From 87c57d95d422fd24eb24d1d769ba9d60b86d8811 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 23:01:07 +0000 Subject: [PATCH 1/7] fix(runtime): resume keeps the run's agent auth config Native continuation set ULTRAFUZZ_CONFIG_PATH to /smithers/resolved-config.json. Every stock agent adapter reads that path with its TOML reader, which finds no [agents.*] tables in JSON and returns an empty config, so each adapter fell back to its defaults on resume: auth, api_key_env and config_dir were silently dropped. With the default ultrafuzz.toml (CodexAgent auth = "api-key"), every resumed Codex task ran with subscription auth and a cleared OPENAI_API_KEY. Launch hands adapters a sealed copy of /smithers/execution-config.toml, the TOML rendering of the same resolved config. Resume now points them at that file. It is written at compile time since #635, and resume only sets the variable when resolved-config.json parses as the current v4 schema (#1120), so every run that reaches this branch was compiled with it. The new regression test resumes a run, rebuilds the generated CodexAgent from the config path the runner received, and checks it still carries the API key. It fails on main (CODEX_API_KEY is empty) and passes with this change. "native resume delegates the persisted workflow after mutable project sources are replaced" asserted the JSON path, i.e. the bug; it now asserts the TOML path. Co-Authored-By: Claude Opus 5.5 --- docs/reference/cli.md | 3 +- packages/runtime/src/start-run.ts | 6 +++- packages/runtime/test/runtime.test.ts | 46 ++++++++++++++++++++++++++- 3 files changed, 52 insertions(+), 3 deletions(-) diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 0e42102b2..68bd3bd33 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -323,7 +323,8 @@ schema bindings, and metadata projections remain provenance for inspection; they are not resume authorization. Smithers decides which finished rows can be reused and which newly rendered or unfinished tasks run. Ultrafuzz does not rewrite historical artifacts or automatically reset, replay, timetravel, or -fork completed work. +fork completed work. Agent adapters in the continued workflow read the same +agent config as at launch, from the run's `smithers/execution-config.toml`. `resume --refresh-controller` first renders the currently installed Ultrafuzz controller and stock adapters beside the historical source, then delegates to diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index ba32e842b..2e8f44da9 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -526,6 +526,10 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { const smithersRoot = safeResolveInside(layout.root, "smithers", "Smithers evidence"); const tasksPath = safeResolveInside(smithersRoot, "tasks.json", "workflow task manifest"); const configPath = safeResolveInside(smithersRoot, "resolved-config.json", "workflow config"); + // Agent adapters parse ULTRAFUZZ_CONFIG_PATH as TOML; given the JSON above + // they find no agent tables and fall back to default auth. Launch writes + // the same config as TOML beside it and hands adapters a copy of that file. + const agentConfigPath = safeResolveInside(smithersRoot, "execution-config.toml", "workflow agent config"); let taskDocument: SmithersTaskManifestDocument | undefined; let config: ResolvedConfig | undefined; if (fs.existsSync(tasksPath)) { @@ -571,7 +575,7 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { ...forgeGuard.env, ULTRAFUZZ_ARTIFACTS_MODULE: import.meta.resolve("@ultrafuzz/artifacts"), ULTRAFUZZ_RUNTIME_MODULE: import.meta.resolve("@ultrafuzz/runtime"), - ...(config === undefined ? {} : { ULTRAFUZZ_CONFIG_PATH: configPath }), + ...(config === undefined ? {} : { ULTRAFUZZ_CONFIG_PATH: agentConfigPath }), ULTRAFUZZ_WORKFLOW_PERSISTED_PATH: workflowPath }; let trustedCli: TrustedCliEnvironment = { diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index 357f3466d..5fd3a3f9f 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -2503,6 +2503,16 @@ function controllerRefreshTerminalEnv( }); } +/** Make a fake lifecycle runner append `$` to `logPath` on every `up`. */ +function logFakeRunnerUpVariable(env: Record, name: string, logPath: string): void { + const shim = env.SMITHERS_BIN; + assert.ok(shim); + fs.writeFileSync( + shim, + fs.readFileSync(shim, "utf8").replace(" up)\n", ` up)\n printf '%s\\n' "$${name}" >> ${shellQuote(logPath)}\n`) + ); +} + function workflowEvents( workflowRunId: string, events: Array<{ @@ -15383,7 +15393,7 @@ test("native resume delegates the persisted workflow after mutable project sourc assert.equal(resumed.ok, true, JSON.stringify(resumed.diagnostics)); const consumed = fs.readFileSync(snapshotBytesLog, "utf8"); assert.equal(consumed.includes(`workflow=${mutableWorkflow}\n`), true); - assert.match(consumed, /^config=.*\/smithers\/resolved-config\.json$/mu); + assert.match(consumed, /^config=.*\/smithers\/execution-config\.toml$/mu); assert.match(consumed, /^agent=.*\/\.smithers\/agents\/codex\.ts$/mu); assert.match(consumed, /HostileReplacement/u); assert.match(consumed, /export const hostile/u); @@ -25060,6 +25070,40 @@ test("native continuation does not use historical trusted CLI identity as an aut ); }); +test("native continuation hands generated agents the run's TOML config, so CodexAgent keeps API-key auth", async () => { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + writeSmallTopology(project); + const runId = "continuation-agent-config"; + const env = controllerRefreshTerminalEnv(project, runId); + const configPathLog = path.join(project, "fake-smithers-up-config-path.log"); + logFakeRunnerUpVariable(env, "ULTRAFUZZ_CONFIG_PATH", configPathLog); + const launched = await startRun({ projectRoot: project, runId, env }); + assert.equal(launched.ok, true, JSON.stringify(launched.diagnostics)); + fs.writeFileSync(configPathLog, "", "utf8"); + const resumed = await resumeRun({ projectRoot: project, runId, env }); + assert.equal(resumed.ok, true, JSON.stringify(resumed.diagnostics)); + assert.equal(resumed.value?.submitted, true); + + // Build the stock adapter from the config path the resumed runner received. + // The init config selects `auth = "api-key"` for CodexAgent; an adapter that + // cannot read it falls back to subscription auth and clears the key. + const { createCodexAgent } = await loadGeneratedCodexAgent(project); + const previous = { config: process.env.ULTRAFUZZ_CONFIG_PATH, key: process.env.OPENAI_API_KEY }; + process.env.ULTRAFUZZ_CONFIG_PATH = fs.readFileSync(configPathLog, "utf8").trim(); + process.env.OPENAI_API_KEY = "continuation-codex-key"; + try { + const agent = createCodexAgent() as { opts: { env: Record } }; + assert.equal(agent.opts.env.CODEX_API_KEY, "continuation-codex-key"); + assert.equal(agent.opts.env.OPENAI_API_KEY, "continuation-codex-key"); + } finally { + if (previous.config === undefined) delete process.env.ULTRAFUZZ_CONFIG_PATH; + else process.env.ULTRAFUZZ_CONFIG_PATH = previous.config; + if (previous.key === undefined) delete process.env.OPENAI_API_KEY; + else process.env.OPENAI_API_KEY = previous.key; + } +}); + test("controller refresh authenticates newly required sealed runner patches and rejects source drift", async () => { const project = tempProject(); initProject({ projectRoot: project, force: true }); From a037bc9f4f5318e7b70d06e6c6a71e3bbebd1433 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 23:01:27 +0000 Subject: [PATCH 2/7] fix(runtime): a no-op resume attach leaves run state alone and says so `resume` treats a run whose derived Smithers state is still active as an idempotent attach and starts no controller. That includes a run still draining a pause and, for up to the 30 s heartbeat window, a run whose controller just died. submitSmithersContinuation nevertheless called recordNativeContinuationState, which rewrote state.json and moved workflow_deadline_at forward on every such no-op, and the CLI printed "Submitted resume" regardless of `submitted`. Only record continuation state when a controller was actually started, and drop the now-unused alreadyRunning parameter and branch. The CLI now prints "Run already active: ; no new controller was started" when nothing was submitted, with the hint to resume again once a draining pause has parked. JSON output is unchanged. #1153 asked for an Ultrafuzz-side controller-generation fence. The pinned engine already provides the guarantee, so this pins it instead of re-implementing it. A real-Smithers integration test shows that: - a graceful pause lets held in-flight tasks finish; - while the owner drains, `up --resume --force` and `timetravel --force` are refused with RUN_OWNER_ALIVE, and inspect still reports the run state as `running`, so `ultrafuzz resume` only attaches; - a resume after the park runs only the remaining task, in a new controller. Swapping --force for --steal-ownership in that test makes it fail, so it detects a takeover. The state test fails on main (workflow_deadline_at moves on a submitted:false resume) and the CLI test fails on main ("Submitted resume: ..."); both pass with this change. The pin test passes on both by design. Co-Authored-By: Claude Opus 5.5 --- docs/reference/cli.md | 17 +- docs/reference/configuration.md | 3 +- packages/cli/src/commands/resume.ts | 6 +- packages/cli/test/cli.test.ts | 20 +++ packages/runtime/src/start-run.ts | 36 ++--- packages/runtime/test/runtime.test.ts | 6 + ...thers-preparation-race.integration.test.ts | 148 +++++++++++++++++- 7 files changed, 204 insertions(+), 32 deletions(-) diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 68bd3bd33..a9ac9e2dc 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -459,13 +459,16 @@ an available agent-written report. `pause` requests a graceful stop: no new tasks are scheduled, in-flight tasks finish, and the run settles in the resumable `paused` state. `resume` reports -`submitted: false` instead of launching a duplicate continuation when the linked -workflow is still in an active state (running, in-progress, started, queued, -retrying, or waiting). `resume --reset-node` retries one failed workflow node and -its dependents in the same linked run; the applied reset is recorded so retrying -the command after a failed continuation resumes the already-reset run instead of -repeating the reset. `fork` may start from a checkpoint frame and may reset one -workflow node before starting the fork. +`submitted: false` (text output `Run already active`) instead of launching a +duplicate continuation when the linked workflow is still active (its Smithers +run state is `running`, `recovering`, or one of the `waiting-*` states), and +leaves the run's recorded state and workflow deadline unchanged. A run still +finishing its in-flight tasks after `pause` is still active; resume it again +once `status` reports `paused`. `resume --reset-node` retries one failed +workflow node and its dependents in the same linked run; the applied reset is +recorded so retrying the command after a failed continuation resumes the +already-reset run instead of repeating the reset. `fork` may start from a +checkpoint frame and may reset one workflow node before starting the fork. Every command in this section takes an Ultrafuzz run ID and resolves the linked workflow run from existing product evidence; none of them require the diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 1095c7155..1f5f87da9 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -139,7 +139,8 @@ resume. `workflow_deadline_seconds` is not a guaranteed wall-clock limit. Ultrafuzz records `workflow_deadline_at` in run state when the run is created, and again -from each `resume`, `replay`, or `fork`, but nothing enforces it on a timer: an +from each `replay`, `fork`, or `resume` that starts a controller (not one that +finds the run still active), but nothing enforces it on a timer: an unattended run keeps executing, and incurring provider cost, past its deadline. The deadline is checked only when a command synchronizes the run: `ultrafuzz status` (including each `--watch` poll), `inspect`, `why`, and `stats`, plus diff --git a/packages/cli/src/commands/resume.ts b/packages/cli/src/commands/resume.ts index 2f3bb79da..e172870b7 100644 --- a/packages/cli/src/commands/resume.ts +++ b/packages/cli/src/commands/resume.ts @@ -40,7 +40,11 @@ export default class Resume extends Command { emitCommandResult( this, "resume", - commandFromRuntime("resume", result, (value) => `Submitted ${value.action}: ${value.workflow_run_id}\n`), + commandFromRuntime("resume", result, (value) => + value.submitted + ? `Submitted ${value.action}: ${value.workflow_run_id}\n` + : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused.\n` + ), flags.json === true ); } diff --git a/packages/cli/test/cli.test.ts b/packages/cli/test/cli.test.ts index 18676ca3a..750066e23 100644 --- a/packages/cli/test/cli.test.ts +++ b/packages/cli/test/cli.test.ts @@ -1918,6 +1918,26 @@ test("status surfaces a terminal product and live workflow lifecycle divergence" ); }); +test("resume of an already-active run says no controller was started instead of claiming a submission", async () => { + const project = tempProject(); + const env = fakeSmithersEnv(project); + assert.equal((await cli(project, ["init", "--json"], env)).code, 0); + writeSmallTopology(project); + const runId = "resume-already-active"; + const run = await cli(project, ["run", "--run-id", runId, "--json"], env); + assert.equal(run.code, 0, run.stderr); + + // The fake runner still reports the run as running, so resume only attaches. + const resumed = await cli(project, ["resume", runId], env); + + assert.equal(resumed.code, 0, `${resumed.stderr}\n${resumed.stdout}`); + assert.match( + resumed.stdout, + /^Run already active: ultrafuzz-resume-already-active; no new controller was started\./mu + ); + assert.doesNotMatch(resumed.stdout, /Submitted/u); +}); + test("status --watch stops immediately on a degraded verdict even while product state is nonterminal", async () => { const project = tempProject(); const env = fakeSmithersEnv(project); diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index 2e8f44da9..9d92cfcc5 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -672,12 +672,11 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { trustedCli.environmentVariableNames ) }); - recordNativeContinuationState({ - layout, - config, - requestedConcurrency: input.maxConcurrency, - alreadyRunning: result.alreadyRunning ?? false - }); + // Attaching to an already-active run started no controller. Its state + // (status, lease, deadline) belongs to the live owner, so leave it alone. + if (result.alreadyRunning !== true) { + recordNativeContinuationState({ layout, config, requestedConcurrency: input.maxConcurrency }); + } return runtimeResult(true, { run_id: runId, workflow_run_id: smithersRunId, @@ -743,7 +742,6 @@ function recordNativeContinuationState(input: { layout: RunLayout; config: ResolvedConfig | undefined; requestedConcurrency: number | undefined; - alreadyRunning: boolean; }): void { try { const submittedAt = new Date().toISOString(); @@ -753,19 +751,17 @@ function recordNativeContinuationState(input: { state.status = "running"; state.started_at ??= submittedAt; delete state.finished_at; - if (!input.alreadyRunning) { - const leaseDurationMs = - (input.config?.run.controllerLeaseSeconds ?? Math.max(1, state.controller_lease.duration_ms / 1_000)) * 1_000; - state.controller_lease = { - ...state.controller_lease, - status: "active", - duration_ms: leaseDurationMs, - renewed_at: submittedAt, - expires_at: new Date(submittedAtMs + leaseDurationMs).toISOString() - }; - state.concurrency.requested_concurrency = - input.requestedConcurrency ?? input.config?.run.maxParallelAgents ?? state.concurrency.requested_concurrency; - } + const leaseDurationMs = + (input.config?.run.controllerLeaseSeconds ?? Math.max(1, state.controller_lease.duration_ms / 1_000)) * 1_000; + state.controller_lease = { + ...state.controller_lease, + status: "active", + duration_ms: leaseDurationMs, + renewed_at: submittedAt, + expires_at: new Date(submittedAtMs + leaseDurationMs).toISOString() + }; + state.concurrency.requested_concurrency = + input.requestedConcurrency ?? input.config?.run.maxParallelAgents ?? state.concurrency.requested_concurrency; if (input.config !== undefined) { state.workflow_deadline_at = new Date( submittedAtMs + input.config.run.workflowDeadlineSeconds * 1_000 diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index 5fd3a3f9f..2394621f2 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -26463,6 +26463,11 @@ test("ordinary resume checks active-run ownership before detached preflight", as }); const run = await startRun({ projectRoot: project, runId: "active-lifecycle-run", env }); assert.equal(run.ok, true, JSON.stringify(run.diagnostics)); + // An attach starts no controller, so the run's state (status, lease and + // workflow deadline) must stay exactly as its live owner left it (#1153). + assert.ok(run.value); + const statePath = path.join(run.value.run_root, "state.json"); + const stateBefore = fs.readFileSync(statePath, "utf8"); fs.writeFileSync(env.SMITHERS_FAKE_LOG!, "", "utf8"); // A duplicate `up --resume --detach` renders the workflow before Smithers // checks ownership. Keep that path fatal so this regression proves active @@ -26492,6 +26497,7 @@ test("ordinary resume checks active-run ownership before detached preflight", as const forcedCommands = fs.readFileSync(env.SMITHERS_FAKE_LOG!, "utf8"); assert.match(forcedCommands, /inspect ultrafuzz-active-lifecycle-run --format json --full-output/u); assert.doesNotMatch(forcedCommands, /^up /mu); + assert.equal(fs.readFileSync(statePath, "utf8"), stateBefore); }); test("resume derives reset identities from the canonical nodes of a failed workflow", async () => { diff --git a/packages/runtime/test/smithers-preparation-race.integration.test.ts b/packages/runtime/test/smithers-preparation-race.integration.test.ts index 204af09e9..ad778d442 100644 --- a/packages/runtime/test/smithers-preparation-race.integration.test.ts +++ b/packages/runtime/test/smithers-preparation-race.integration.test.ts @@ -1,6 +1,6 @@ import assert from "node:assert/strict"; import { temporaryRoot } from "./temporary-root.js"; -import { execFileSync } from "node:child_process"; +import { execFileSync, spawnSync } from "node:child_process"; import fs from "node:fs"; import path from "node:path"; import test from "node:test"; @@ -226,6 +226,96 @@ process.stdout.write(JSON.stringify(result));` } }); +// #1153 asked Ultrafuzz for its own controller-generation fence around pause +// and continuation handoff. The pinned engine already provides the guarantee: +// a graceful pause lets in-flight tasks finish instead of aborting them, a +// second controller is refused while the owner is alive (`--force` does not +// take ownership), and a resume after the park runs only the remaining work. +// Pin those facts so an engine bump that regresses them fails here. +test("graceful pause drains in-flight tasks and refuses a second controller until the run parks", async () => { + const root = temporaryRoot("ultrafuzz-smithers-pause-handoff-"); + const workflowDir = path.join(root, ".smithers", "workflows"); + const workflowPath = path.join(workflowDir, "pause-handoff.tsx"); + const traceLog = path.join(root, "trace.log"); + const releasePath = path.join(root, "release-in-flight"); + const runId = `pause-handoff-${process.pid}-${Date.now()}`; + const runner = (args: string[]) => + spawnSync(smithersBinary(), args, { + cwd: root, + encoding: "utf8", + env: { ...process.env, SMITHERS_POST_FAILURE: "0" } + }); + const trace = (): string[] => + fs.existsSync(traceLog) ? fs.readFileSync(traceLog, "utf8").trim().split("\n").filter(Boolean) : []; + const pidOf = (lines: readonly string[], event: string) => + lines.find((line) => line.endsWith(` ${event}`))?.split(" ")[0]; + + try { + fs.mkdirSync(workflowDir, { recursive: true }); + initFixtureRepository(root); + const smithersPackageRoot = fs.realpathSync(path.join(runtimePackageRoot(), "node_modules", "smthrs")); + fs.symlinkSync(path.dirname(smithersPackageRoot), path.join(root, ".smithers", "node_modules"), "dir"); + fs.writeFileSync(workflowPath, pauseHandoffWorkflowSource({ traceLog, releasePath }), "utf8"); + + const launched = runner([ + "up", + workflowPath, + "--detach", + "--run-id", + runId, + "--root", + root, + "--input", + "{}", + "--format", + "json" + ]); + assert.equal(launched.status, 0, launched.stderr); + await waitUntil(() => trace().filter((line) => line.endsWith(" start")).length === 2, 60_000, "a and b start"); + + const pause = runner(["pause", runId, "--format", "json"]); + assert.match(pause.stdout, /"pause-requested"/u, pause.stderr); + // Both in-flight tasks are held open, so the owner is still draining: a + // replacement controller and a node reset are refused despite `--force`. + for (const takeover of [ + ["up", workflowPath, "--resume", runId, "--run-id", runId, "--force", "--detach", "--format", "json"], + ["timetravel", workflowPath, "--run-id", runId, "--node-id", "a", "--no-vcs", "--force", "--format", "json"] + ]) { + const refused = runner(takeover); + assert.notEqual(refused.status, 0, takeover.join(" ")); + assert.match(refused.stdout + refused.stderr, /RUN_OWNER_ALIVE/u, takeover.join(" ")); + } + // The draining run still reports an active state, so `ultrafuzz resume` + // only attaches to it instead of starting a controller. + const draining = JSON.parse(runner(["inspect", runId, "--format", "json", "--full-output"]).stdout) as { + data?: { runState?: { state?: string } }; + }; + assert.equal(draining.data?.runState?.state, "running"); + assert.equal(trace().length, 2, "no task ended or started while the pause drained"); + + fs.writeFileSync(releasePath, "", "utf8"); + await waitForStatus(root, runId, "paused", 60_000); + const parked = trace(); + const ownerPid = pidOf(parked, "a start"); + assert.equal(pidOf(parked, "a end"), ownerPid, "the draining owner finished a"); + assert.equal(pidOf(parked, "b end"), ownerPid, "the draining owner finished b"); + assert.equal(pidOf(parked, "c start"), undefined, "the pause stopped new scheduling"); + + const resumed = runner(["up", workflowPath, "--resume", runId, "--run-id", runId, "--detach", "--format", "json"]); + assert.equal(resumed.status, 0, resumed.stderr); + await waitForSuccessfulCompletion(root, runId, 60_000); + const finished = trace(); + for (const event of ["a start", "b start", "c start"]) { + const runs = finished.filter((line) => line.endsWith(` ${event}`)).length; + assert.equal(runs, 1, `${event} ran ${runs} times: ${JSON.stringify(finished)}`); + } + assert.notEqual(pidOf(finished, "c start"), ownerPid, "only the replacement controller ran c"); + } finally { + fs.writeFileSync(releasePath, "", "utf8"); + fs.rmSync(root, { recursive: true, force: true }); + } +}); + function initFixtureRepository(root: string): void { execGit(root, ["init", "--quiet", "--initial-branch=main"]); execGit(root, ["config", "user.name", "Ultrafuzz Synthetic Test"]); @@ -368,6 +458,45 @@ export default smithers((ctx) => ( `; } +function pauseHandoffWorkflowSource(input: { traceLog: string; releasePath: string }): string { + return `/** @jsxImportSource smthrs */ +import fs from "node:fs"; +import { createSmithers } from "smthrs"; +import { z } from "zod/v4"; + +const traceLog = ${JSON.stringify(input.traceLog)}; +const releasePath = ${JSON.stringify(input.releasePath)}; +const trace = (event) => fs.appendFileSync(traceLog, process.pid + " " + event + "\\n", "utf8"); +const { Workflow, Task, Parallel, Sequence, smithers, outputs } = createSmithers({ + input: z.object({}), + step: z.object({ done: z.literal(true) }) +}); +// Hold each in-flight task until the test releases it, bounded so a failed +// test cannot leave the detached engine polling forever. +const held = (id) => async () => { + trace(id + " start"); + const deadline = Date.now() + 120000; + while (!fs.existsSync(releasePath) && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 50)); + } + trace(id + " end"); + return { done: true }; +}; + +export default smithers(() => ( + + + + {held("a")} + {held("b")} + + {() => (trace("c start"), { done: true })} + + +)); +`; +} + async function waitForFile(filePath: string, timeoutMs: number): Promise { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { @@ -377,7 +506,20 @@ async function waitForFile(filePath: string, timeoutMs: number): Promise { throw new Error(`timed out waiting for ${filePath}`); } +async function waitUntil(condition: () => boolean, timeoutMs: number, label: string): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (condition()) return; + await new Promise((resolve) => setTimeout(resolve, 100)); + } + throw new Error(`timed out waiting until ${label}`); +} + async function waitForSuccessfulCompletion(root: string, runId: string, timeoutMs: number): Promise { + await waitForStatus(root, runId, "finished", timeoutMs); +} + +async function waitForStatus(root: string, runId: string, expected: string, timeoutMs: number): Promise { const deadline = Date.now() + timeoutMs; let status = "unknown"; while (Date.now() < deadline) { @@ -393,13 +535,13 @@ async function waitForSuccessfulCompletion(root: string, runId: string, timeoutM await new Promise((resolve) => setTimeout(resolve, 100)); continue; } - if (status === "finished") return; + if (status === expected) return; if (["failed", "cancelled", "canceled"].includes(status)) { throw new Error(`synthetic Smithers workflow ended with status ${status}`); } await new Promise((resolve) => setTimeout(resolve, 100)); } - throw new Error(`synthetic Smithers workflow did not finish; final status ${status}`); + throw new Error(`synthetic Smithers workflow did not reach ${expected}; final status ${status}`); } function smithersNodeOutput(root: string, runId: string, nodeId: string): Buffer { From 8298212f4ccebeefe176c1be5bdf2319efddaf40 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 23:01:44 +0000 Subject: [PATCH 3/7] fix(runtime): resume warns instead of dropping the trusted launcher or failing on worktree cleanup Two resume-time preparation steps were wrong in opposite directions. When prepareTrustedCliEnvironment threw on resume, for example because the run was launched under a different Node binary and the launcher bytes no longer match a fresh render, the error was swallowed and the trusted CLI variables were cleared. composeSmithersCommandPath then left /trusted-bin off the engine PATH, so every task's validator preflight ran whatever `ultrafuzz` the operator's PATH held (or hit ENOENT). Keep ULTRAFUZZ_TRUSTED_BIN pointed at the run-owned launcher when it exists: the launcher re-verifies its closure on every dispatch. The failure is now reported as a WORKFLOW_TRUSTED_CLI_UNVERIFIED warning. repairPrunableRunWorktreeRegistrations is cleanup, but a failed `git worktree remove` (or an unreadable workspace path) threw out of resume and failed the whole continuation. It now yields a WORKFLOW_WORKTREE_REPAIR_FAILED warning and the continuation proceeds. Resume returns these warnings as diagnostics (also appended after the error on a failed resume), and the CLI prints them in text mode. Both tests fail on main (the runner PATH starts with the caller's PATH instead of trusted-bin; the resume fails with WORKFLOW_LIFECYCLE_FAILED) and pass with this change. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + docs/reference/cli.md | 4 +- docs/schemas.md | 8 ++-- packages/cli/src/commands/resume.ts | 18 ++++---- packages/runtime/src/start-run.ts | 56 +++++++++++++++++++---- packages/runtime/test/runtime.test.ts | 66 +++++++++++++++++++++++++++ 6 files changed, 130 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 960ba9394..66f2072d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,7 @@ ### Other changes +- **[runtime] [cli] [docs]** `resume` now points agent adapters at the run's TOML agent config (`smithers/execution-config.toml`) instead of `resolved-config.json`, which their TOML reader parsed as empty, so a resumed run keeps each agent's `auth`, `api_key_env`, and `config_dir` (the default `CodexAgent` API-key auth had fallen back to subscription). A resume that only attaches to an already-active run no longer rewrites `state.json` or pushes back `workflow_deadline_at`, and prints `Run already active` instead of `Submitted resume`. A failed stale task-worktree cleanup no longer fails the resume and a failed trusted-CLI re-verification is no longer silent: both are now warnings, and in the latter case the run-owned launcher stays first on `PATH` instead of leaving tasks to whatever `ultrafuzz` the operator's `PATH` holds (#1153, #1143, #1110). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955). diff --git a/docs/reference/cli.md b/docs/reference/cli.md index a9ac9e2dc..3c16047e1 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -324,7 +324,9 @@ they are not resume authorization. Smithers decides which finished rows can be reused and which newly rendered or unfinished tasks run. Ultrafuzz does not rewrite historical artifacts or automatically reset, replay, timetravel, or fork completed work. Agent adapters in the continued workflow read the same -agent config as at launch, from the run's `smithers/execution-config.toml`. +agent config as at launch, from the run's `smithers/execution-config.toml`. If +resume cannot prune stale task-worktree registrations, it reports a +`WORKFLOW_WORKTREE_REPAIR_FAILED` warning and continues. `resume --refresh-controller` first renders the currently installed Ultrafuzz controller and stock adapters beside the historical source, then delegates to diff --git a/docs/schemas.md b/docs/schemas.md index c8e8caedc..da37ec8e9 100644 --- a/docs/schemas.md +++ b/docs/schemas.md @@ -84,9 +84,11 @@ a working-tree rebuild cannot change an active run. Ambient Node loader/search variables are removed and both ESM and CommonJS module resolution must stay inside that snapshot; document reads are unaffected. A path lookup alone is not a preflight. Ordinary resume now delegates continuation to Smithers instead of -using the historical launcher or closure as an authorization gate. A current -controller refresh publishes a new controller path without rewriting the -historical closure. +using the historical launcher or closure as an authorization gate. When resume +cannot re-verify the launcher, it reports a `WORKFLOW_TRUSTED_CLI_UNVERIFIED` +warning and keeps the run-owned launcher first on `PATH`; that launcher still +verifies its closure before every dispatch. A current controller refresh +publishes a new controller path without rewriting the historical closure. Exit `0` establishes portable document-shape conformance only. Cross-file joins, projected-key uniqueness, filesystem and Git facts, digest relationships, diff --git a/packages/cli/src/commands/resume.ts b/packages/cli/src/commands/resume.ts index e172870b7..09f2fc96d 100644 --- a/packages/cli/src/commands/resume.ts +++ b/packages/cli/src/commands/resume.ts @@ -5,6 +5,7 @@ import { cliEntrypoint, cliIo, commandFromRuntime, + diagnosticsText, emitCommandResult, globalFlags, projectRoot @@ -37,15 +38,14 @@ export default class Resume extends Command { resetNode: flags["reset-node"], env: cliIo().env }); - emitCommandResult( - this, - "resume", - commandFromRuntime("resume", result, (value) => - value.submitted - ? `Submitted ${value.action}: ${value.workflow_run_id}\n` - : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused.\n` - ), - flags.json === true + const commandResult = commandFromRuntime("resume", result, (value) => + value.submitted + ? `Submitted ${value.action}: ${value.workflow_run_id}\n` + : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused.\n` ); + if (result.ok && result.diagnostics.length > 0) { + commandResult.text = `${commandResult.text ?? ""}${diagnosticsText(result.diagnostics)}`; + } + emitCommandResult(this, "resume", commandResult, flags.json); } } diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index 9d92cfcc5..eb03c2139 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -54,6 +54,7 @@ import { prepareTrustedCliEnvironment, runTrustedJsonValidatorPreflight, TRUSTED_CLI_ENVIRONMENT_VARIABLES, + ULTRAFUZZ_TRUSTED_BIN_ENV, type TrustedCliEnvironment } from "./trusted-cli.js"; import { hasRuntimeErrors, runtimeFailure, runtimeResult } from "./utils.js"; @@ -478,6 +479,7 @@ export async function resumeRun(input: WorkflowLifecycleInput) { async function submitSmithersContinuation(input: WorkflowLifecycleInput) { let releaseLifecycleLock: (() => Promise) | undefined; + const diagnostics: RuntimeDiagnostic[] = []; try { const projectRoot = path.resolve(input.projectRoot); const runsRoot = await runsRootForProject(projectRoot); @@ -597,9 +599,23 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { }); if (prepared.active) runTrustedJsonValidatorPreflight({ layout, trusted: prepared }); trustedCli = prepared; - } catch { + } catch (error) { // Historical validator identity is task setup provenance, not authority - // to prevent Smithers from continuing the workflow. + // to prevent Smithers from continuing the workflow. Keep the run-owned + // launcher first on PATH anyway: it re-verifies its closure on every + // call, while dropping it lets tasks run whatever `ultrafuzz` is on PATH. + const trustedBin = path.join(layout.root, "trusted-bin"); + const launcherKept = fs.existsSync( + path.join(trustedBin, process.platform === "win32" ? "ultrafuzz.cmd" : "ultrafuzz") + ); + if (launcherKept) trustedCli.env[ULTRAFUZZ_TRUSTED_BIN_ENV] = trustedBin; + diagnostics.push( + resumeWarning( + "WORKFLOW_TRUSTED_CLI_UNVERIFIED", + `resume could not re-verify the run's trusted Ultrafuzz CLI${launcherKept ? "; tasks keep calling the run-owned launcher, which checks itself on every call" : ""}`, + error + ) + ); } } const agentRefs = tasks.flatMap((task) => task.agentChain.map((profile) => profile.agentRef)); @@ -615,7 +631,15 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { assertCurrentCloudAgentCredentialEnvironment(config, tasks, lifecycleEnvironment); } if (typeof metadata.source_revision === "string") { - repairPrunableRunWorktreeRegistrations({ projectRoot, runRoot: layout.root, runId }); + try { + repairPrunableRunWorktreeRegistrations({ projectRoot, runRoot: layout.root, runId }); + } catch (error) { + // Pruning is cleanup: a stale registration it leaves behind surfaces + // when Smithers recreates that task's worktree, so do not stop here. + diagnostics.push( + resumeWarning("WORKFLOW_WORKTREE_REPAIR_FAILED", "resume could not prune stale task worktrees", error) + ); + } } const result = await runSmithersLifecycleCommand({ action: "resume", @@ -677,14 +701,21 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { if (result.alreadyRunning !== true) { recordNativeContinuationState({ layout, config, requestedConcurrency: input.maxConcurrency }); } - return runtimeResult(true, { - run_id: runId, - workflow_run_id: smithersRunId, - action: "resume" as const, - submitted: !result.alreadyRunning - }); + return runtimeResult( + true, + { + run_id: runId, + workflow_run_id: smithersRunId, + action: "resume" as const, + submitted: !result.alreadyRunning + }, + diagnostics + ); } catch (error) { - return runtimeFailure([smithersDiagnostic(error, "WORKFLOW_LIFECYCLE_FAILED")]); + return runtimeFailure([ + smithersDiagnostic(error, "WORKFLOW_LIFECYCLE_FAILED"), + ...diagnostics + ]); } finally { await releaseLifecycleLock?.(); } @@ -779,6 +810,11 @@ function recordNativeContinuationState(input: { } } +function resumeWarning(code: string, context: string, error: unknown): RuntimeDiagnostic { + const diagnostic = smithersDiagnostic(error, code); + return { ...diagnostic, message: `${context}: ${diagnostic.message}`, severity: "warning", source: "runtime" }; +} + export async function replayRun(input: WorkflowLifecycleInput) { return submitLifecycleAction(input, "replay"); } diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index 2394621f2..ca30a97d7 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -25046,11 +25046,14 @@ test("native continuation does not use historical trusted CLI identity as an aut writeSmallTopology(project); const runId = "controller-refresh-trusted-cli"; const env = controllerRefreshTerminalEnv(project, runId); + const upPathLog = path.join(project, "fake-smithers-up-path.log"); + logFakeRunnerUpVariable(env, "PATH", upPathLog); const launched = await startRun({ projectRoot: project, runId, env }); assert.equal(launched.ok, true, JSON.stringify(launched.diagnostics)); const trustedMetadataPath = path.join(launched.value!.run_root, "trusted-cli.json"); fs.writeFileSync(trustedMetadataPath, "{}\n", "utf8"); fs.writeFileSync(env.SMITHERS_FAKE_LOG!, "", "utf8"); + fs.writeFileSync(upPathLog, "", "utf8"); const ordinary = await resumeRun({ projectRoot: project, runId, env }); assert.equal(ordinary.ok, true, JSON.stringify(ordinary.diagnostics)); @@ -25060,6 +25063,23 @@ test("native continuation does not use historical trusted CLI identity as an aut assert.equal(refreshed.ok, true, JSON.stringify(refreshed.diagnostics)); assert.equal(refreshed.value?.submitted, true); assert.equal(fs.readFileSync(trustedMetadataPath, "utf8"), "{}\n"); + // The run-owned launcher, which re-verifies itself on every call, stays + // first on the runner's PATH instead of leaving tasks to whatever + // `ultrafuzz` the operator's PATH holds, and the failure is reported (#1143). + const upPaths = fs.readFileSync(upPathLog, "utf8").trim().split("\n"); + assert.equal(upPaths.length, 2); + for (const upPath of upPaths) { + assert.equal(upPath.split(path.delimiter)[0], path.join(path.dirname(trustedMetadataPath), "trusted-bin")); + } + for (const resumed of [ordinary, refreshed]) { + assert.equal( + resumed.diagnostics.some( + (diagnostic) => diagnostic.code === "WORKFLOW_TRUSTED_CLI_UNVERIFIED" && diagnostic.severity === "warning" + ), + true, + JSON.stringify(resumed.diagnostics) + ); + } assert.equal( fs .readFileSync(env.SMITHERS_FAKE_LOG!, "utf8") @@ -26500,6 +26520,52 @@ test("ordinary resume checks active-run ownership before detached preflight", as assert.equal(fs.readFileSync(statePath, "utf8"), stateBefore); }); +test("resume continues with a warning when stale task-worktree cleanup fails", async () => { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + writeSmallTopology(project); + const runId = "worktree-cleanup-failure"; + const env = controllerRefreshTerminalEnv(project, runId); + const commandLog = env.SMITHERS_FAKE_LOG; + assert.ok(commandLog); + const launched = await startRun({ projectRoot: project, runId, env }); + assert.equal(launched.ok, true, JSON.stringify(launched.diagnostics)); + assert.ok(launched.value); + const runRoot = launched.value.run_root; + // Cleanup runs only for runs launched from a Git revision. + const metadataPath = path.join(runRoot, "run.json"); + const metadata = JSON.parse(fs.readFileSync(metadataPath, "utf8")) as Record; + fs.writeFileSync(metadataPath, `${JSON.stringify({ ...metadata, source_revision: "0".repeat(40) }, null, 2)}\n`); + // Git lists a prunable registration owned by this run, then cannot remove it. + const previousPath = process.env.PATH ?? ""; + const gitBin = temporaryRoot("ufz-failing-git-"); + fs.writeFileSync( + path.join(gitBin, "git"), + [ + "#!/bin/sh", + 'case "$*" in', + ` *"worktree list --porcelain"*) printf 'worktree %s\\nbranch refs/heads/ultrafuzz/%s/stale\\nprunable\\n\\n' ${shellQuote(path.join(runRoot, "workspaces", "stale"))} ${runId} ;;`, + ' *"worktree remove"*) echo "fatal: synthetic removal failure" >&2; exit 1 ;;', + ` *) PATH=${shellQuote(previousPath)} exec git "$@" ;;`, + "esac", + "" + ].join("\n"), + { mode: 0o755 } + ); + process.env.PATH = [gitBin, previousPath].join(path.delimiter); + const resumed = await resumeRun({ projectRoot: project, runId, env }).finally(() => { + process.env.PATH = previousPath; + }); + + assert.equal(resumed.ok, true, JSON.stringify(resumed.diagnostics)); + assert.equal(resumed.value?.submitted, true); + const warning = resumed.diagnostics.find((diagnostic) => diagnostic.code === "WORKFLOW_WORKTREE_REPAIR_FAILED"); + assert.ok(warning, JSON.stringify(resumed.diagnostics)); + assert.equal(warning.severity, "warning"); + assert.match(warning.message, /synthetic removal failure/u); + assert.match(fs.readFileSync(commandLog, "utf8"), /^up .*--resume ultrafuzz-worktree-cleanup-failure/mu); +}); + test("resume derives reset identities from the canonical nodes of a failed workflow", async () => { const project = tempProject(); initProject({ projectRoot: project, force: true }); From 2143c44c93f8703ef0b17c204180d24d2330c0d9 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 03:21:46 +0000 Subject: [PATCH 4/7] fix(runtime): say what a kept trusted launcher can and cannot do on resume When resume cannot re-verify the trusted CLI it keeps /trusted-bin first on PATH. The WORKFLOW_TRUSTED_CLI_UNVERIFIED warning described that launcher as a working fallback ("checks itself on every call"). That is only true when the launcher itself is intact. With damaged metadata, a changed closure, or a removed Node binary, the launcher refuses or fails to exec. Then every task's preflight-json-validator step fails while resume reports ok. The warning now says that tasks still call the named launcher and that their validator preflight fails while it cannot verify itself. When no launcher exists, it says tasks use whatever `ultrafuzz` is on PATH. I checked the damaged case by running the launcher after the existing test corrupts trusted-cli.json and resumes twice. It exits 1, both resumes return ok, and each carries the new warning. The existing #1143 test drove the catch branch only through that corrupted metadata. In that state keeping the launcher cannot help, so its PATH assertion checked the mechanism, not the outcome. It keeps only the warning assertion. A new test covers the case the change exists for: resume without a CLI entrypoint, so preparation throws while the launcher is intact. It resolves `ultrafuzz` through the PATH the runner received, and that must be the run's launcher. It then runs the task's `json validate` call through it and parses the success envelope. On main the lookup finds no `ultrafuzz` at all (undefined instead of the run's trusted-bin/ultrafuzz); with this change it passes. docs/schemas.md now states the existence condition and the failure modes, and that --refresh-controller does not repair such a launcher. Co-Authored-By: Claude Opus 5.5 --- docs/schemas.md | 10 +++-- packages/runtime/src/start-run.ts | 14 +++--- packages/runtime/test/runtime.test.ts | 63 ++++++++++++++++++++++----- 3 files changed, 68 insertions(+), 19 deletions(-) diff --git a/docs/schemas.md b/docs/schemas.md index da37ec8e9..6aa919644 100644 --- a/docs/schemas.md +++ b/docs/schemas.md @@ -86,9 +86,13 @@ inside that snapshot; document reads are unaffected. A path lookup alone is not a preflight. Ordinary resume now delegates continuation to Smithers instead of using the historical launcher or closure as an authorization gate. When resume cannot re-verify the launcher, it reports a `WORKFLOW_TRUSTED_CLI_UNVERIFIED` -warning and keeps the run-owned launcher first on `PATH`; that launcher still -verifies its closure before every dispatch. A current controller refresh -publishes a new controller path without rewriting the historical closure. +warning and, if `/trusted-bin/ultrafuzz` exists, keeps it first on `PATH` +rather than letting tasks reach another `ultrafuzz`. That launcher still +verifies its closure before every dispatch, so if its metadata or closure is +damaged, or the Node binary it names is gone, each task's validator preflight +fails; `resume --refresh-controller` does not repair such a launcher. A current +controller refresh publishes a new controller path without rewriting the +historical closure. Exit `0` establishes portable document-shape conformance only. Cross-file joins, projected-key uniqueness, filesystem and Git facts, digest relationships, diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index eb03c2139..c380631c9 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -604,15 +604,19 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { // to prevent Smithers from continuing the workflow. Keep the run-owned // launcher first on PATH anyway: it re-verifies its closure on every // call, while dropping it lets tasks run whatever `ultrafuzz` is on PATH. - const trustedBin = path.join(layout.root, "trusted-bin"); - const launcherKept = fs.existsSync( - path.join(trustedBin, process.platform === "win32" ? "ultrafuzz.cmd" : "ultrafuzz") + const launcher = path.join( + layout.root, + "trusted-bin", + process.platform === "win32" ? "ultrafuzz.cmd" : "ultrafuzz" ); - if (launcherKept) trustedCli.env[ULTRAFUZZ_TRUSTED_BIN_ENV] = trustedBin; + const launcherKept = fs.existsSync(launcher); + if (launcherKept) trustedCli.env[ULTRAFUZZ_TRUSTED_BIN_ENV] = path.dirname(launcher); diagnostics.push( resumeWarning( "WORKFLOW_TRUSTED_CLI_UNVERIFIED", - `resume could not re-verify the run's trusted Ultrafuzz CLI${launcherKept ? "; tasks keep calling the run-owned launcher, which checks itself on every call" : ""}`, + launcherKept + ? `resume could not re-verify the run's trusted Ultrafuzz CLI (tasks still call ${launcher}, and their preflight-json-validator step fails while that launcher cannot verify itself)` + : `resume could not re-verify the run's trusted Ultrafuzz CLI (${launcher} does not exist, so tasks call whatever \`ultrafuzz\` is on PATH)`, error ) ); diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index ca30a97d7..65db202c1 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -26,8 +26,10 @@ import { artifactContractDefinition, artifactContractSchemaBinding, artifactSchemaBundleDigest, + artifactSchemaDirectory, artifactSchemaRegistry, artifactSchemaRegistryFromDirectory, + artifactValidatorSmokeFixturePath, ARTIFACT_VALIDATOR_SMOKE_FIXTURE_SHA256, GOAL_PLAN_JSON_SCHEMA_ID, THREAT_MODEL_JSON_SCHEMA_ID, @@ -36,6 +38,7 @@ import { goalPlanJsonSchema, layoutForRunRoot, manifestDigest, + parseJsonValidatorPreflightSuccessEnvelope, readPlannedGraphDocument, readRunState, promptArtifactAuthorityPathSelectorId, @@ -25046,14 +25049,11 @@ test("native continuation does not use historical trusted CLI identity as an aut writeSmallTopology(project); const runId = "controller-refresh-trusted-cli"; const env = controllerRefreshTerminalEnv(project, runId); - const upPathLog = path.join(project, "fake-smithers-up-path.log"); - logFakeRunnerUpVariable(env, "PATH", upPathLog); const launched = await startRun({ projectRoot: project, runId, env }); assert.equal(launched.ok, true, JSON.stringify(launched.diagnostics)); const trustedMetadataPath = path.join(launched.value!.run_root, "trusted-cli.json"); fs.writeFileSync(trustedMetadataPath, "{}\n", "utf8"); fs.writeFileSync(env.SMITHERS_FAKE_LOG!, "", "utf8"); - fs.writeFileSync(upPathLog, "", "utf8"); const ordinary = await resumeRun({ projectRoot: project, runId, env }); assert.equal(ordinary.ok, true, JSON.stringify(ordinary.diagnostics)); @@ -25063,14 +25063,7 @@ test("native continuation does not use historical trusted CLI identity as an aut assert.equal(refreshed.ok, true, JSON.stringify(refreshed.diagnostics)); assert.equal(refreshed.value?.submitted, true); assert.equal(fs.readFileSync(trustedMetadataPath, "utf8"), "{}\n"); - // The run-owned launcher, which re-verifies itself on every call, stays - // first on the runner's PATH instead of leaving tasks to whatever - // `ultrafuzz` the operator's PATH holds, and the failure is reported (#1143). - const upPaths = fs.readFileSync(upPathLog, "utf8").trim().split("\n"); - assert.equal(upPaths.length, 2); - for (const upPath of upPaths) { - assert.equal(upPath.split(path.delimiter)[0], path.join(path.dirname(trustedMetadataPath), "trusted-bin")); - } + // The failed re-verification is reported instead of swallowed (#1143). for (const resumed of [ordinary, refreshed]) { assert.equal( resumed.diagnostics.some( @@ -25090,6 +25083,54 @@ test("native continuation does not use historical trusted CLI identity as an aut ); }); +test("a resume that cannot re-verify the trusted CLI leaves tasks on the run's own working launcher", async () => { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + writeSmallTopology(project); + const runId = "resume-keeps-trusted-launcher"; + const env = controllerRefreshTerminalEnv(project, runId); + const upPathLog = path.join(project, "fake-smithers-up-path.log"); + logFakeRunnerUpVariable(env, "PATH", upPathLog); + const launched = await startRun({ projectRoot: project, runId, env }); + assert.equal(launched.ok, true, JSON.stringify(launched.diagnostics)); + assert.ok(launched.value); + fs.writeFileSync(upPathLog, "", "utf8"); + + // Without a CLI entrypoint resume cannot re-verify the launcher, although + // the launcher and its closure are intact (#1143). + const resumed = await runtimeResumeRun({ projectRoot: project, runId, env }); + assert.equal(resumed.ok, true, JSON.stringify(resumed.diagnostics)); + assert.equal(resumed.value?.submitted, true); + + // Each task's validator preflight runs `ultrafuzz` from the runner's PATH. + // It must still reach the run's launcher, which verifies its closure and + // dispatches to the recorded (fake) CLI, never another `ultrafuzz`. + const runnerPath = fs.readFileSync(upPathLog, "utf8").trim(); + const resolved = runnerPath + .split(path.delimiter) + .map((entry) => path.join(entry, "ultrafuzz")) + .find((candidate) => fs.existsSync(candidate)); + assert.equal(resolved, path.join(launched.value.run_root, "trusted-bin", "ultrafuzz")); + const findings = artifactSchemaRegistry().find((entry) => entry.filename === "findings.schema.json"); + assert.ok(findings); + const stdout = execFileSync( + resolved, + [ + "json", + "validate", + "--schema", + path.join(artifactSchemaDirectory(), findings.filename), + "--file", + artifactValidatorSmokeFixturePath(), + "--json" + ], + { encoding: "utf8", env: { ...process.env, PATH: runnerPath } } + ); + parseJsonValidatorPreflightSuccessEnvelope(Buffer.from(stdout, "utf8")); + const warning = resumed.diagnostics.find((diagnostic) => diagnostic.code === "WORKFLOW_TRUSTED_CLI_UNVERIFIED"); + assert.equal(warning?.severity, "warning", JSON.stringify(resumed.diagnostics)); +}); + test("native continuation hands generated agents the run's TOML config, so CodexAgent keeps API-key auth", async () => { const project = tempProject(); initProject({ projectRoot: project, force: true }); From 9029ff805d9a797b16314d6ef3acf87993e13e7e Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 03:21:57 +0000 Subject: [PATCH 5/7] fix(cli): the no-op resume message also covers a controller that just exited An attach happens in two cases. One is a pause that is still draining. The other is a controller process that exited less than 30 seconds ago. In the pinned Smithers, deriveRunState keeps a run `running` until its heartbeat is 30 s stale. `ultrafuzz status` also maps the later `stale`/`orphaned` states to `running`. So "resume again once status reports paused" never fires in the second case, and an operator who follows it waits indefinitely. The message now adds: "if its controller process just exited, resume again after 30 seconds". After that window the run is no longer active and resume submits a continuation. Three statements were wider than the code: - docs/reference/cli.md said adapters read the launch config unconditionally. They do so only when the run's resolved-config.json parses as the current schema; otherwise resume sets no ULTRAFUZZ_CONFIG_PATH. - The start-run comment said the state "belongs to the live owner", but in the second case no owner is alive. - The CHANGELOG entry carried the same unconditional claims. All three now match the code. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- docs/reference/cli.md | 15 ++++++++++----- packages/cli/src/commands/resume.ts | 2 +- packages/runtime/src/start-run.ts | 5 +++-- 4 files changed, 15 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 66f2072d1..0b9cd32d3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,7 @@ ### Other changes -- **[runtime] [cli] [docs]** `resume` now points agent adapters at the run's TOML agent config (`smithers/execution-config.toml`) instead of `resolved-config.json`, which their TOML reader parsed as empty, so a resumed run keeps each agent's `auth`, `api_key_env`, and `config_dir` (the default `CodexAgent` API-key auth had fallen back to subscription). A resume that only attaches to an already-active run no longer rewrites `state.json` or pushes back `workflow_deadline_at`, and prints `Run already active` instead of `Submitted resume`. A failed stale task-worktree cleanup no longer fails the resume and a failed trusted-CLI re-verification is no longer silent: both are now warnings, and in the latter case the run-owned launcher stays first on `PATH` instead of leaving tasks to whatever `ultrafuzz` the operator's `PATH` holds (#1153, #1143, #1110). +- **[runtime] [cli] [docs]** When a run's resolved config is readable, `resume` now points agent adapters at the run's TOML agent config (`smithers/execution-config.toml`) instead of `resolved-config.json`, which their TOML reader parsed as empty, so a resumed run keeps each agent's `auth`, `api_key_env`, and `config_dir` (the default `CodexAgent` API-key auth had fallen back to subscription). A resume that only attaches to an already-active run no longer rewrites `state.json` or pushes back `workflow_deadline_at`, and prints `Run already active` instead of `Submitted resume`. A failed stale task-worktree cleanup no longer fails the resume, and a failed trusted-CLI re-verification is no longer silent: both are now warnings, and in the latter case an existing run-owned launcher stays first on `PATH` instead of leaving tasks to whatever `ultrafuzz` the operator's `PATH` holds (a launcher that cannot verify itself still fails each task's validator preflight) (#1153, #1143, #1110). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955). diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 3c16047e1..5267afc0b 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -323,10 +323,13 @@ schema bindings, and metadata projections remain provenance for inspection; they are not resume authorization. Smithers decides which finished rows can be reused and which newly rendered or unfinished tasks run. Ultrafuzz does not rewrite historical artifacts or automatically reset, replay, timetravel, or -fork completed work. Agent adapters in the continued workflow read the same -agent config as at launch, from the run's `smithers/execution-config.toml`. If -resume cannot prune stale task-worktree registrations, it reports a -`WORKFLOW_WORKTREE_REPAIR_FAILED` warning and continues. +fork completed work. When the run's `smithers/resolved-config.json` parses as +the current resolved-config schema, agent adapters in the continued workflow +read the run's `smithers/execution-config.toml` (launch gave them a copy of the +same file); for a run whose resolved config does not parse, resume sets no +`ULTRAFUZZ_CONFIG_PATH`. If resume cannot prune stale task-worktree +registrations, it reports a `WORKFLOW_WORKTREE_REPAIR_FAILED` warning and +continues. `resume --refresh-controller` first renders the currently installed Ultrafuzz controller and stock adapters beside the historical source, then delegates to @@ -466,7 +469,9 @@ duplicate continuation when the linked workflow is still active (its Smithers run state is `running`, `recovering`, or one of the `waiting-*` states), and leaves the run's recorded state and workflow deadline unchanged. A run still finishing its in-flight tasks after `pause` is still active; resume it again -once `status` reports `paused`. `resume --reset-node` retries one failed +once `status` reports `paused`. Smithers also reports a run as `running` for up +to 30 seconds after its controller process exits (its heartbeat window), so +resume such a run again after that. `resume --reset-node` retries one failed workflow node and its dependents in the same linked run; the applied reset is recorded so retrying the command after a failed continuation resumes the already-reset run instead of repeating the reset. `fork` may start from a diff --git a/packages/cli/src/commands/resume.ts b/packages/cli/src/commands/resume.ts index 09f2fc96d..44e701df8 100644 --- a/packages/cli/src/commands/resume.ts +++ b/packages/cli/src/commands/resume.ts @@ -41,7 +41,7 @@ export default class Resume extends Command { const commandResult = commandFromRuntime("resume", result, (value) => value.submitted ? `Submitted ${value.action}: ${value.workflow_run_id}\n` - : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused.\n` + : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused; if its controller process just exited, resume again after 30 seconds.\n` ); if (result.ok && result.diagnostics.length > 0) { commandResult.text = `${commandResult.text ?? ""}${diagnosticsText(result.diagnostics)}`; diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index c380631c9..62b3a79e6 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -700,8 +700,9 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { trustedCli.environmentVariableNames ) }); - // Attaching to an already-active run started no controller. Its state - // (status, lease, deadline) belongs to the live owner, so leave it alone. + // An attach to a run Smithers still reports active started no controller, + // so it must not re-record status, lease or deadline; the resume that + // starts the next controller does. if (result.alreadyRunning !== true) { recordNativeContinuationState({ layout, config, requestedConcurrency: input.maxConcurrency }); } From 10ce4e5d3040757fa103fef977e2308928dfddb3 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 03:22:31 +0000 Subject: [PATCH 6/7] test(runtime): let the pause-handoff guard's engines exit before deleting their root The guard's finally block wrote the release marker and then deleted the temp root in the same tick, so the marker was deleted too. The held tasks poll every 50 ms and almost never saw it. After any assertion failure the detached owner kept running until its 120 s hold expired and then wrote its logs back under the deleted root. Passing runs were not affected, because by then every task had finished. Teardown now writes the marker and then waits, bounded at 30 s, until every process that traced a task has exited before it deletes the root. Checked with a copy of the test where the takeover uses --steal-ownership instead of --force. The test fails as intended (DETACHED_ADMISSION_FAILED / RUN_RESUME_CLAIM_FAILED instead of RUN_OWNER_ALIVE). The owner engine had exited by the time the run returned. Two and a half minutes later, past the 120 s hold, the root had not reappeared. The unmodified file still passes, all 4 tests. Co-Authored-By: Claude Opus 5.5 --- ...ithers-preparation-race.integration.test.ts | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/packages/runtime/test/smithers-preparation-race.integration.test.ts b/packages/runtime/test/smithers-preparation-race.integration.test.ts index ad778d442..7bed7acff 100644 --- a/packages/runtime/test/smithers-preparation-race.integration.test.ts +++ b/packages/runtime/test/smithers-preparation-race.integration.test.ts @@ -311,7 +311,16 @@ test("graceful pause drains in-flight tasks and refuses a second controller unti } assert.notEqual(pidOf(finished, "c start"), ownerPid, "only the replacement controller ran c"); } finally { + // Release any task still held, then give every engine that ran a task + // time to exit. Deleting the root first would delete the release marker + // too, leaving a detached engine polling until its 120 s hold expires and + // writing its logs back under the deleted root. fs.writeFileSync(releasePath, "", "utf8"); + await waitUntil( + () => !trace().some((line) => processIsAlive(Number(line.split(" ")[0]))), + 30_000, + "the detached engines exit" + ).catch(() => undefined); fs.rmSync(root, { recursive: true, force: true }); } }); @@ -506,6 +515,15 @@ async function waitForFile(filePath: string, timeoutMs: number): Promise { throw new Error(`timed out waiting for ${filePath}`); } +function processIsAlive(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch (error) { + return (error as NodeJS.ErrnoException).code === "EPERM"; + } +} + async function waitUntil(condition: () => boolean, timeoutMs: number, label: string): Promise { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { From a87aaa5c88e32762f7d399f4d3998f23bcc2a26f Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 03:55:23 +0000 Subject: [PATCH 7/7] chore: move the changelog entry to the consolidated release notes Every pull request in this batch inserts its entry at the same place in CHANGELOG.md, so each merge would conflict with the next. The entries are collected into one changelog update instead. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 - 1 file changed, 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0b9cd32d3..960ba9394 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,7 +12,6 @@ ### Other changes -- **[runtime] [cli] [docs]** When a run's resolved config is readable, `resume` now points agent adapters at the run's TOML agent config (`smithers/execution-config.toml`) instead of `resolved-config.json`, which their TOML reader parsed as empty, so a resumed run keeps each agent's `auth`, `api_key_env`, and `config_dir` (the default `CodexAgent` API-key auth had fallen back to subscription). A resume that only attaches to an already-active run no longer rewrites `state.json` or pushes back `workflow_deadline_at`, and prints `Run already active` instead of `Submitted resume`. A failed stale task-worktree cleanup no longer fails the resume, and a failed trusted-CLI re-verification is no longer silent: both are now warnings, and in the latter case an existing run-owned launcher stays first on `PATH` instead of leaving tasks to whatever `ultrafuzz` the operator's `PATH` holds (a launcher that cannot verify itself still fails each task's validator preflight) (#1153, #1143, #1110). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955).