From f1e07b9650b077c11551329a4bbaa5d3c063e88d Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 12:30:15 +0000 Subject: [PATCH 1/3] fix(cli): plain status, inspect and why print the warnings of a successful poll commandFromRuntime rendered a result's diagnostics only when the result failed. getRunHealth returns ok unless a diagnostic is an error, and the sync and deadline problems it reports are warnings (WORKFLOW_DEADLINE_CANCEL_FAILED, WORKFLOW_STATE_SYNC_SKIPPED, WORKFLOW_CONTROL_EVIDENCE_DIVERGED and the transient sync codes), so plain `ultrafuzz status` exited 0 without them. inspect and why render through the same helper. docs/reference/configuration.md tells operators to run `ultrafuzz status` from cron and act on its warnings. A successful result now prints its diagnostics after the rendered value, and a failure still leads with them. resume and init appended successful diagnostics themselves; those copies are gone. JSON output is unchanged. The status reference now says where the warnings appear. The new CLI test sets a past workflow_deadline_at and has the fake runner answer cancel outside its status contract. On origin/main plain status prints no warning line; with this change it prints `warning: WORKFLOW_DEADLINE_CANCEL_FAILED: ...`. Co-Authored-By: Claude Opus 5.5 --- docs/reference/cli.md | 4 ++++ packages/cli/src/command-shared.ts | 5 ++++- packages/cli/src/commands/init.ts | 5 +---- packages/cli/src/commands/resume.ts | 4 ---- packages/cli/test/lifecycle-commands.test.ts | 13 +++++++++++++ 5 files changed, 22 insertions(+), 9 deletions(-) diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 83f79ad0b..984c3e1c1 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -388,6 +388,10 @@ Time on current step: 10 minutes on stateful-invariant-campaign Pace: 4 finished in the last 10m ``` +A poll that succeeds with diagnostics, such as a `WORKFLOW_DEADLINE_CANCEL_FAILED` +or `WORKFLOW_STATE_SYNC_SKIPPED` warning, prints each one after these lines as +`: : `, the same diagnostics `--json` reports. + `--watch` re-polls every `--interval` seconds (default 30) until the run reaches a terminal state (`succeeded`, `failed`, `timed-out`, or `canceled`), needs attention, is paused, or the poll fails. With `--json --watch`, every poll writes one diff --git a/packages/cli/src/command-shared.ts b/packages/cli/src/command-shared.ts index 9ae43b5d8..62b9873ff 100644 --- a/packages/cli/src/command-shared.ts +++ b/packages/cli/src/command-shared.ts @@ -131,8 +131,11 @@ export function commandFromRuntime( : project === undefined ? (result.value as CliCommandDataMap[CommandName]) : project(result.value), + // A successful result still prints its diagnostics, after the value; a failure leads with them. text: result.value - ? `${result.ok ? "" : diagnosticsText(result.diagnostics)}${text(result.value)}` + ? result.ok + ? `${text(result.value)}${diagnosticsText(result.diagnostics)}` + : `${diagnosticsText(result.diagnostics)}${text(result.value)}` : diagnosticsText(result.diagnostics), diagnostics: result.diagnostics }; diff --git a/packages/cli/src/commands/init.ts b/packages/cli/src/commands/init.ts index c35c03eaa..13b00176d 100644 --- a/packages/cli/src/commands/init.ts +++ b/packages/cli/src/commands/init.ts @@ -1,7 +1,7 @@ import { Command, Flags } from "@oclif/core"; import { initProject } from "@ultrafuzz/runtime"; -import { commandFromRuntime, diagnosticsText, emitCommandResult, globalFlags, projectRoot } from "../command-shared.js"; +import { commandFromRuntime, emitCommandResult, globalFlags, projectRoot } from "../command-shared.js"; export default class Init extends Command { static override summary = "Create project-owned Ultrafuzz surfaces"; @@ -22,9 +22,6 @@ export default class Init extends Command { "" ].join("\n") ); - if (result.ok && result.diagnostics.length > 0) { - commandResult.text = `${commandResult.text ?? ""}${diagnosticsText(result.diagnostics)}`; - } emitCommandResult(this, "init", commandResult, flags.json === true); } } diff --git a/packages/cli/src/commands/resume.ts b/packages/cli/src/commands/resume.ts index 68deba422..1e6c6f748 100644 --- a/packages/cli/src/commands/resume.ts +++ b/packages/cli/src/commands/resume.ts @@ -5,7 +5,6 @@ import { cliEntrypoint, cliIo, commandFromRuntime, - diagnosticsText, emitCommandResult, globalFlags, projectRoot @@ -43,9 +42,6 @@ export default class Resume extends Command { ? `Submitted ${value.action}: ${value.workflow_run_id ?? value.run_id}\n` : `Run already active: ${value.workflow_run_id ?? value.run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused; if its controller process just exited, resume again after 30 seconds.\n` ); - if (result.ok && result.diagnostics.length > 0) { - commandResult.text = `${commandResult.text ?? ""}${diagnosticsText(result.diagnostics)}`; - } emitCommandResult(this, "resume", commandResult, flags.json); } } diff --git a/packages/cli/test/lifecycle-commands.test.ts b/packages/cli/test/lifecycle-commands.test.ts index 9bb832fa1..5a38aa885 100644 --- a/packages/cli/test/lifecycle-commands.test.ts +++ b/packages/cli/test/lifecycle-commands.test.ts @@ -585,6 +585,19 @@ test("cancel distinguishes a submitted request from a confirmed cancellation", a assert.match(humanConfirmed.stdout, /^Cancellation confirmed: lifecycle-cli-run is canceled$/mu); }); +test("plain status prints the warnings of a successful poll", async (t) => { + // A cancel reply outside the runner's status contract fails the deadline cancel. + const { project, env, runRoot } = await launchedProject(t, { cancelStatus: "busy" }); + const statePath = path.join(runRoot, "state.json"); + const state = JSON.parse(fs.readFileSync(statePath, "utf8")) as Record; + fs.writeFileSync(statePath, `${JSON.stringify({ ...state, workflow_deadline_at: "2000-01-01T00:00:00.000Z" })}\n`); + + const status = await cli(project, ["status", RUN_ID], env); + + assert.equal(status.code, 0, status.stderr); + assert.match(status.stdout, /^warning: WORKFLOW_DEADLINE_CANCEL_FAILED: /mu); +}); + test("doctor reports install posture in human and JSON output", async (t) => { const { project, env } = await launchedProject(t); writeSmallTopology(project, "recon"); From 36819a162dcd71229acbad0ad876e1361bd134ed Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 12:30:24 +0000 Subject: [PATCH 2/3] test(cli): the campaign e2e checks that a supervisor relaunch of a sealed engine re-runs the interrupted node No test checked that a fresh `ultrafuzz run`, whose engine runs from the sealed execution snapshot, recovers when only its engine dies. The real-engine relaunch test runs a plain-path engine, the fd-transfer harness runs a toy script, and this e2e killed the engine together with its supervisor, then recovered through `ultrafuzz resume`. The e2e now first SIGKILLs only the detached engine (the `up` process, not `supervise`) while summarize is held, and waits, with no ultrafuzz command, for the relaunched engine to hold summarize again in a new process. It then kills the whole controller, which is now the relaunched engine and its supervisor, and resumes, so summarize starts three times and the event log has three RunStarted and a RunAutoResumed. The relaunched engine does not finish the run: it is killed while it holds summarize, and the engine `ultrafuzz resume` starts finishes it. The resume phase therefore starts from a supervisor-relaunched engine instead of the one `ultrafuzz run` started. Resume takes the workflow path from Ultrafuzz's run metadata, so what differs is the state the relaunched engine wrote to the Smithers database. Co-Authored-By: Claude Opus 5.5 --- packages/cli/test/e2e/campaign-resume.test.ts | 51 ++++++++++++++----- 1 file changed, 39 insertions(+), 12 deletions(-) diff --git a/packages/cli/test/e2e/campaign-resume.test.ts b/packages/cli/test/e2e/campaign-resume.test.ts index 3b5765028..00e682569 100644 --- a/packages/cli/test/e2e/campaign-resume.test.ts +++ b/packages/cli/test/e2e/campaign-resume.test.ts @@ -358,7 +358,24 @@ function assertNoFinishedTaskRestarted(events: WorkflowEvent[]): void { assert.ok(finished.size > 0, "the workflow event stream has no finished tasks"); } -/** Launches the campaign and SIGKILLs its detached controller while `INTERRUPTED_NODE` runs. */ +/** Waits for the stub to hold `INTERRUPTED_NODE` in a process other than the `earlier` ones. */ +async function heldCall(campaign: Campaign, workflowRunId: string, label: string, earlier: number[] = []) { + return waitFor(label, 15 * MINUTE, () => { + const calls = agentCalls(campaign); + const call = calls.find( + (entry) => entry.node === INTERRUPTED_NODE && entry.event === "held" && !earlier.includes(entry.pid) + ); + if (call === undefined && processesMentioning(workflowRunId).length === 0) { + assert.fail(`the workflow stopped while waiting for ${label}; agent calls: ${JSON.stringify(calls)}`); + } + return call; + }); +} + +/** + * Launches the campaign and SIGKILLs its detached engine while `INTERRUPTED_NODE` runs. Once the + * supervisor has relaunched the engine and the node runs again, SIGKILLs the whole controller. + */ async function interruptMidRun(campaign: Campaign, mark: (phase: string) => void): Promise { await ultrafuzz(campaign, ["init"], 2 * MINUTE); const configPath = path.join(campaign.project, "ultrafuzz.toml"); @@ -383,14 +400,19 @@ async function interruptMidRun(campaign: Campaign, mark: (phase: string) => void const [workflowRunId] = launched.workflow_ids; assert.ok(workflowRunId !== undefined, "run did not report its workflow run ID"); - const held = await waitFor(`${INTERRUPTED_NODE} to start`, 15 * MINUTE, () => { - const calls = agentCalls(campaign); - const call = calls.find((entry) => entry.node === INTERRUPTED_NODE && entry.event === "held"); - if (call === undefined && processesMentioning(workflowRunId).length === 0) { - assert.fail(`the workflow stopped before ${INTERRUPTED_NODE} started; agent calls: ${JSON.stringify(calls)}`); - } - return call; + const first = await heldCall(campaign, workflowRunId, `${INTERRUPTED_NODE} to start`); + // An engine killed on its own, as by the OOM killer, leaves its supervisor running. The supervisor + // relaunches it from the run's sealed execution snapshot with no `ultrafuzz` command. + const engines = processesMentioning(workflowRunId).filter((pid) => { + const argv = commandLine(pid) ?? ""; + return argv.includes("\0up\0") && !argv.includes("\0supervise\0"); }); + assert.ok(engines.length > 0, "no detached engine process is running the workflow"); + for (const pid of engines) kill(pid); + const held = await heldCall(campaign, workflowRunId, `the relaunched engine to rerun ${INTERRUPTED_NODE}`, [ + first.pid + ]); + mark("engine relaunched"); // A host crash takes down the detached engine and the supervisor that would otherwise restart it. const controller = processesMentioning(workflowRunId); assert.ok(controller.length > 0, "no detached controller process is running the workflow"); @@ -403,7 +425,7 @@ async function interruptMidRun(campaign: Campaign, mark: (phase: string) => void } test( - "a campaign whose controller is SIGKILLed mid-node resumes on the pinned engine without re-running finished nodes", + "a campaign whose engine and then whole controller are SIGKILLed mid-node resumes on the pinned engine without re-running finished nodes", { timeout: 45 * MINUTE, skip: process.platform === "linux" ? false : "finds the detached controller through /proc" }, async (t) => { const campaign = prepareCampaign(); @@ -452,14 +474,19 @@ test( "RunFinished", `agent calls: ${JSON.stringify(calls)}` ); - // Every agent ran once, except the node the kill interrupted, which ran again after resume. + // Every agent ran once, except the node the kills interrupted, which ran again in the relaunched + // engine and after resume. const starts = calls.filter((call) => call.event === "started"); assert.deepEqual( AGENT_NODES.map((node) => [node, starts.filter((call) => call.node === node).length]), - AGENT_NODES.map((node) => [node, node === INTERRUPTED_NODE ? 2 : 1]) + AGENT_NODES.map((node) => [node, node === INTERRUPTED_NODE ? 3 : 1]) ); assert.equal(events.truncated, false); - assert.equal(events.events.filter((event) => event.category === "RunStarted").length, 2); + assert.equal(events.events.filter((event) => event.category === "RunStarted").length, 3); + assert.ok( + events.events.some((event) => event.category === "RunAutoResumed"), + "no supervisor relaunch event" + ); assertNoFinishedTaskRestarted(events.events); const health = await ultrafuzz(campaign, ["status", runId]); From 0f50e93577af159a2613a48aa485818060e97e52 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 14:23:02 +0000 Subject: [PATCH 3/3] docs: status warnings print on stdout, and scripts read the --json diagnostics A successful status poll prints its warnings on stdout without changing the exit status, and a message can continue on lines without a severity prefix. A cron job that discards stdout, or greps for `warning:` lines, can therefore miss them or see only the first line of each. The status reference now says so, and the deadline advice in the configuration reference points scripts at the `diagnostics` of `ultrafuzz status --json`. Co-Authored-By: Claude Opus 5.5 --- docs/reference/cli.md | 6 ++++-- docs/reference/configuration.md | 5 +++-- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 984c3e1c1..db35ac128 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -389,8 +389,10 @@ Pace: 4 finished in the last 10m ``` A poll that succeeds with diagnostics, such as a `WORKFLOW_DEADLINE_CANCEL_FAILED` -or `WORKFLOW_STATE_SYNC_SKIPPED` warning, prints each one after these lines as -`: : `, the same diagnostics `--json` reports. +or `WORKFLOW_STATE_SYNC_SKIPPED` warning, prints each one on stdout after these +lines as `: : `, where a message can span several +lines. The warnings do not change the exit status, so scripts should read them +from the `diagnostics` array of `--json` output. `--watch` re-polls every `--interval` seconds (default 30) until the run reaches a terminal state (`succeeded`, `failed`, `timed-out`, or `canceled`), diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 37d538ae7..f528a9465 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -159,8 +159,9 @@ next synchronization requests cancellation again; a synchronization that fails or is skipped (for example `WORKFLOW_STATE_SYNC_SKIPPED` or `WORKFLOW_SYNC_IN_PROGRESS`) does not check the deadline at all. A run that finished first keeps its terminal outcome, with no timeout record. To bound an -unattended run, run `ultrafuzz status ` periodically (for example from -cron) and act on its warnings, or cancel it with `ultrafuzz cancel `. +unattended run, run `ultrafuzz status --json` periodically (for example +from cron) and act on the warnings in its `diagnostics` (plain `status` prints +them on stdout), or cancel it with `ultrafuzz cancel `. Workflow-side enforcement is tracked in [#1110](https://github.com/monad-developers/ultrafuzz/issues/1110).