From 1cae41cdee58a3cc8fac2a118e8570f1d14278f3 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 06:48:55 +0000 Subject: [PATCH 01/10] feat(runtime): continuing results are optional to every other group, not only review #1120 made a failure_policy: continue producer's output optional only to consumers in the group literally named `review`, in both the compiler and dynamic lowering. Any other consumer that reconciles partial results, such as a property fan-in or the exhaustive dynamic strategy generator, was skipped as soon as one continuing input failed. reconcilesPartialResults now takes the producer's group: a continuing producer's output is optional to a consumer in any other group and stays required inside its own group. Chains inside one group, such as the stateful invariant stages, therefore still never run without their predecessor, which is the case #1120 protected. The compiler and dynamic lowering both call the helper. The launch task-manifest gate checks only that every optional input names a continuing producer, so it is unchanged. With the shipped topologies as they are, one thing changes: in the exhaustive profile dynamic-strategy-generator (specialists) runs with the strategy attempts that succeeded instead of being skipped when any of its 58 strategy inputs fails. It reads them through the verifier-admitted ancestor authorities, which omit a failed producer. Co-Authored-By: Claude Opus 5.5 --- .../artifacts/src/smithers-task-manifest.ts | 2 +- packages/runtime/src/dynamic-runtime.ts | 25 +++++++++++++------ packages/runtime/src/smithers.ts | 20 ++++++++------- .../runtime/test/dynamic-expansion.test.ts | 7 ++++-- .../test/workflow-dependency-policy.test.ts | 7 +++--- 5 files changed, 38 insertions(+), 23 deletions(-) diff --git a/packages/artifacts/src/smithers-task-manifest.ts b/packages/artifacts/src/smithers-task-manifest.ts index 98ee0fca4..cfe746cea 100644 --- a/packages/artifacts/src/smithers-task-manifest.ts +++ b/packages/artifacts/src/smithers-task-manifest.ts @@ -913,7 +913,7 @@ export interface SmithersTaskManifestTask { vulnerabilityDatabaseCatalog?: { path: string; sha256: string }; /** Stable byte authorities for every transitive reference ancestor's outer artifact manifest. */ referenceArtifactManifestAuthorities?: SmithersTaskManifestReferenceArtifactManifestAuthority[]; - /** Exact subset whose producer group uses failure_policy=continue. */ + /** Subset whose producer group uses failure_policy=continue and differs from this task's group. */ optionalDependencyArtifactDirs?: string[]; /** Canonical union of compact ancestor-output selectors used by the rendered prompt. */ promptArtifactAuthoritySelectors?: SmithersTaskManifestPromptArtifactAuthoritySelector[]; diff --git a/packages/runtime/src/dynamic-runtime.ts b/packages/runtime/src/dynamic-runtime.ts index 501adea7b..4741c1522 100644 --- a/packages/runtime/src/dynamic-runtime.ts +++ b/packages/runtime/src/dynamic-runtime.ts @@ -332,13 +332,18 @@ function instantiateDynamicGraphNode(input: { } /** - * Continuation lets independent tasks settle; it does not make a strategy's required inputs - * optional. Only the review group reconciles partial results, so only its tasks treat a continuing - * producer's output as optional (#1120). The compiler and dynamic lowering share this one rule; the - * task-manifest gate checks only that optional inputs come from continuing producers. + * Whether `consumer` treats the output of a continuing producer in `producerGroup` as optional. + * Consumers outside that group reconcile whichever results succeeded (the review group, and the + * property fan-in reading its lenses). Inside the group a chain keeps its required inputs, so a + * strategy or stateful stage never runs without its predecessor's output (#1120). The compiler and + * dynamic lowering share this one rule; the task-manifest gate checks only that optional inputs + * come from continuing producers. */ -export function reconcilesPartialResults(task: Pick): boolean { - return task.metadata.node.group === "review"; +export function reconcilesPartialResults( + consumer: Pick, + producerGroup: string | undefined +): boolean { + return consumer.metadata.node.group !== producerGroup; } function lowerTaskDynamicDependencies( @@ -381,8 +386,12 @@ function lowerTaskDynamicDependencies( dependencies.push(...generated.map((candidate) => candidate.attemptId)); dependencySmithersNodeIds.push(...generated.map((candidate) => candidate.verifierSmithersNodeId)); dependencyArtifactDirs.push(...generated.map((candidate) => candidate.artifactDir)); - if (group.continueOnFail && reconcilesPartialResults(task)) { - optionalDependencyArtifactDirs.push(...generated.map((candidate) => candidate.artifactDir)); + if (group.continueOnFail) { + optionalDependencyArtifactDirs.push( + ...generated + .filter((candidate) => reconcilesPartialResults(task, candidate.metadata.node.group)) + .map((candidate) => candidate.artifactDir) + ); } concreteNodeIds.push(...manifest.items.map((item) => item.node_id)); } diff --git a/packages/runtime/src/smithers.ts b/packages/runtime/src/smithers.ts index 7ff87dff0..a975179ec 100644 --- a/packages/runtime/src/smithers.ts +++ b/packages/runtime/src/smithers.ts @@ -3768,19 +3768,21 @@ export function compileSmithersWorkflow(input: SmithersCompileInput): CompiledSm workflowName }) ); - const nonBlockingAttemptIds = compiledTasks - .filter((task) => { + const nonBlockingGroupByAttemptId = new Map( + compiledTasks.flatMap((task) => { const group = task.metadata.node.group; - return group !== undefined && input.graph.groups[group]?.defaults?.failure_policy === "continue"; + return group !== undefined && input.graph.groups[group]?.defaults?.failure_policy === "continue" + ? [[task.attemptId, group] as const] + : []; }) - .map((task) => task.attemptId) - .sort(); - const nonBlockingAttemptIdSet = new Set(nonBlockingAttemptIds); + ); + const nonBlockingAttemptIds = [...nonBlockingGroupByAttemptId.keys()].sort(); const tasks = compiledTasks.map((task) => ({ ...task, - optionalDependencyArtifactDirs: reconcilesPartialResults(task) - ? task.dependencyArtifactDirs.filter((directory) => nonBlockingAttemptIdSet.has(path.basename(directory))) - : [] + optionalDependencyArtifactDirs: task.dependencyArtifactDirs.filter((directory) => { + const producerGroup = nonBlockingGroupByAttemptId.get(path.basename(directory)); + return producerGroup !== undefined && reconcilesPartialResults(task, producerGroup); + }) })); const smithersDir = path.join(input.runLayout.root, "smithers"); fs.mkdirSync(smithersDir, { recursive: true }); diff --git a/packages/runtime/test/dynamic-expansion.test.ts b/packages/runtime/test/dynamic-expansion.test.ts index d375937b9..1db710d6c 100644 --- a/packages/runtime/test/dynamic-expansion.test.ts +++ b/packages/runtime/test/dynamic-expansion.test.ts @@ -642,10 +642,12 @@ test("a lock file left by a killed materializer blocks neither expansion nor a s ); }); -test("runtime materialization preserves required inputs and allows partial review joins", () => { +test("runtime materialization keeps inputs inside the fan-out's group required and allows partial joins", () => { + // The continuing fan-out belongs to `strategies`; a join in any other group reconciles it. for (const [count, consumerGroup] of [ [0, "review"], [1, "review"], + [1, "catalog"], [0, "strategies"], [1, "strategies"] ] as const) { @@ -669,6 +671,7 @@ test("runtime materialization preserves required inputs and allows partial revie fs.writeFileSync(graphPath, `${JSON.stringify(graph)}\n`, "utf8"); fs.writeFileSync(tasksPath, `${JSON.stringify({ schema_version: "1.0", run_id: runId, tasks: [] })}\n`, "utf8"); const templateTask = compiledTask(projectRoot, runRoot, "fanout", "fanout", templatePath); + templateTask.metadata.node.group = "strategies"; const joinTask = compiledTask(projectRoot, runRoot, "join", "join", undefined, ["fanout"]); joinTask.metadata.node.group = consumerGroup; if (count === 0) joinTask.optionalDependencyArtifactDirs = [path.join(runRoot, "artifacts", "planner")]; @@ -731,7 +734,7 @@ test("runtime materialization preserves required inputs and allows partial revie assert.deepEqual(storedJoin.dependencySmithersNodeIds, [generated.verifierSmithersNodeId]); assert.deepEqual( storedJoin.optionalDependencyArtifactDirs, - consumerGroup === "review" ? [generated.artifactDir] : [] + consumerGroup === "strategies" ? [] : [generated.artifactDir] ); assert.match(fs.readFileSync(generated.renderedPromptPath!, "utf8"), /goal 0 using context 0/u); } diff --git a/packages/runtime/test/workflow-dependency-policy.test.ts b/packages/runtime/test/workflow-dependency-policy.test.ts index 43a24cc51..7979c7790 100644 --- a/packages/runtime/test/workflow-dependency-policy.test.ts +++ b/packages/runtime/test/workflow-dependency-policy.test.ts @@ -40,21 +40,22 @@ function dependencyStateHelpers(): { >; } -test("workflow dependency policy preserves required strategy inputs and partial review fan-in", () => { +test("workflow dependency policy keeps inputs inside a continuing group required and lets another group reconcile", () => { const projectRoot = temporaryRoot("ufz-dependency-policy-"); const runId = "dependency-policy"; const runLayout = createRunLayout({ outputRoot: path.join(projectRoot, "runs"), runId }); + // Any group can reconcile a continuing group's results; the name `review` is not special. const topology: ProjectTopology = { version: 2, defaults: { strategy_loops: 1 }, - groups: { strategies: { defaults: { failure_policy: "continue" } }, review: {} }, + groups: { strategies: { defaults: { failure_policy: "continue" } }, catalog: {} }, nodes: [ { id: "__start__", kind: "meta", role: "start", depends_on: [] }, ...[ { id: "producer", group: "strategies", depends_on: ["__start__"] }, { id: "dependent", group: "strategies", depends_on: ["producer"] }, { id: "independent", group: "strategies", depends_on: ["__start__"] }, - { id: "review", group: "review", depends_on: ["dependent", "independent"] } + { id: "review", group: "catalog", depends_on: ["dependent", "independent"] } ].map((node) => ({ ...node, kind: "agentic" as const, From fdea1e686aded751283e3182a4e98c7c71f8a8ff Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 06:49:02 +0000 Subject: [PATCH 02/10] fix(runtime): host property fan-in gates skip a lens its verifier did not admit When the host re-verifies a finalized property fan-in (semanticPropertyLenses and verifyLensReferenceExpectationPreservation, both through sealedDirectArtifactDependencies), it loaded every declared lens dependency, including an optional one that the in-engine verifier left out because it failed. A fan-in that correctly consolidated only the verified lenses was then rejected on the host with PROPERTY_LENS_AUTHORITY_INVALID ("has no current finalized output authority"). sealedDirectArtifactDependencies now drops an optional producer that is absent from the verifier-persisted admission, using the same optionalDeclaredProducerWasNotAdmitted check that finalizedDeclaredContractProducers and the in-engine gate already apply. Required lenses, and optional lenses the verifier admitted, are still loaded and authenticated. A catalog that cites a property from the omitted lens is still rejected by the property-source-join gate as an unknown source. Co-Authored-By: Claude Opus 5.5 --- packages/runtime/src/artifact-gates.ts | 9 +- packages/runtime/test/artifact-gates.test.ts | 86 ++++++++++++++++++++ 2 files changed, 93 insertions(+), 2 deletions(-) diff --git a/packages/runtime/src/artifact-gates.ts b/packages/runtime/src/artifact-gates.ts index 7633c41f8..f34655b51 100644 --- a/packages/runtime/src/artifact-gates.ts +++ b/packages/runtime/src/artifact-gates.ts @@ -835,7 +835,10 @@ type DirectArtifactDependency = { task?: SmithersTaskManifestTask; }; -/** Resolve the current attempt's exact direct dependencies from its sealed task declaration. */ +/** + * Resolve the current attempt's exact direct dependencies from its sealed task declaration, + * omitting an optional producer its verifier did not admit. + */ function sealedDirectArtifactDependencies( layout: RunLayout, consumer: PlannedGraphNode, @@ -869,7 +872,9 @@ function sealedDirectArtifactDependencies( `sealed Smithers dependency ${JSON.stringify(dependencyAttemptId)} does not bind one planned agentic attempt` ); } - dependencies.push({ attemptId: dependencyAttemptId, node, task }); + if (!optionalDeclaredProducerWasNotAdmitted(dependencyAttemptId, task, authority)) { + dependencies.push({ attemptId: dependencyAttemptId, node, task }); + } continue; } diff --git a/packages/runtime/test/artifact-gates.test.ts b/packages/runtime/test/artifact-gates.test.ts index ca8bfaecf..c882bd65a 100644 --- a/packages/runtime/test/artifact-gates.test.ts +++ b/packages/runtime/test/artifact-gates.test.ts @@ -7389,6 +7389,92 @@ for (const finalizationState of ["failed", "unfinalized"] as const) { }); } +test("property fan-in consumes the admitted lenses when an optional lens failed", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-optional-failed-lens" }); + const node = writeMinimalPropertyFaninFixture(layout, { + dependsOn: ["property-specification-recon", "property-specification-crytic"] + }); + writePropertyLens(layout, "property-specification-recon", ["recon-1"]); + // The continue-policy lens failed before its verifier wrote a marker. + registerArtifactNode(layout, "property-specification-crytic", [ + boundOutput("properties/crytic.json", "ultrafuzz/property-lens@2", true) + ]); + updateNodeState(layout, "property-specification-crytic", { status: "failed" }); + // As a required input, the failed lens still rejects the fan-in. + const required = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.ok( + required.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_LENS_AUTHORITY_INVALID"), + JSON.stringify(required.diagnostics) + ); + + // As packaged: the lenses continue on failure and the fan-in, in another group, reconciles them. + const graph = JSON.parse(fs.readFileSync(layout.graphPath, "utf8")) as PlannedGraph; + graph.groups = { properties: { defaults: { failure_policy: "continue" } }, "property-catalog": {} }; + for (const candidate of graph.nodes) { + if (candidate.id === node.id) candidate.group = "property-catalog"; + else if (candidate.outputs.some((output) => output.contract === "ultrafuzz/property-lens@2")) { + candidate.group = "properties"; + } + } + fs.writeFileSync(layout.graphPath, JSON.stringify(graph), "utf8"); + const sealed = sealedFixtureAuthority(layout, node.id); + const lensDirs = ["property-specification-recon", "property-specification-crytic"].map((attemptId) => + getNodeArtifactDir(layout, attemptId) + ); + const task = { ...sealed.task, optionalDependencyArtifactDirs: lensDirs }; + const tasks = sealed.tasks.map((candidate) => (candidate.attemptId === task.attemptId ? task : candidate)); + writeSealedFixtureTaskAuthority(layout, graph.nodes, tasks); + const planned = (JSON.parse(fs.readFileSync(layout.graphPath, "utf8")) as PlannedGraph).nodes.find( + (candidate) => candidate.id === node.id + ); + assert.ok(planned); + const declared = task.dependencyArtifactDirs.map((directory) => path.basename(directory)); + const withoutFailedLens = { + task, + tasks, + admittedDependencyAttemptIds: declared.filter((attemptId) => attemptId !== "property-specification-crytic") + }; + const result = verifyRuntimeRequiredArtifactsForAttempt(layout, planned, node.id, withoutFailedLens); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); + + // The fan-in cannot recreate the failed lens's rows: a source it was not given is unknown. + const fabricated = JSON.stringify({ + schema_version: "ultrafuzz.properties.v2", + properties: [ + { + id: "property-1", + description: "Supply accounting remains consistent.", + category: "accounting", + priority: "high", + sources: [ + { source_node_id: "property-specification-recon", source_property_id: "recon-1" }, + { source_node_id: "property-specification-crytic", source_property_id: "crytic-1" } + ], + ledger_ids: ["evidence-1"] + } + ] + }); + writeArtifactFile(layout, node.id, "properties.json", fabricated); + writeArtifactFile(layout, node.id, "properties.md", fixtureCanonicalPropertiesMarkdown(fabricated)); + const fabricatedResult = verifyRuntimeRequiredArtifactsForAttempt(layout, planned, node.id, withoutFailedLens); + assert.deepEqual( + gateIssuePaths(fabricatedResult, "property-source-join"), + ["$.properties[0].sources[1]"], + JSON.stringify(fabricatedResult.diagnostics) + ); + + // Only the verifier's admission drops a lens: an admitted lens is still read and authenticated. + const admitted = verifyRuntimeRequiredArtifactsForAttempt(layout, planned, node.id, { + task, + tasks, + admittedDependencyAttemptIds: declared + }); + assert.ok( + admitted.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_LENS_AUTHORITY_INVALID"), + JSON.stringify(admitted.diagnostics) + ); +}); + for (const missingAuthority of ["manifest", "verification marker"] as const) { test(`property fan-in rejects a declared lens with a missing ${missingAuthority}`, () => { const layout = createRunLayout({ From ca6bf2b650837085d0d5452dd0b3dce7070c7b6f Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 06:49:38 +0000 Subject: [PATCH 03/10] feat(topology)!: a failed property lens no longer skips the rest of the campaign In the shipped topologies behind the default, low-cost, exhaustive and invariant-only audit profiles, the eight property lenses shared the halting `properties` group with the property fan-in. One lens that ended in a terminal failure failed the workflow: the fan-in was skipped and no strategy, stateful stage or review node ran, so the campaign ended without a report. The `properties` group now uses failure_policy: continue, and the fan-in moves to its own halting `property-catalog` group. With the previous two commits a failed lens is then an optional input to every later node: the fan-in consolidates the lenses that passed verification, and the strategies, specialists and review run as before. The fan-in itself still halts, so no strategy runs without a property catalog. Leaving the fan-in inside a continuing `properties` group would have been worse: its own output would then be optional to the strategies. The fan-in prompt now says its lens authority lists only verified lenses and that a missing lens's properties must not be recreated. The existing property-source-join gate rejects a source the fan-in was not given. The docs describe the general continue rule instead of implying review is the only consumer of partial results. BREAKING CHANGE: in the shipped topologies a failed property lens no longer fails the campaign. Its properties are absent from the catalog, the lens is reported as a failed node, and under best-effort completion the run can still succeed with a PARTIAL report. `--require-complete` runs still end unsuccessful. The fan-in node is in the new `property-catalog` group. Projects initialized earlier keep their copied .ultrafuzz/topology.yml, and the old behaviour, until they apply the same two group edits. Co-Authored-By: Claude Opus 5.5 --- .../property-specification-fanin.md | 20 +-- .ultrafuzz/topology.yml | 7 +- docs/explanation/campaigns.md | 3 +- docs/reference/artifacts-reports.md | 10 +- docs/reference/prompt-catalog-data.yml | 2 +- docs/reference/prompt-catalog.md | 2 +- docs/reference/topology-yaml.md | 11 +- packages/config/topologies/default.yml | 7 +- packages/config/topologies/exhaustive.yml | 7 +- packages/config/topologies/invariant-only.yml | 7 +- ...ithers-dependency-skip.integration.test.ts | 122 ++++++++++++++++++ .../topology/test/packaged-topologies.test.ts | 8 +- 12 files changed, 181 insertions(+), 25 deletions(-) diff --git a/.ultrafuzz/prompts/properties/property-specification-fanin.md b/.ultrafuzz/prompts/properties/property-specification-fanin.md index bf70e3bc1..4d37de6d5 100644 --- a/.ultrafuzz/prompts/properties/property-specification-fanin.md +++ b/.ultrafuzz/prompts/properties/property-specification-fanin.md @@ -26,13 +26,14 @@ Base test setup: ## 1. Consolidate -Consolidate properties from every topology-required lens JSON into a single +Consolidate properties from every lens JSON selected below into a single catalog. Each lens emits one `ultrafuzz/property-lens@2` output. Read and validate every selected lens JSON; it is the machine-readable source of truth. Use `{{schema_path}}/property-lens.schema.json` to validate each source catalog and assign every retained priority as `high`, `medium`, or `low`. -Sealed JSON authority for every declared ancestor property-lens artifact: +Sealed JSON authority for every ancestor property-lens artifact that passed +verification: {{ancestor_contract_artifact_authority:ultrafuzz/property-lens@2}} @@ -40,7 +41,10 @@ Read the manifest definition instead of expecting an expanded lens path or source array in this prompt. Use each selected producer task's exact `logical_node_id` as `source_node_id`, and retain the selector's required run-relative `localeCompare` order. Do not infer a source from a Markdown -companion, hard-coded lens list, or same-named workspace file. +companion, hard-coded lens list, or same-named workspace file. A lens that did +not pass verification is absent from this authority: consolidate the lenses it +selects, and do not recreate a missing lens's properties from its reference +material or from memory. Deduplicate equivalent properties across artifacts. When in doubt, err on the side of retaining multiple similar properties rather than risk removing one that represents a distinct concept or carries different meaning. @@ -66,11 +70,11 @@ property with every distinct contributing source in `sources`. Never keep only the first source. Canonical IDs only need to remain stable within this run, but all downstream artifacts must use them unchanged. -Coverage of the lens artifacts is total and machine-checked. Every property ID -in every lens artifact must appear exactly once across the whole catalog as a -source-node/property-ID pair in some canonical property's `sources`: the runtime rejects a lens ID that -appears in no canonical property and rejects the same pair listed on two -canonical properties. Deduplicating two rows therefore means listing both source +Coverage of the selected lens artifacts is total and machine-checked. Every +property ID in every selected lens artifact must appear exactly once across the +whole catalog as a source-node/property-ID pair in some canonical property's +`sources`: the runtime rejects a lens ID that appears in no canonical property +and rejects the same pair listed on two canonical properties. Deduplicating two rows therefore means listing both source pairs on the one merged canonical property, never dropping one. You may not drop a lens row because it duplicates wording inside its own lens, reads as non-testable, or looks out of scope; merge it into the canonical property it diff --git a/.ultrafuzz/topology.yml b/.ultrafuzz/topology.yml index c763c203a..4030baaa3 100644 --- a/.ultrafuzz/topology.yml +++ b/.ultrafuzz/topology.yml @@ -8,6 +8,11 @@ groups: properties: label: Properties color: "#a16207" + defaults: + failure_policy: continue + property-catalog: + label: Property catalog + color: "#854d0e" goals: label: Threat goals color: "#be123c" @@ -426,7 +431,7 @@ nodes: - id: property-specification-fanin kind: agentic prompt: properties/property-specification-fanin.md - group: properties + group: property-catalog depends_on: - property-specification-certora - property-specification-crytic diff --git a/docs/explanation/campaigns.md b/docs/explanation/campaigns.md index 7e718b210..f756d3b6c 100644 --- a/docs/explanation/campaigns.md +++ b/docs/explanation/campaigns.md @@ -39,7 +39,8 @@ phase; they are evidence to inspect, not an automatic security submission. The default scaffold is the direct bug-finding workflow: - Eight property-discovery lenses produce the shared property catalog and may - publish concrete findings discovered while deriving properties. + publish concrete findings discovered while deriving properties. A lens that + fails is left out of the catalog; the campaign continues with the others. - Twenty direct strategies investigate boundary, accounting, input, round-trip, workflow, time, state-machine, dependency, parity, lifecycle, and coverage-expansion hypotheses. diff --git a/docs/reference/artifacts-reports.md b/docs/reference/artifacts-reports.md index b7ee43ba6..2ea94695e 100644 --- a/docs/reference/artifacts-reports.md +++ b/docs/reference/artifacts-reports.md @@ -638,10 +638,12 @@ of production issues and, in bounded classification mode, the triage classification, lifecycle enrichment, and severity assessment of production issues. -New runs use `run.completion_policy = "best-effort"` by default. Stock strategy -groups continue after ordinary task failures. Independent work can finish; -work that needs a missing required result is skipped. Review uses successful -results that pass the existing input checks. User-authored topologies retain +New runs use `run.completion_policy = "best-effort"` by default. The stock +property-lens, goal, strategy, and specialist groups continue after ordinary +task failures. Independent work can finish; work that needs a missing required +result is skipped. Nodes that combine another group's results, such as the +property fan-in and review, use the successful results that pass the existing +input checks. User-authored topologies retain their declared failure policies. The configured attempts and time limits remain in effect; reporting does not restart analysis or add recovery attempts. diff --git a/docs/reference/prompt-catalog-data.yml b/docs/reference/prompt-catalog-data.yml index c60a473aa..b48caf72d 100644 --- a/docs/reference/prompt-catalog-data.yml +++ b/docs/reference/prompt-catalog-data.yml @@ -48,7 +48,7 @@ prompts: description: Derives target invariants and liveness properties using the Crytic reference. - path: properties/property-specification-fanin.md category: Property consolidation - description: Validates, merges, and deduplicates every property lens into one canonical property catalog. + description: Validates, merges, and deduplicates every verified property lens into one canonical property catalog. - path: properties/property-specification-runtime-verification.md category: Property design description: Derives target invariants using the Runtime Verification reference. diff --git a/docs/reference/prompt-catalog.md b/docs/reference/prompt-catalog.md index 7f112dc43..ddb058421 100644 --- a/docs/reference/prompt-catalog.md +++ b/docs/reference/prompt-catalog.md @@ -27,7 +27,7 @@ The default topology binds 48 agentic nodes to 47 distinct prompt files. [`strat | [`properties/aviggiano-lens.md`](../../.ultrafuzz/prompts/properties/aviggiano-lens.md) | Property design | Derives target properties using Antonio Viggiano's property reference. | | [`properties/0kn0t-lens.md`](../../.ultrafuzz/prompts/properties/0kn0t-lens.md) | Property design | Derives target properties using the 0kn0t reference. | | [`properties/josselin-feist-lens.md`](../../.ultrafuzz/prompts/properties/josselin-feist-lens.md) | Property design | Derives DeFi rounding properties using the Montyly rounding reference under the Josselin Feist lens identity. | -| [`properties/property-specification-fanin.md`](../../.ultrafuzz/prompts/properties/property-specification-fanin.md) | Property consolidation | Validates, merges, and deduplicates every property lens into one canonical property catalog. | +| [`properties/property-specification-fanin.md`](../../.ultrafuzz/prompts/properties/property-specification-fanin.md) | Property consolidation | Validates, merges, and deduplicates every verified property lens into one canonical property catalog. | | [`strategies/boundary-tests.md`](../../.ultrafuzz/prompts/strategies/boundary-tests.md) | Strategy | Turns properties and workflows into adversarial boundary recipes while investigating confirmed production bugs. | | [`strategies/encode-decode.md`](../../.ultrafuzz/prompts/strategies/encode-decode.md) | Strategy | Investigates encoding, decoding, parsing, serialization, and promised inverse relationships. | | [`strategies/differential-library-tests.md`](../../.ultrafuzz/prompts/strategies/differential-library-tests.md) | Strategy | Performs lightweight public-surface differential checks using independently justified comparators. | diff --git a/docs/reference/topology-yaml.md b/docs/reference/topology-yaml.md index eb3bb9548..a9da591b1 100644 --- a/docs/reference/topology-yaml.md +++ b/docs/reference/topology-yaml.md @@ -97,10 +97,13 @@ completed agent session whose required output is missing or schema-invalid; that post-agent contract failure is terminal. `failure_policy: continue` marks every node in that group as nonblocking. The -generated workflow waits for such a node to settle, consumes its artifacts only -when its verifier succeeded, and lets unrelated or downstream reconciliation -continue when it failed or was skipped. Use this only for optional specialist -lanes; ordinary groups retain fail-closed `halt` semantics. +generated workflow waits for such a node to settle and consumes its artifacts +only when its verifier succeeded. When it fails or is skipped, unrelated work +continues and nodes in other groups run with the results that did succeed, +while a node of the same group that depends on it is skipped, so a chain inside +one group never runs without its predecessor. The packaged topologies use this +for the property lenses, goals, strategies, and specialists; ordinary groups +retain fail-closed `halt` semantics. ## Node Fields diff --git a/packages/config/topologies/default.yml b/packages/config/topologies/default.yml index c763c203a..4030baaa3 100644 --- a/packages/config/topologies/default.yml +++ b/packages/config/topologies/default.yml @@ -8,6 +8,11 @@ groups: properties: label: Properties color: "#a16207" + defaults: + failure_policy: continue + property-catalog: + label: Property catalog + color: "#854d0e" goals: label: Threat goals color: "#be123c" @@ -426,7 +431,7 @@ nodes: - id: property-specification-fanin kind: agentic prompt: properties/property-specification-fanin.md - group: properties + group: property-catalog depends_on: - property-specification-certora - property-specification-crytic diff --git a/packages/config/topologies/exhaustive.yml b/packages/config/topologies/exhaustive.yml index 3dafaa7ec..87e2c3be6 100644 --- a/packages/config/topologies/exhaustive.yml +++ b/packages/config/topologies/exhaustive.yml @@ -8,6 +8,11 @@ groups: properties: label: Properties color: "#a16207" + defaults: + failure_policy: continue + property-catalog: + label: Property catalog + color: "#854d0e" goals: label: Threat goals color: "#be123c" @@ -426,7 +431,7 @@ nodes: - id: property-specification-fanin kind: agentic prompt: properties/property-specification-fanin.md - group: properties + group: property-catalog depends_on: - property-specification-certora - property-specification-crytic diff --git a/packages/config/topologies/invariant-only.yml b/packages/config/topologies/invariant-only.yml index efc7b0d7f..27c64a9c4 100644 --- a/packages/config/topologies/invariant-only.yml +++ b/packages/config/topologies/invariant-only.yml @@ -8,6 +8,11 @@ groups: properties: label: Properties color: "#a16207" + defaults: + failure_policy: continue + property-catalog: + label: Property catalog + color: "#854d0e" strategies: label: Strategies color: "#7c3aed" @@ -317,7 +322,7 @@ nodes: - id: property-specification-fanin kind: agentic prompt: properties/property-specification-fanin.md - group: properties + group: property-catalog depends_on: - property-specification-certora - property-specification-crytic diff --git a/packages/runtime/test/smithers-dependency-skip.integration.test.ts b/packages/runtime/test/smithers-dependency-skip.integration.test.ts index 624149f33..98c767d32 100644 --- a/packages/runtime/test/smithers-dependency-skip.integration.test.ts +++ b/packages/runtime/test/smithers-dependency-skip.integration.test.ts @@ -5,6 +5,11 @@ import path from "node:path"; import test from "node:test"; import { fileURLToPath } from "node:url"; +import { loadReferenceCatalog } from "@ultrafuzz/references"; + +import { initProject, planRun } from "../src/index.js"; +import { compileSmithersWorkflow, type CompiledSmithersWorkflow } from "../src/smithers.js"; +import { writeShippedDocumentReferenceCaches, writeShippedVulnerabilityDatabaseCache } from "./reference-fixtures.js"; import { temporaryRoot } from "./temporary-root.js"; interface Inspection { @@ -45,6 +50,123 @@ test("a dependency admission failure fails its preparation once while a transien ]); }); +// One failed property lens used to halt the whole default campaign: the lenses shared a halting +// group with the fan-in, and only the review group could consume a continuing producer's output. +test("a failed property lens leaves the packaged default fan-in, strategies and review to finish", async () => { + const compiled = await compilePackagedDefaultTopology(); + const lenses = compiled.tasks.filter((task) => + task.metadata.artifacts.outputs.some((output) => output.contract === "ultrafuzz/property-lens@2") + ); + const lensDirs = new Set(lenses.map((task) => task.artifactDir)); + assert.equal(lenses.length, 8); + for (const task of compiled.tasks) { + const optional = task.optionalDependencyArtifactDirs ?? []; + // No attempt requires a lens's output, so a failed lens fails no downstream input admission. + assert.deepEqual( + task.dependencyArtifactDirs.filter((directory) => lensDirs.has(directory) && !optional.includes(directory)), + [], + task.attemptId + ); + // Outside review nothing else became optional: the strategies still require the fan-in, and + // each stateful stage still requires the one before it. + if (task.metadata.node.group !== "review") { + assert.deepEqual( + optional.filter((directory) => !lensDirs.has(directory)), + [], + task.attemptId + ); + } + } + + const failing = "property-specification-a16z"; + assert.ok(lenses.some((task) => task.attemptId === failing)); + const { state } = await runSyntheticWorkflow("lens-failure", (root, template) => + compiledSchedulingWorkflowSource(root, template, compiled, failing) + ); + assert.equal(state(`verify:${failing}`), "failed"); + for (const task of compiled.tasks) { + if (task.attemptId !== failing) assert.equal(state(task.verifierSmithersNodeId), "finished", task.attemptId); + } +}); + +async function compilePackagedDefaultTopology(): Promise { + const project = temporaryRoot("ufz-lens-failure-project-"); + assert.equal(initProject({ projectRoot: project, force: true }).ok, true); + const xdgCacheHome = path.join(project, "xdg-cache"); + writeShippedDocumentReferenceCaches(xdgCacheHome, loadReferenceCatalog(project)); + writeShippedVulnerabilityDatabaseCache(xdgCacheHome); + const previousXdgCacheHome = process.env.XDG_CACHE_HOME; + process.env.XDG_CACHE_HOME = xdgCacheHome; + try { + const runId = "lens-failure"; + const plan = await planRun({ projectRoot: project, runId, env: {}, runtimeOverrides: { auditProfile: "default" } }); + assert.ok(plan.ok && plan.value, JSON.stringify(plan.diagnostics)); + return compileSmithersWorkflow({ + projectRoot: project, + config: plan.value.resolved_config, + graph: plan.value.expanded_graph, + runLayout: plan.value.layout, + workflowName: `ultrafuzz-${runId}`, + renderedPrompts: plan.value.rendered_prompts + }); + } finally { + if (previousXdgCacheHome === undefined) delete process.env.XDG_CACHE_HOME; + else process.env.XDG_CACHE_HOME = previousXdgCacheHome; + } +} + +/** + * Each compiled attempt becomes one engine task with the generated workflow's scheduling inputs: its + * verifier ID, its producers' verifier IDs, continue-on-failure for non-blocking attempts, and the + * template's own skip decision over required producers. Only `failing` throws. + */ +function compiledSchedulingWorkflowSource( + root: string, + template: string, + compiled: CompiledSmithersWorkflow, + failing: string +): string { + const compiledBaseTasks = compiled.tasks.map((task) => ({ + attemptId: task.attemptId, + verifierId: task.verifierSmithersNodeId, + dependsOn: task.dependencySmithersNodeIds, + dependencyArtifactDirs: task.dependencyArtifactDirs, + optionalDependencyArtifactDirs: task.optionalDependencyArtifactDirs ?? [], + metadata: { dependencies: task.metadata.dependencies } + })); + return `/** @jsxImportSource smthrs */ +import fs from "node:fs"; +import path from "node:path"; +import { createSmithers } from "smthrs"; +import { z } from "zod/v4"; +${templateSlice(template, "type WorkflowTaskStateContext =", "const agentPromptTemplate =")} +const compiledBaseTasks = ${JSON.stringify(compiledBaseTasks)}; +${templateSlice(template, "function dependencyVerificationProducersFromCompiledTask", "\n\nfunction dynamicExecutionMetadata")} +const nonBlocking = new Set(${JSON.stringify(compiled.nonBlockingAttemptIds)}); +const evidence = ${JSON.stringify(path.join(root, "executed.log"))}; +const { Workflow, Parallel, Task, smithers, outputs } = createSmithers({ + input: z.object({}), + result: z.object({ value: z.string() }) +}); +const record = (value: string) => { fs.appendFileSync(evidence, value + "\\n"); return { value }; }; +export default smithers((ctx) => + {compiledBaseTasks.map((task) => { + const required = dependencyVerificationProducersFromCompiledTask(task) + .filter((producer) => !producer.optional) + .map((producer) => producer.verifierId); + return + {() => { + if (task.attemptId === ${JSON.stringify(failing)}) throw new Error("synthetic lens failure"); + return record(task.attemptId); + }} + ; + })} +); +`; +} + async function runSyntheticWorkflow( name: string, workflowSource: (root: string, template: string) => string diff --git a/packages/topology/test/packaged-topologies.test.ts b/packages/topology/test/packaged-topologies.test.ts index 271aa89fd..57cf08747 100644 --- a/packages/topology/test/packaged-topologies.test.ts +++ b/packages/topology/test/packaged-topologies.test.ts @@ -142,11 +142,15 @@ function expandedTopology(topologyPath: string): ExpandedGraph { } describe("packaged topology collection", () => { - it.each(PACKAGED_TOPOLOGY_IDS)("continues independent strategies in %s without changing setup failures", (id) => { + it.each(PACKAGED_TOPOLOGY_IDS)("continues strategies and property lenses in %s but halts on setup failures", (id) => { const topology = loadedTopology(path.join(TOPOLOGY_ROOT, `${id}.yml`)); expect(topology.groups.strategies?.defaults?.failure_policy).toBe("continue"); expect(topology.groups.setup?.defaults?.failure_policy).toBeUndefined(); - expect(topology.groups.properties?.defaults?.failure_policy).toBeUndefined(); + // A failed lens leaves the fan-in, in its own halting group, to consolidate the other lenses. + const faninGroup = topology.nodes.find((node) => node.id === "property-specification-fanin")?.group; + expect(topology.groups.properties?.defaults?.failure_policy).toBe(id === "smoke" ? undefined : "continue"); + expect(faninGroup).toBe(id === "smoke" ? undefined : "property-catalog"); + expect(topology.groups["property-catalog"]?.defaults?.failure_policy).toBeUndefined(); const expanded = expandedTopology(path.join(TOPOLOGY_ROOT, `${id}.yml`)); expect(expanded.groups.strategies?.defaults?.failure_policy).toBe("continue"); expect(expanded.nodes.some((node) => node.group === "strategies")).toBe(true); From b326522b3473d25fa90465e198847b407b587d36 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 10:07:09 +0000 Subject: [PATCH 04/10] fix(runtime): a task keeps its result when an optional input it ran without is retried successfully With the property lenses continuing, a lens failure no longer ends the run. When an interrupted run is resumed with --retry-failed, which Modal's durable resume always passes, the failed lens reruns while the strategies already admitted without it keep running. Once the retried lens finalized as succeeded, host finalization failed each of those strategies with "finalized optional dependency is missing from verifier admission". dedupe-findings had admitted those strategies in the engine, so it then failed the inverse check, which halts review and fails the run over bookkeeping. Base has the same race for a retried strategy and the review tasks admitted without it. The check now accepts that omission when the producer's verifier finished after the consumer's preparation started, ordered by the Smithers event timestamps the sync pass already reads. It still rejects the omission of a producer that verified before the admission began, which is the deleted-marker case it exists for. The artifact gates already treat the consumer's recorded admission as the authority. Co-Authored-By: Claude Opus 5.5 --- docs/how-to/restart-continue.md | 7 +++ packages/runtime/src/workflow-sync.ts | 31 ++++++++-- packages/runtime/test/runtime.test.ts | 86 +++++++++++++++++++++++++++ 3 files changed, 120 insertions(+), 4 deletions(-) diff --git a/docs/how-to/restart-continue.md b/docs/how-to/restart-continue.md index 0e8c5b2ed..8f2e75c8d 100644 --- a/docs/how-to/restart-continue.md +++ b/docs/how-to/restart-continue.md @@ -82,6 +82,13 @@ the resumed workflow decides the skip again from the restored task states. It stays skipped while the prerequisite is still failed, and runs once a reset lets the prerequisite succeed. +Retrying a failed node of a `failure_policy: continue` group does not undo work +that already ran without it. A task that started before the retried node +verified keeps the result it produced without that node, and only tasks that +start afterwards read the new output. A retried artifact verifier is different: +resetting its agent producer also resets every node that started after that +producer's attempt, so those nodes run again. + ## Replay A Linked Run Use replay when you want the workflow engine to replay the linked run from the diff --git a/packages/runtime/src/workflow-sync.ts b/packages/runtime/src/workflow-sync.ts index 3caf878fa..788cba85a 100644 --- a/packages/runtime/src/workflow-sync.ts +++ b/packages/runtime/src/workflow-sync.ts @@ -2983,6 +2983,16 @@ async function synchronizeTasks(input: { }) ); const tasksByAttempt = new Map(input.tasks.map((task) => [task.attemptId, task])); + const nodeEvidence = (nodeId: string) => mergeNodeWorkflowEvidence(steps.get(nodeId), eventsByNode.get(nodeId) ?? []); + // Admission begins when the consumer's preparation task starts. + const verifiedAfterAdmissionOf = (consumer: StoredWorkflowTask) => { + const admittedAt = nodeEvidence(consumer.preparationSmithersNodeId)?.startedAt; + return (attemptId: string): boolean => { + const producer = tasksByAttempt.get(attemptId); + const verifiedAt = producer === undefined ? undefined : nodeEvidence(producer.verifierSmithersNodeId)?.finishedAt; + return admittedAt !== undefined && verifiedAt !== undefined && Date.parse(verifiedAt) > Date.parse(admittedAt); + }; + }; let syncedNodes = 0; let changed = initialStateChanged; @@ -3012,7 +3022,8 @@ async function synchronizeTasks(input: { evidence, evidenceSource: attemptEvidence.source, tasksByAttempt, - control: input.control + control: input.control, + verifiedAfterAdmission: verifiedAfterAdmissionOf(task) }) : { status: previousIsImmutable ? (previous?.status ?? evidence.status) : evidence.status, @@ -3235,6 +3246,8 @@ async function finalizeTerminalTask(input: { evidenceSource: "agent" | "verifier" | "preparation"; tasksByAttempt: Map; control: WorkflowSynchronizationControl; + /** Whether an optional producer verified after this task's admission began. */ + verifiedAfterAdmission: (attemptId: string) => boolean; }): Promise { if (input.evidence.status !== "succeeded") { // Preparation and verification wrappers both enforce the artifact contract @@ -3331,7 +3344,8 @@ async function finalizeTerminalTask(input: { input.layout, input.task, verifierAuthority.admittedDependencyAttemptIds, - input.tasksByAttempt + input.tasksByAttempt, + input.verifiedAfterAdmission ); prerequisiteManifestAuthority = capturePrerequisiteManifestAuthority( input.layout, @@ -3481,7 +3495,8 @@ async function finalizeTerminalTask(input: { input.layout, input.task, verifierAuthority.admittedDependencyAttemptIds, - input.tasksByAttempt + input.tasksByAttempt, + input.verifiedAfterAdmission ); const currentPrerequisiteManifestAuthority = capturePrerequisiteManifestAuthority( input.layout, @@ -3685,12 +3700,19 @@ function captureReferenceManifestDigest( * must persist it in the admitted closure. This prevents a deleted, dangling, * or malformed optional marker from turning a previously admitted success into * an unauthenticated omission during a later synchronization pass. + * + * The one exception is a producer that verified only after the consumer's + * admission began, which happens when `resume --retry-failed` reruns a failed + * producer while its consumers keep running. The consumer ran without it, which + * is the omission its verifier recorded, so failing it would stop the run over + * bookkeeping. */ function assertOptionalDependencyAuthoritiesCurrent( layout: RunLayout, task: StoredWorkflowTask, admittedDependencyAttemptIds: readonly string[], - tasksByAttempt: ReadonlyMap + tasksByAttempt: ReadonlyMap, + verifiedAfterAdmission: (attemptId: string) => boolean ): void { const optionalArtifactDirs = task.optionalDependencyArtifactDirs ?? []; if (optionalArtifactDirs.length === 0) return; @@ -3711,6 +3733,7 @@ function assertOptionalDependencyAuthoritiesCurrent( const dependencyStatus = state.nodes[attemptId]?.status; const finalizedSuccess = dependencyStatus !== undefined && NODE_RECOVERED_STATUSES.has(dependencyStatus); + if (finalizedSuccess && !admitted.has(attemptId) && verifiedAfterAdmission(attemptId)) continue; if (finalizedSuccess !== admitted.has(attemptId)) { throw new Error( finalizedSuccess diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index cf84b76f0..9a73680d0 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -22311,6 +22311,92 @@ for (const markerAuthority of ["malformed leaf", "dangling leaf", "symlinked roo }); } +// `resume --retry-failed` reruns a failed optional producer while consumers that were already +// admitted without it keep running. Only a producer that verified before the admission began +// shows the admission lost a success. +for (const retryVerified of ["after", "before"] as const) { + test(`syncRun ${retryVerified === "after" ? "keeps" : "rejects"} a consumer admitted without an optional prerequisite whose retry verified ${retryVerified} the admission`, async () => { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + writeOptionalSpecialistTopology(project); + const workflowRunId = `ultrafuzz-sync-optional-retry-${retryVerified}`; + const runId = `sync-optional-retry-${retryVerified}`; + const failedEvents = [ + { type: "NodeFinished", nodeId: "node:direct-strategy", attempt: 1 }, + { type: "NodeFailed", nodeId: "node:optional-specialist", attempt: 1 } + ]; + const failedEnv = fakeLifecycleSmithersEnv(project, { + inspect: workflowInspect({ + workflowRunId, + status: "running", + steps: [ + { id: "node:direct-strategy", state: "finished", attempt: 1 }, + { id: "node:optional-specialist", state: "failed", attempt: 1 }, + { id: "node:final-report", state: "pending", attempt: 0 } + ] + }), + events: workflowEvents(workflowRunId, failedEvents) + }); + const run = await startRun({ projectRoot: project, runId, env: failedEnv }); + assert.equal(run.ok, true, JSON.stringify(run.diagnostics)); + assert.ok(run.value); + const runRoot = run.value.run_root; + writeRequiredArtifactSet(runRoot, "direct-strategy", [GENERIC_RUNTIME_MARKDOWN_PATH]); + assert.equal((await syncRun({ projectRoot: project, runId, env: failedEnv })).ok, true); + const nodeStatus = (nodeId: string) => + ( + JSON.parse(fs.readFileSync(path.join(runRoot, "state.json"), "utf8")) as { + nodes: Record; + } + ).nodes[nodeId]?.status; + assert.equal(nodeStatus("optional-specialist"), "failed"); + + // The report's verifier records an admission made while the specialist had no marker. + writeRequiredArtifactSet(runRoot, "final-report", [GENERIC_RUNTIME_MARKDOWN_PATH]); + writeRequiredArtifactSet(runRoot, "optional-specialist", [GENERIC_RUNTIME_MARKDOWN_PATH]); + const admission = [{ type: "NodeStarted", nodeId: "prepare:final-report", attempt: 1 }]; + const retry = [ + { type: "NodeStarted", nodeId: "node:optional-specialist", attempt: 2 }, + { type: "NodeFinished", nodeId: "node:optional-specialist", attempt: 2 }, + { type: "NodeFinished", nodeId: "verify:optional-specialist", attempt: 1 } + ]; + const finalEnv = fakeLifecycleSmithersEnv(project, { + inspect: workflowInspect({ + workflowRunId, + steps: [ + { id: "node:direct-strategy", state: "finished", attempt: 1 }, + { id: "node:optional-specialist", state: "finished", attempt: 2 }, + { id: "node:final-report", state: "finished", attempt: 1 } + ] + }), + events: workflowEvents(workflowRunId, [ + ...failedEvents, + ...(retryVerified === "after" ? [...admission, ...retry] : [...retry, ...admission]), + { type: "NodeFinished", nodeId: "node:final-report", attempt: 1 }, + { type: "RunFinished" } + ]) + }); + const sync = await syncRun({ projectRoot: project, runId, env: finalEnv }); + + assert.equal(sync.ok, true, JSON.stringify(sync.diagnostics)); + assert.equal(nodeStatus("optional-specialist"), "succeeded"); + if (retryVerified === "after") { + assert.equal(nodeStatus("final-report"), "succeeded", JSON.stringify(sync.diagnostics)); + assert.equal(sync.value?.status, "succeeded", JSON.stringify(sync.diagnostics)); + } else { + assert.equal(nodeStatus("final-report"), "failed"); + assert.ok( + sync.diagnostics.some( + (diagnostic) => + diagnostic.code === "ARTIFACT_VERIFICATION_AUTHORITY_INVALID" && + diagnostic.message.includes("finalized optional dependency is missing from verifier admission") + ), + JSON.stringify(sync.diagnostics) + ); + } + }); +} + test("syncRun binds an optional prerequisite digest before a final-boundary manifest swap", async () => { const project = tempProject(); initProject({ projectRoot: project, force: true }); From 9a7da73e1d92b31ba60365fb667c0998798d9cad Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 09:20:24 +0000 Subject: [PATCH 05/10] docs(topology): describe what continue does to a same-group chain reached through another group A same-group ancestor reached only through another group does not skip its dependent: the intermediate node runs, and the dependent fails its input admission instead. The topology reference now says so, and no longer calls continue a policy for optional branches. The reconcilesPartialResults comment is scoped to chains within one group, and the dependency-policy test names its reconciling node `join` so the old `review` name does not look significant. Co-Authored-By: Claude Opus 5.5 --- docs/reference/topology-yaml.md | 25 ++++++++++--------- packages/runtime/src/dynamic-runtime.ts | 8 +++--- .../test/workflow-dependency-policy.test.ts | 10 ++++---- 3 files changed, 22 insertions(+), 21 deletions(-) diff --git a/docs/reference/topology-yaml.md b/docs/reference/topology-yaml.md index a9da591b1..57d822937 100644 --- a/docs/reference/topology-yaml.md +++ b/docs/reference/topology-yaml.md @@ -75,13 +75,13 @@ colors such as `#7c3aed`. Group defaults may include: -| Field | Meaning | -| ----------------- | ----------------------------------------------------- | -| `loops` | Default loop count for nodes in the group. | -| `timeout_seconds` | Default timeout for nodes in the group. | -| `max_attempts` | Maximum attempts for an agent task. | -| `model_profiles` | Explicit model profile list for nodes in the group. | -| `failure_policy` | `halt` (default) or `continue` for optional branches. | +| Field | Meaning | +| ----------------- | -------------------------------------------------------- | +| `loops` | Default loop count for nodes in the group. | +| `timeout_seconds` | Default timeout for nodes in the group. | +| `max_attempts` | Maximum attempts for an agent task. | +| `model_profiles` | Explicit model profile list for nodes in the group. | +| `failure_policy` | `halt` (default) or `continue` (nonblocking; see below). | Node fields override group defaults. A node or group `model_profiles` list is the model fan-out surface. When neither a node nor its group selects model @@ -99,11 +99,12 @@ that post-agent contract failure is terminal. `failure_policy: continue` marks every node in that group as nonblocking. The generated workflow waits for such a node to settle and consumes its artifacts only when its verifier succeeded. When it fails or is skipped, unrelated work -continues and nodes in other groups run with the results that did succeed, -while a node of the same group that depends on it is skipped, so a chain inside -one group never runs without its predecessor. The packaged topologies use this -for the property lenses, goals, strategies, and specialists; ordinary groups -retain fail-closed `halt` semantics. +continues and nodes in other groups run with the results that did succeed. A +node of the same group never runs without it: a node that depends on it +directly, or through nodes of the same group, is skipped, and one that reaches +it only through another group fails its input admission instead. The packaged +topologies use this for the property lenses, goals, strategies, and +specialists; ordinary groups retain fail-closed `halt` semantics. ## Node Fields diff --git a/packages/runtime/src/dynamic-runtime.ts b/packages/runtime/src/dynamic-runtime.ts index 4741c1522..773b04d7c 100644 --- a/packages/runtime/src/dynamic-runtime.ts +++ b/packages/runtime/src/dynamic-runtime.ts @@ -334,10 +334,10 @@ function instantiateDynamicGraphNode(input: { /** * Whether `consumer` treats the output of a continuing producer in `producerGroup` as optional. * Consumers outside that group reconcile whichever results succeeded (the review group, and the - * property fan-in reading its lenses). Inside the group a chain keeps its required inputs, so a - * strategy or stateful stage never runs without its predecessor's output (#1120). The compiler and - * dynamic lowering share this one rule; the task-manifest gate checks only that optional inputs - * come from continuing producers. + * property fan-in reading its lenses). Inside the group the output stays required, so a chain within + * one group, such as a stateful stage after its setup, never runs without its predecessor (#1120). + * The compiler and dynamic lowering share this one rule; the task-manifest gate checks only that + * optional inputs come from continuing producers. */ export function reconcilesPartialResults( consumer: Pick, diff --git a/packages/runtime/test/workflow-dependency-policy.test.ts b/packages/runtime/test/workflow-dependency-policy.test.ts index 7979c7790..8c8d48402 100644 --- a/packages/runtime/test/workflow-dependency-policy.test.ts +++ b/packages/runtime/test/workflow-dependency-policy.test.ts @@ -55,14 +55,14 @@ test("workflow dependency policy keeps inputs inside a continuing group required { id: "producer", group: "strategies", depends_on: ["__start__"] }, { id: "dependent", group: "strategies", depends_on: ["producer"] }, { id: "independent", group: "strategies", depends_on: ["__start__"] }, - { id: "review", group: "catalog", depends_on: ["dependent", "independent"] } + { id: "join", group: "catalog", depends_on: ["dependent", "independent"] } ].map((node) => ({ ...node, kind: "agentic" as const, prompt: "fixture.md", outputs: [{ path: "result.md", contract: "ultrafuzz/nonempty-markdown@1" as const, primary: true }] })), - { id: "__finish__", kind: "meta", role: "finish", depends_on: ["review"] } + { id: "__finish__", kind: "meta", role: "finish", depends_on: ["join"] } ] }; const config = createDefaultResolvedConfig(); @@ -82,12 +82,12 @@ test("workflow dependency policy keeps inputs inside a continuing group required assert.deepEqual(compiled.nonBlockingAttemptIds, ["dependent", "independent", "producer"]); const dependent = compiled.tasks.find((task) => task.attemptId === "dependent"); const independent = compiled.tasks.find((task) => task.attemptId === "independent"); - const review = compiled.tasks.find((task) => task.attemptId === "review"); - assert.ok(dependent && independent && review); + const join = compiled.tasks.find((task) => task.attemptId === "join"); + assert.ok(dependent && independent && join); assert.deepEqual(dependent.dependencySmithersNodeIds, ["verify:producer"]); assert.deepEqual(dependent.optionalDependencyArtifactDirs, []); assert.deepEqual(independent.dependencySmithersNodeIds, []); - assert.deepEqual(review.optionalDependencyArtifactDirs?.map((directory) => path.basename(directory)).sort(), [ + assert.deepEqual(join.optionalDependencyArtifactDirs?.map((directory) => path.basename(directory)).sort(), [ "dependent", "independent", "producer" From 7d6e063207e84b7bca81a89b6606000e35cc22e8 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Wed, 30 Sep 2026 07:43:56 +0000 Subject: [PATCH 06/10] docs(changelog): record that a failed property lens no longer stops the campaign The packaged exhaustive and invariant-only topologies and new project scaffolds change on upgrade, while an existing project's .ultrafuzz/topology.yml keeps the old halting properties group until it is edited. The breaking-changes entry names the three edits and the prompt refresh, and states the general rule for custom topologies: a continuing group's results are optional to every other group, not only to review. The other-changes entry records the resume --retry-failed fix for a task admitted without an optional input that is later rerun successfully. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index c787fba40..8139a8efa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ ### Breaking changes +- **[config] [runtime] [prompts] [docs]** A property lens that fails no longer stops the campaign in the packaged `exhaustive` and `invariant-only` topologies or in projects initialized by this release. Their `properties` group now has `failure_policy: continue`, and `property-specification-fanin` moves to a new `property-catalog` group, which still halts: the fan-in consolidates the lenses that passed verification, and the strategies, specialists and review run without the failed lens's properties. The report is PARTIAL when the lens ran out of attempts, and unverified when it failed its output contract. In `exhaustive`, `dynamic-strategy-generator` now also runs when some strategy attempts failed, using the ones that succeeded, instead of being skipped. An existing project keeps the old behaviour under the `default` and `low-cost` profiles, where one failed lens fails the run before any strategy starts and leaves it without a report, until you make three edits to its `.ultrafuzz/topology.yml`: add `defaults: {failure_policy: continue}` to the `properties` group; add a `property-catalog` group with no `failure_policy` (the scaffold gives it `label: Property catalog` and `color: "#854d0e"`); and change the `property-specification-fanin` node's `group` from `properties` to `property-catalog`. Then delete `.ultrafuzz/prompts/properties/property-specification-fanin.md` and rerun `ultrafuzz init` to take the prompt that has the fan-in consolidate only the lenses it is given. The rule behind this changes for custom topologies too: a `failure_policy: continue` group's results are now optional to nodes in every other group, not only to the group named `review`, so such a node runs without a failed input instead of being skipped, while nodes in the same group still require them. To keep a chain strict, put all of it in one group. - **[runtime] [docs]** A final-report producer retry or verifier that runs in a restarted controller (resume, quota park, supervisor relaunch) now rebuilds `report.json#run_metadata.agent_execution` from `smithers/final-report-selections/.json` in the run directory, which each producer attempt writes just before starting its agent, instead of running `smithers node` as a subprocess (a `--full-output` read of up to 64 MiB under a 180 s budget). This applies to workflows rendered by this release, at launch or by `resume --refresh-controller`; plain `resume` keeps running the workflow persisted at launch, even after a refresh. On a run launched by an earlier release, a verifier refreshed after its producer succeeded under the old workflow fails with `report producer selection was never recorded`, and mixing plain and refreshed resumes can make it fail with `run_metadata.agent_execution differs from the controller-observed producer`; in either case run `ultrafuzz resume --refresh-controller --reset-node ` (`node:final-report` in the packaged topologies), which reruns the report producer and its verifier rather than every failed task. Producer attempts after such a refresh leave the earlier release's attempts out of `agent_execution.failed_attempts`. After a restart the verifier compares against a run-directory file instead of the Smithers database. The reference docs now say that neither is a boundary against an agent running unsandboxed as the same user, which retires #585's claim that the run filesystem cannot be used to forge the producer (Refs #1143). - **[config] [runtime] [modal] [docs]** Removes per-node cloud execution (#134), which never completed a node in any release (#712, #713). `validate`, `doctor`, `run` and `eval run` reject `[execution] mode = "cloud"`, `execution.provider` and `[execution.providers.modal]` with `CONFIG_EXECUTION_CLOUD_REMOVED` before a run directory exists; remove `execution.provider` and `[execution.providers.modal]`, and set `mode = "local"` or delete it. The rest of `[execution]` is still accepted and validated but no longer affects execution: every agentic attempt runs locally. To use Modal, run the whole campaign inside one sandbox, as the unchanged `ultrafuzz-modal` eval runner does. A run planned with `mode = "cloud"` cannot be continued: `resume`, with or without `--refresh-controller`, reads the mode from the run's `smithers/resolved-config.json` and fails with `WORKFLOW_CLOUD_EXECUTION_REMOVED` before starting Smithers, so start a new run. It also fails, with `WORKFLOW_LIFECYCLE_FAILED`, for a run whose resolved config is missing or does not parse, which a plain `resume` used to continue without it. `ultrafuzz clean` no longer terminates Modal sandboxes tagged `purpose=ultrafuzz-node` or deletes the `ultrafuzz-node-*` volumes of earlier cloud attempts; remove those by hand in the Modal app the run used, which its `plan.json` records under `execution.providers.modal` until `ultrafuzz clean` deletes the run. The runtime `cloud-execution-generation` schema and the seven Modal node schemas are deleted, so `json validate` no longer recognizes them. - **[runtime] [docs]** A run launched by an earlier release whose dynamic groups have already expanded cannot be synchronized, resumed or reported after upgrading. Each synchronization re-renders the run's published prompts with the current code and compares the bytes, and this release changes the coverage text the final report prompt embeds (#1176) and the order of multi-path authority selectors (#1195), so such a run fails with `runtime rendered prompt changed`. Finish or cancel it before upgrading; `pause` and `cancel` still work on it. @@ -24,6 +25,7 @@ ### Other changes +- **[runtime] [docs]** A task that started without a failed optional input, such as a review task without a failed strategy, keeps its result when `resume --retry-failed` reruns that input and the rerun succeeds before the task's own result is synchronized. Before, the task failed with "finalized optional dependency is missing from verifier admission", which could halt review and fail the run. - **[runtime] [cli] [docs]** Production copies of `brace-expansion` and `undici` move past the High advisories published on 2026-09-29 (GHSA-6j4f-fj2g-mc7p and GHSA-qhr7-859c-m2p7 for `brace-expansion`, GHSA-rfgv-xxqx-mfg5 for `undici`) and the Moderate and Low advisories fixed alongside them (GHSA-q2hr-2g5m-vwhr, GHSA-3wwx-pv8p-q78v, GHSA-r53p-7pc4-xj5r). The CLI's two `brace-expansion` copies, both pulled in by `@oclif/core` through `minimatch`, move from 2.1.4 and 5.0.9 to 2.1.7 and 5.0.12. No npm release bundles these fixes yet, so the `npm@11.19.0` patch now also carries `brace-expansion` 5.0.12 and `undici` 6.28.1 inside the operator npm, and the operator npm's pinned closure digest changes with it. - **[artifacts]** Concurrent ultrafuzz commands, such as `ultrafuzz status --watch` while `ultrafuzz cancel` runs, no longer lose or tear records in a run's `events.jsonl`, `usage.jsonl` or `attempts.jsonl`, or in the `.ultrafuzz/` audit journals. Each append holds a short-lived `.lock` next to the journal; a lock left by a killed process on the same host is taken over at once, and one that cannot be checked after 10 seconds. - **[runtime] [docs]** When `resume --retry-failed` retries a dynamic source whose verifier failed, it now also withdraws the rendered prompts of later nodes that wait on the group, so they render again from the new expansion. Before, if the retried source planned different items, a node whose prompt names the group's children (for example with `{{artifact_path:}}`) failed every render with `runtime rendered prompt changed for `. The packaged topologies were not affected (#1141). From 2005840aa7fc282e634e83f2789fe36fe801a1ba Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Wed, 30 Sep 2026 09:38:55 +0000 Subject: [PATCH 07/10] fix(runtime): a task synchronized while its optional input is still retrying keeps its result `resume --retry-failed` reruns a failed continuing producer, such as a property lens, while the tasks admitted without it keep running. A sync pass during that rerun (Modal's `waitForTerminalRun` runs `ultrafuzz inspect` every 60 s) failed each of those tasks with "optional dependency is not terminal for verifier admission". That failure recovered on a later pass, but a dependent whose host gate reads the task's output, such as triage after dedupe, failed its gate in the same pass, recorded `terminal_disposition: task-output-validation-failure`, which is immutable, and left the run `failed` for good. `assertOptionalDependencyAuthoritiesCurrent` now accepts the omission of an optional producer that is not terminal. Tasks synchronize in dependency order and start only after their producers settle, so such a producer is being rerun and had no verified output when the task was admitted: if the rerun fails, the omission matches, and if it succeeds, it verified after the admission, which the retry exception added earlier in this PR accepts. A producer with no recorded state still fails the check, and so does a finalized success missing from an admission that began after it verified. A stricter rule that accepts only a rerun started after the consumer's admission would not fix the common order, where the resume starts the rerun first and the consumers are admitted while it runs. The new sync test reproduces the review finding: lens (continue) -> catalog -> dedupe (findings@2) -> triage (triaged-findings@1), with one sync while the lens is rerunning. Before this change triage ends failed and the run fails. Co-Authored-By: Claude Opus 5.5 --- packages/runtime/src/workflow-sync.ts | 16 ++- packages/runtime/test/runtime.test.ts | 182 ++++++++++++++++++++++++++ 2 files changed, 191 insertions(+), 7 deletions(-) diff --git a/packages/runtime/src/workflow-sync.ts b/packages/runtime/src/workflow-sync.ts index 788cba85a..677bc7994 100644 --- a/packages/runtime/src/workflow-sync.ts +++ b/packages/runtime/src/workflow-sync.ts @@ -3701,11 +3701,11 @@ function captureReferenceManifestDigest( * or malformed optional marker from turning a previously admitted success into * an unauthenticated omission during a later synchronization pass. * - * The one exception is a producer that verified only after the consumer's - * admission began, which happens when `resume --retry-failed` reruns a failed - * producer while its consumers keep running. The consumer ran without it, which - * is the omission its verifier recorded, so failing it would stop the run over - * bookkeeping. + * The one exception is a producer that `resume --retry-failed` reruns while its + * consumers keep running. A consumer admitted without it ran without it, which + * is the omission its verifier recorded, so failing the consumer would stop the + * run over bookkeeping. The omission stands while the rerun is in progress, and + * after it when the rerun verified only after the consumer's admission began. */ function assertOptionalDependencyAuthoritiesCurrent( layout: RunLayout, @@ -3742,8 +3742,10 @@ function assertOptionalDependencyAuthoritiesCurrent( ); } if (!finalizedSuccess) { - if (dependencyStatus === undefined || !terminalStatus(dependencyStatus)) { - throw new Error(`optional dependency is not terminal for verifier admission ${attemptId}`); + // Tasks synchronize in dependency order and start only after their producers settle, so a + // producer that is not terminal here is being rerun and was unverified when this task was admitted. + if (dependencyStatus === undefined) { + throw new Error(`optional dependency has no recorded state for verifier admission ${attemptId}`); } continue; } diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index 9a73680d0..054060564 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -22397,6 +22397,188 @@ for (const retryVerified of ["after", "before"] as const) { }); } +/** A continuing lens, the halting catalog after it, and two review stages whose host gates chain. */ +function writeRetriedLensReviewProject(): string { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + const markdownOutput = ` + - path: ${GENERIC_RUNTIME_MARKDOWN_PATH} + contract: ultrafuzz/nonempty-markdown@1 + primary: true`; + fs.writeFileSync( + path.join(project, ".ultrafuzz", "topology.yml"), + `version: 2 +defaults: + strategy_loops: 1 +groups: + properties: + label: Properties + defaults: + failure_policy: continue + property-catalog: + label: Property catalog + review: + label: Review +nodes: + - id: __start__ + kind: meta + role: start + depends_on: [] + - id: lens + kind: agentic + prompt: setup/runtime-fixture.md + group: properties + depends_on: [__start__] + outputs:${markdownOutput} + - id: catalog + kind: agentic + prompt: setup/runtime-fixture.md + group: property-catalog + depends_on: [lens] + outputs:${markdownOutput} + - id: dedupe + kind: agentic + prompt: setup/runtime-fixture.md + group: review + depends_on: [catalog] + outputs: + - path: findings.json + contract: ultrafuzz/findings@2 + primary: true + - path: finding-lifecycle-ledger.json + contract: ultrafuzz/finding-lifecycle-ledger@1 + - id: triage + kind: agentic + prompt: setup/runtime-fixture.md + group: review + depends_on: [dedupe] + outputs: + - path: triaged-findings.json + contract: ultrafuzz/triaged-findings@1 + primary: true + - path: finding-lifecycle-ledger.json + contract: ultrafuzz/finding-lifecycle-ledger@1 + - id: __finish__ + kind: meta + role: finish + depends_on: [triage] +`, + "utf8" + ); + writeNeutralRuntimeFixturePrompt(project); + fs.appendFileSync( + path.join(project, ".ultrafuzz", "prompts", "setup", "runtime-fixture.md"), + REPORT_VOCABULARY_PROMPT_REFERENCES, + "utf8" + ); + return project; +} + +// A sync pass can run while `resume --retry-failed` still reruns a failed optional producer, as +// Modal's `ultrafuzz inspect` poll does. A consumer that ran without the producer must finalize in +// that pass too: while it reads failed, the host gate of a dependent that reads its outputs fails, +// and that output-validation failure is final. +test("syncRun finalizes consumers admitted without an optional prerequisite whose retry is still running", async () => { + const project = writeRetriedLensReviewProject(); + const workflowRunId = "ultrafuzz-sync-optional-retry-running"; + const runId = "sync-optional-retry-running"; + const failedEvents = [ + { type: "NodeFailed", nodeId: "node:lens", attempt: 1 }, + { type: "NodeStarted", nodeId: "prepare:catalog", attempt: 1 }, + { type: "NodeFinished", nodeId: "node:catalog", attempt: 1 } + ]; + const failedEnv = fakeLifecycleSmithersEnv(project, { + inspect: workflowInspect({ + workflowRunId, + status: "running", + steps: [ + { id: "node:lens", state: "failed", attempt: 1 }, + { id: "node:catalog", state: "finished", attempt: 1 }, + { id: "node:dedupe", state: "pending", attempt: 0 }, + { id: "node:triage", state: "pending", attempt: 0 } + ] + }), + events: workflowEvents(workflowRunId, failedEvents) + }); + const run = await startRun({ projectRoot: project, runId, env: failedEnv }); + assert.equal(run.ok, true, JSON.stringify(run.diagnostics)); + assert.ok(run.value); + const runRoot = run.value.run_root; + const writeArtifacts = (attemptId: string, files: Record) => { + for (const [relative, contents] of Object.entries(files)) { + const filePath = path.join(runRoot, "artifacts", attemptId, ...relative.split("/")); + fs.mkdirSync(path.dirname(filePath), { recursive: true }); + fs.writeFileSync(filePath, contents, "utf8"); + } + writeCurrentArtifactVerificationMarker(runRoot, attemptId); + }; + const emptyLedger = JSON.stringify({ schema_version: "ultrafuzz.finding-lifecycle-ledger.v1", records: [] }); + const nodeStatus = (nodeId: string) => + ( + JSON.parse(fs.readFileSync(path.join(runRoot, "state.json"), "utf8")) as { + nodes: Record; + } + ).nodes[nodeId]?.status; + writeArtifacts("catalog", { [GENERIC_RUNTIME_MARKDOWN_PATH]: "catalog without the lens\n" }); + assert.equal((await syncRun({ projectRoot: project, runId, env: failedEnv })).ok, true); + assert.equal(nodeStatus("lens"), "failed"); + assert.equal(nodeStatus("catalog"), "succeeded"); + + // The retried lens is still running when dedupe and triage finish without it. + writeArtifacts("dedupe", { "findings.json": "[]", "finding-lifecycle-ledger.json": emptyLedger }); + writeArtifacts("triage", { "triaged-findings.json": "[]", "finding-lifecycle-ledger.json": emptyLedger }); + const retryEvents = [ + ...failedEvents, + { type: "NodeStarted", nodeId: "node:lens", attempt: 2 }, + { type: "NodeStarted", nodeId: "prepare:dedupe", attempt: 1 }, + { type: "NodeFinished", nodeId: "node:dedupe", attempt: 1 }, + { type: "NodeStarted", nodeId: "prepare:triage", attempt: 1 }, + { type: "NodeFinished", nodeId: "node:triage", attempt: 1 } + ]; + const retryEnv = fakeLifecycleSmithersEnv(project, { + inspect: workflowInspect({ + workflowRunId, + status: "running", + steps: [ + { id: "node:lens", state: "in-progress", attempt: 2 }, + { id: "node:catalog", state: "finished", attempt: 1 }, + { id: "node:dedupe", state: "finished", attempt: 1 }, + { id: "node:triage", state: "finished", attempt: 1 } + ] + }), + events: workflowEvents(workflowRunId, retryEvents) + }); + const retrySync = await syncRun({ projectRoot: project, runId, env: retryEnv }); + assert.equal(retrySync.ok, true, JSON.stringify(retrySync.diagnostics)); + assert.equal(nodeStatus("dedupe"), "succeeded", JSON.stringify(retrySync.diagnostics)); + + writeArtifacts("lens", { [GENERIC_RUNTIME_MARKDOWN_PATH]: "lens retried\n" }); + const finalEnv = fakeLifecycleSmithersEnv(project, { + inspect: workflowInspect({ + workflowRunId, + steps: [ + { id: "node:lens", state: "finished", attempt: 2 }, + { id: "node:catalog", state: "finished", attempt: 1 }, + { id: "node:dedupe", state: "finished", attempt: 1 }, + { id: "node:triage", state: "finished", attempt: 1 } + ] + }), + events: workflowEvents(workflowRunId, [ + ...retryEvents, + { type: "NodeFinished", nodeId: "node:lens", attempt: 2 }, + { type: "NodeFinished", nodeId: "verify:lens", attempt: 1 }, + { type: "RunFinished" } + ]) + }); + const sync = await syncRun({ projectRoot: project, runId, env: finalEnv }); + + assert.equal(sync.ok, true, JSON.stringify(sync.diagnostics)); + for (const nodeId of ["lens", "catalog", "dedupe", "triage"]) { + assert.equal(nodeStatus(nodeId), "succeeded", `${nodeId}: ${JSON.stringify(sync.diagnostics)}`); + } + assert.equal(sync.value?.status, "succeeded", JSON.stringify(sync.diagnostics)); +}); + test("syncRun binds an optional prerequisite digest before a final-boundary manifest swap", async () => { const project = tempProject(); initProject({ projectRoot: project, force: true }); From 3847aad8e7fbf9df23e01aca9d5ac5fd07a165ed Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Wed, 30 Sep 2026 09:38:29 +0000 Subject: [PATCH 08/10] fix(runtime): dynamic group templates follow the optional-input rule `compileSmithersWorkflow` computed `optionalDependencyArtifactDirs` only for static tasks, so a dynamic group's task templates, and the children cloned from them, kept every inherited ancestor required. In a custom topology where a continuing producer in another group is an ancestor of a dynamic group, a failed producer let a static sibling run but failed every generated child's input admission, an `artifact-contract` failure that leaves the report unverified. That contradicted the rule the topology reference and changelog state, and main had the same gap for a dynamic group in `review`. The compiler now applies the same filter to the templates. A template without optional inputs keeps its bytes, so the packaged topologies, whose dynamic goal groups have only setup ancestors, compile exactly as before. Generated children remap their inherited optional directories the way they already remap `dependencyArtifactDirs`, so the two stay consistent in a relocated project. A run launched before this change keeps its sealed templates, so its dynamic lowering re-derives unchanged. Co-Authored-By: Claude Opus 5.5 --- packages/runtime/src/dynamic-runtime.ts | 7 + packages/runtime/src/smithers.ts | 16 +- .../runtime/test/dynamic-workflow.test.ts | 138 ++++++++++++++++++ 3 files changed, 157 insertions(+), 4 deletions(-) diff --git a/packages/runtime/src/dynamic-runtime.ts b/packages/runtime/src/dynamic-runtime.ts index 773b04d7c..83700e354 100644 --- a/packages/runtime/src/dynamic-runtime.ts +++ b/packages/runtime/src/dynamic-runtime.ts @@ -270,6 +270,13 @@ function instantiateDynamicTasks(input: { dependencyArtifactDirs: template.dependencyArtifactDirs.map((directory) => remapProjectPath(directory, input.group.promptContext.projectRoot, input.projectRoot) ), + ...(template.optionalDependencyArtifactDirs === undefined + ? {} + : { + optionalDependencyArtifactDirs: template.optionalDependencyArtifactDirs.map((directory) => + remapProjectPath(directory, input.group.promptContext.projectRoot, input.projectRoot) + ) + }), renderedPromptPath: path.join(artifactDir, "prompt.rendered.md"), promptTemplatePath: remapProjectPath( input.group.templatePath, diff --git a/packages/runtime/src/smithers.ts b/packages/runtime/src/smithers.ts index a975179ec..e16e9921d 100644 --- a/packages/runtime/src/smithers.ts +++ b/packages/runtime/src/smithers.ts @@ -3749,7 +3749,7 @@ export function compileSmithersWorkflow(input: SmithersCompileInput): CompiledSm }); }) ); - const dynamicGroups = input.graph.nodes + const compiledDynamicGroups = input.graph.nodes .filter( (node): node is ExpandedNode & { dynamic: NonNullable } => node.dynamic !== undefined ) @@ -3777,11 +3777,19 @@ export function compileSmithersWorkflow(input: SmithersCompileInput): CompiledSm }) ); const nonBlockingAttemptIds = [...nonBlockingGroupByAttemptId.keys()].sort(); - const tasks = compiledTasks.map((task) => ({ - ...task, - optionalDependencyArtifactDirs: task.dependencyArtifactDirs.filter((directory) => { + const optionalInputs = (task: CompiledSmithersTask): string[] => + task.dependencyArtifactDirs.filter((directory) => { const producerGroup = nonBlockingGroupByAttemptId.get(path.basename(directory)); return producerGroup !== undefined && reconcilesPartialResults(task, producerGroup); + }); + const tasks = compiledTasks.map((task) => ({ ...task, optionalDependencyArtifactDirs: optionalInputs(task) })); + // Generated children are cloned from their group's templates, so the templates follow the same + // rule. A template without optional inputs keeps its bytes. + const dynamicGroups = compiledDynamicGroups.map((group) => ({ + ...group, + taskTemplates: group.taskTemplates.map((template) => { + const optionalDependencyArtifactDirs = optionalInputs(template); + return optionalDependencyArtifactDirs.length === 0 ? template : { ...template, optionalDependencyArtifactDirs }; }) })); const smithersDir = path.join(input.runLayout.root, "smithers"); diff --git a/packages/runtime/test/dynamic-workflow.test.ts b/packages/runtime/test/dynamic-workflow.test.ts index d941a0864..e825583a2 100644 --- a/packages/runtime/test/dynamic-workflow.test.ts +++ b/packages/runtime/test/dynamic-workflow.test.ts @@ -197,6 +197,144 @@ test("compiled dynamic workflow defers templates, emits syntactically valid Type assert.ok(materialized.tasks.every((task) => task.sourceRef === plan.value!.source_ref)); }); +test("dynamic templates and their children treat a continuing producer in another group as optional", async () => { + const project = tempProject(); + initProject({ projectRoot: project, force: true }); + writePrompt(project, "dynamic/context.md", "dynamic-context", "Write context to {{artifact_path}}/context.md."); + writePrompt(project, "dynamic/planner.md", "dynamic-planner", "Write the plan to {{artifact_path}}/plan.json."); + writePrompt( + project, + "dynamic/worker.md", + "dynamic-worker", + "Your /goal is {{item.goal_prompt}}.\n{{finding_reachability_vocabulary}}\n{{finding_note_key_vocabulary}}" + ); + writePrompt( + project, + "dynamic/hunter.md", + "dynamic-hunter", + "Hunt the plan.\n{{finding_reachability_vocabulary}}\n{{finding_note_key_vocabulary}}" + ); + // `producer` continues on failure and reaches the dynamic group only through `planner`, a node of + // another group; the static `hunter` beside the group already treats its output as optional. + fs.writeFileSync( + path.join(project, ".ultrafuzz", "topology.yml"), + `version: 2 +defaults: + strategy_loops: 1 +groups: + strategies: + label: Strategies + defaults: + failure_policy: continue + planning: + label: Planning + goals: + label: Goals + defaults: + failure_policy: continue +nodes: + - id: __start__ + kind: meta + role: start + depends_on: [] + - id: producer + kind: agentic + prompt: dynamic/context.md + group: strategies + depends_on: [__start__] + outputs: + - path: context.md + contract: ultrafuzz/nonempty-markdown@1 + primary: true + - id: planner + kind: agentic + prompt: dynamic/planner.md + group: planning + depends_on: [producer] + outputs: + - path: plan.json + contract: ultrafuzz/goal-plan@1 + primary: true + - id: fanout + kind: agentic + prompt: dynamic/worker.md + group: goals + depends_on: [planner] + dynamic: + from: + node: planner + path: $.goals + key: id + node_id: "dynamic:item:{{ item.id }}" + outputs: + - path: findings.json + contract: ultrafuzz/findings@2 + primary: true + - id: hunter + kind: agentic + prompt: dynamic/hunter.md + group: goals + depends_on: [planner] + outputs: + - path: findings.json + contract: ultrafuzz/findings@2 + primary: true + - id: __finish__ + kind: meta + role: finish + depends_on: [fanout, hunter] +`, + "utf8" + ); + const plan = await planRun({ projectRoot: project, runId: "dynamic-optional-inputs", env: {} }); + assert.equal(plan.ok, true, JSON.stringify(plan.diagnostics)); + assert.ok(plan.value); + const planned = plan.value; + const compiled = compileSmithersWorkflow({ + projectRoot: project, + config: planned.resolved_config, + graph: planned.expanded_graph, + runLayout: planned.layout, + workflowName: "ultrafuzz-dynamic-optional-inputs", + renderedPrompts: planned.rendered_prompts + }); + const producerDir = compiled.tasks.find((task) => task.concreteNodeId === "producer")?.artifactDir; + assert.ok(producerDir); + for (const nodeId of ["planner", "hunter"]) { + const task = compiled.tasks.find((candidate) => candidate.concreteNodeId === nodeId); + assert.deepEqual(task?.optionalDependencyArtifactDirs, [producerDir], nodeId); + } + const [group] = compiled.dynamicGroups; + assert.ok(group); + assert.deepEqual( + group.taskTemplates.map((task) => task.optionalDependencyArtifactDirs), + [[producerDir]] + ); + + const sourceArtifactPath = path.resolve(project, group.source.artifactPath); + fs.mkdirSync(path.dirname(sourceArtifactPath), { recursive: true }); + fs.writeFileSync( + sourceArtifactPath, + `${JSON.stringify({ goals: [0, 1].map((index) => ({ id: `goal-${index}`, goal_prompt: `find ${index}` })) })}\n`, + "utf8" + ); + const materialized = materializeDynamicRuntime({ + runId: planned.run_id, + projectRoot: project, + runRoot: planned.run_root, + graphPath: path.join(planned.run_root, "graph.json"), + tasksPath: compiled.tasksPath, + baseTasks: compiled.tasks, + groups: compiled.dynamicGroups, + readyGroupIds: ["fanout"] + }); + const children = materialized.tasks.filter((task) => task.metadata.node.dynamic !== undefined); + assert.equal(children.length, 2); + for (const child of children) { + assert.deepEqual(child.optionalDependencyArtifactDirs, [producerDir], child.attemptId); + } +}); + test("compilation snapshots the exact transformed prompt body used during planning", async () => { const project = tempProject(); writeDynamicProject(project, { excludableContextNode: true }); From 64b9ad314201a50ff5f7a4fe9959a5e74d0c6fc4 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Wed, 30 Sep 2026 09:38:29 +0000 Subject: [PATCH 09/10] fix(dashboard): the property fan-in still renders as a property node The fan-in moved from group `properties` to `property-catalog`, and `flowNodeType` keyed the property card on `group === "properties"`, so `/api/flow` reported the fan-in as `agentAttempt` and the frontend showed it as a one-attempt strategy card. `property-catalog` now maps to `propertySpecification` too, and the flow API test pins the fan-in's type on a freshly initialized project. Co-Authored-By: Claude Opus 5.5 --- packages/dashboard/src/index.ts | 2 +- packages/dashboard/test/dashboard.test.ts | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/dashboard/src/index.ts b/packages/dashboard/src/index.ts index c046e5873..8d4f95fe8 100644 --- a/packages/dashboard/src/index.ts +++ b/packages/dashboard/src/index.ts @@ -2063,7 +2063,7 @@ function flowNodeType(node: TopologyNode): string { if (node.group === "setup") { return "projectDiscovery"; } - if (node.group === "properties") { + if (node.group === "properties" || node.group === "property-catalog") { return "propertySpecification"; } return "agentAttempt"; diff --git a/packages/dashboard/test/dashboard.test.ts b/packages/dashboard/test/dashboard.test.ts index 6cd75f9df..edbd2d77d 100644 --- a/packages/dashboard/test/dashboard.test.ts +++ b/packages/dashboard/test/dashboard.test.ts @@ -52,6 +52,7 @@ interface FlowResponse { }; nodes: Array<{ id: string; + type: string; data: { logicalNodeId: string; artifacts?: Record; @@ -79,6 +80,8 @@ test("serves logical topology flow with expanded attempt details", async () => { assert.ok(flow.nodes.length > 0); assert.ok(flow.run.expanded_nodes > flow.nodes.length); assert.ok(flow.nodes.every((node) => node.id === node.data.logicalNodeId)); + // The property fan-in sits in its own `property-catalog` group and still renders as a property node. + assert.equal(flow.nodes.find((node) => node.id === "property-specification-fanin")?.type, "propertySpecification"); assert.equal(flow.capabilities.runNewCampaign, true); assert.equal(flow.capabilities.referencesStatus, true); } finally { From caf49a3236d97a8d0fb6ed8ab9b7b1d32211b940 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Wed, 30 Sep 2026 09:38:29 +0000 Subject: [PATCH 10/10] docs(changelog): state what a failed or retried lens costs and what custom topologies change Review follow-ups for the Unreleased entries and the reference docs: - The breaking entry is split in two, so the optional-input rule that affects every custom topology is its own bullet. That bullet now names nodes with no `group` and the nodes a dynamic group generates, and says a node reaching a failed same-group ancestor only through another group fails its admission as an `artifact-contract` failure, which leaves the report unverified. - The packaged-topology entry names the packaged `default` topology, qualifies "PARTIAL" with a successful `resume --retry-failed`, says what that resume reruns (the lens, and after a verifier failure every node that started after it), records that verifier failures of the fan-in, strategies and specialists lose `output_contracts.missing`, their terminal disposition and the public eval `failure_code`, offers `ultrafuzz topology copy default` for an uncustomized project, and applies the fan-in prompt refresh to every existing project, since a project prompt overrides the built-in one under every profile. - The retry entry drops "before the task's own result is synchronized", covers the in-progress case fixed two commits back, and says the retried node then counts as succeeded, so completion can read COMPLETE. restart-continue.md says the same. - topology-yaml.md states the report consequence of the cross-group chain. - A test comment no longer attributes the emitted optional-input shape to "only the review task". Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 5 +++-- docs/how-to/restart-continue.md | 10 +++++++--- docs/reference/topology-yaml.md | 11 +++++++---- .../artifacts/test/smithers-task-manifest.test.ts | 5 +++-- 4 files changed, 20 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8139a8efa..f9a0eac22 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,7 +4,8 @@ ### Breaking changes -- **[config] [runtime] [prompts] [docs]** A property lens that fails no longer stops the campaign in the packaged `exhaustive` and `invariant-only` topologies or in projects initialized by this release. Their `properties` group now has `failure_policy: continue`, and `property-specification-fanin` moves to a new `property-catalog` group, which still halts: the fan-in consolidates the lenses that passed verification, and the strategies, specialists and review run without the failed lens's properties. The report is PARTIAL when the lens ran out of attempts, and unverified when it failed its output contract. In `exhaustive`, `dynamic-strategy-generator` now also runs when some strategy attempts failed, using the ones that succeeded, instead of being skipped. An existing project keeps the old behaviour under the `default` and `low-cost` profiles, where one failed lens fails the run before any strategy starts and leaves it without a report, until you make three edits to its `.ultrafuzz/topology.yml`: add `defaults: {failure_policy: continue}` to the `properties` group; add a `property-catalog` group with no `failure_policy` (the scaffold gives it `label: Property catalog` and `color: "#854d0e"`); and change the `property-specification-fanin` node's `group` from `properties` to `property-catalog`. Then delete `.ultrafuzz/prompts/properties/property-specification-fanin.md` and rerun `ultrafuzz init` to take the prompt that has the fan-in consolidate only the lenses it is given. The rule behind this changes for custom topologies too: a `failure_policy: continue` group's results are now optional to nodes in every other group, not only to the group named `review`, so such a node runs without a failed input instead of being skipped, while nodes in the same group still require them. To keep a chain strict, put all of it in one group. +- **[config] [runtime] [prompts] [docs]** A property lens that fails no longer stops the campaign. The packaged `default` topology, which `ultrafuzz init` scaffolds, and the packaged `exhaustive` and `invariant-only` topologies change: their `properties` group now has `failure_policy: continue`, and `property-specification-fanin` moves to a new `property-catalog` group, which still halts. The fan-in consolidates the lenses that passed verification, and the strategies, specialists and review run without the failed lens's properties. The report is PARTIAL when the lens ran out of attempts and unverified when it failed its output contract, unless a later `resume --retry-failed` reruns the lens successfully. That resume, which Modal's durable resume always runs, reruns the failed lens, and when the lens's artifact verifier failed, also every node that started after the lens's attempt, which is most of the campaign. In `exhaustive`, `dynamic-strategy-generator` now also runs when some strategy attempts failed, using the ones that succeeded, instead of being skipped. The fan-in, strategies and specialists now have optional inputs, so when the artifact verifier of one of them fails, `state.json` no longer records `output_contracts.missing` or `terminal_disposition: task-output-validation-failure` for it and public eval diagnostics omit its `failure_code`; the verifier's error stays in `last_error`, as it already did for review tasks. The `exhaustive` and `invariant-only` profiles use the new topology after the upgrade. The `default` and `low-cost` profiles run the project's own `.ultrafuzz/topology.yml`, so an existing project keeps the old behaviour there, where one failed lens fails the run before any strategy starts and leaves it without a report, until you make three edits to that file: add `defaults: {failure_policy: continue}` to the `properties` group; add a `property-catalog` group with no `failure_policy` (the scaffold gives it `label: Property catalog` and `color: "#854d0e"`); and change the `property-specification-fanin` node's `group` from `properties` to `property-catalog`. If you have not customized that file, `ultrafuzz topology copy default .ultrafuzz/topology.yml --force` replaces it with the new default instead. Every existing project, whichever profile it runs, should also delete `.ultrafuzz/prompts/properties/property-specification-fanin.md` and rerun `ultrafuzz init`: a project prompt overrides the built-in one under every profile, and only the new prompt tells the fan-in to consolidate the lenses it is given instead of every lens in the topology. +- **[runtime] [docs]** A `failure_policy: continue` group's results are now optional to every node outside that group, including nodes with no `group` and the nodes a dynamic group generates, not only to the group named `review`. Such a node runs without a failed input instead of being skipped, while nodes in the same group still require it. A node that reaches a failed ancestor of its own group only through another group, such as strategy `s2` after specialist `x` after strategy `s1`, now fails its input admission instead of being skipped; that is an `artifact-contract` failure, so the run's report is published unverified. To keep a chain strict, put all of it in one group. - **[runtime] [docs]** A final-report producer retry or verifier that runs in a restarted controller (resume, quota park, supervisor relaunch) now rebuilds `report.json#run_metadata.agent_execution` from `smithers/final-report-selections/.json` in the run directory, which each producer attempt writes just before starting its agent, instead of running `smithers node` as a subprocess (a `--full-output` read of up to 64 MiB under a 180 s budget). This applies to workflows rendered by this release, at launch or by `resume --refresh-controller`; plain `resume` keeps running the workflow persisted at launch, even after a refresh. On a run launched by an earlier release, a verifier refreshed after its producer succeeded under the old workflow fails with `report producer selection was never recorded`, and mixing plain and refreshed resumes can make it fail with `run_metadata.agent_execution differs from the controller-observed producer`; in either case run `ultrafuzz resume --refresh-controller --reset-node ` (`node:final-report` in the packaged topologies), which reruns the report producer and its verifier rather than every failed task. Producer attempts after such a refresh leave the earlier release's attempts out of `agent_execution.failed_attempts`. After a restart the verifier compares against a run-directory file instead of the Smithers database. The reference docs now say that neither is a boundary against an agent running unsandboxed as the same user, which retires #585's claim that the run filesystem cannot be used to forge the producer (Refs #1143). - **[config] [runtime] [modal] [docs]** Removes per-node cloud execution (#134), which never completed a node in any release (#712, #713). `validate`, `doctor`, `run` and `eval run` reject `[execution] mode = "cloud"`, `execution.provider` and `[execution.providers.modal]` with `CONFIG_EXECUTION_CLOUD_REMOVED` before a run directory exists; remove `execution.provider` and `[execution.providers.modal]`, and set `mode = "local"` or delete it. The rest of `[execution]` is still accepted and validated but no longer affects execution: every agentic attempt runs locally. To use Modal, run the whole campaign inside one sandbox, as the unchanged `ultrafuzz-modal` eval runner does. A run planned with `mode = "cloud"` cannot be continued: `resume`, with or without `--refresh-controller`, reads the mode from the run's `smithers/resolved-config.json` and fails with `WORKFLOW_CLOUD_EXECUTION_REMOVED` before starting Smithers, so start a new run. It also fails, with `WORKFLOW_LIFECYCLE_FAILED`, for a run whose resolved config is missing or does not parse, which a plain `resume` used to continue without it. `ultrafuzz clean` no longer terminates Modal sandboxes tagged `purpose=ultrafuzz-node` or deletes the `ultrafuzz-node-*` volumes of earlier cloud attempts; remove those by hand in the Modal app the run used, which its `plan.json` records under `execution.providers.modal` until `ultrafuzz clean` deletes the run. The runtime `cloud-execution-generation` schema and the seven Modal node schemas are deleted, so `json validate` no longer recognizes them. - **[runtime] [docs]** A run launched by an earlier release whose dynamic groups have already expanded cannot be synchronized, resumed or reported after upgrading. Each synchronization re-renders the run's published prompts with the current code and compares the bytes, and this release changes the coverage text the final report prompt embeds (#1176) and the order of multi-path authority selectors (#1195), so such a run fails with `runtime rendered prompt changed`. Finish or cancel it before upgrading; `pause` and `cancel` still work on it. @@ -25,7 +26,7 @@ ### Other changes -- **[runtime] [docs]** A task that started without a failed optional input, such as a review task without a failed strategy, keeps its result when `resume --retry-failed` reruns that input and the rerun succeeds before the task's own result is synchronized. Before, the task failed with "finalized optional dependency is missing from verifier admission", which could halt review and fail the run. +- **[runtime] [docs]** A task that started without a failed optional input, such as a review task without a failed strategy, keeps its result when `resume --retry-failed` reruns that input, whether the task is synchronized while the rerun is still running or after it succeeded. Before, the task failed with "optional dependency is not terminal for verifier admission" or "finalized optional dependency is missing from verifier admission", which could halt review and fail the run; in the first case a later task that checks the task's output, such as triage after dedupe, also failed for good. The retried node then counts as succeeded, so the report's completion can read COMPLETE and a `--require-complete` run can succeed, although the tasks that started before it verified never read its output. For a property lens, those are the property fan-in and everything after it. - **[runtime] [cli] [docs]** Production copies of `brace-expansion` and `undici` move past the High advisories published on 2026-09-29 (GHSA-6j4f-fj2g-mc7p and GHSA-qhr7-859c-m2p7 for `brace-expansion`, GHSA-rfgv-xxqx-mfg5 for `undici`) and the Moderate and Low advisories fixed alongside them (GHSA-q2hr-2g5m-vwhr, GHSA-3wwx-pv8p-q78v, GHSA-r53p-7pc4-xj5r). The CLI's two `brace-expansion` copies, both pulled in by `@oclif/core` through `minimatch`, move from 2.1.4 and 5.0.9 to 2.1.7 and 5.0.12. No npm release bundles these fixes yet, so the `npm@11.19.0` patch now also carries `brace-expansion` 5.0.12 and `undici` 6.28.1 inside the operator npm, and the operator npm's pinned closure digest changes with it. - **[artifacts]** Concurrent ultrafuzz commands, such as `ultrafuzz status --watch` while `ultrafuzz cancel` runs, no longer lose or tear records in a run's `events.jsonl`, `usage.jsonl` or `attempts.jsonl`, or in the `.ultrafuzz/` audit journals. Each append holds a short-lived `.lock` next to the journal; a lock left by a killed process on the same host is taken over at once, and one that cannot be checked after 10 seconds. - **[runtime] [docs]** When `resume --retry-failed` retries a dynamic source whose verifier failed, it now also withdraws the rendered prompts of later nodes that wait on the group, so they render again from the new expansion. Before, if the retried source planned different items, a node whose prompt names the group's children (for example with `{{artifact_path:}}`) failed every render with `runtime rendered prompt changed for `. The packaged topologies were not affected (#1141). diff --git a/docs/how-to/restart-continue.md b/docs/how-to/restart-continue.md index 8f2e75c8d..48b426d4b 100644 --- a/docs/how-to/restart-continue.md +++ b/docs/how-to/restart-continue.md @@ -85,9 +85,13 @@ the prerequisite succeed. Retrying a failed node of a `failure_policy: continue` group does not undo work that already ran without it. A task that started before the retried node verified keeps the result it produced without that node, and only tasks that -start afterwards read the new output. A retried artifact verifier is different: -resetting its agent producer also resets every node that started after that -producer's attempt, so those nodes run again. +start afterwards read the new output. The retried node then counts as +succeeded, so the report's completion can read COMPLETE and a +`--require-complete` run can succeed, although those earlier tasks never read +its output. For a property lens, those are the property fan-in and everything +after it. A retried artifact verifier is different: resetting its agent +producer also resets every node that started after that producer's attempt, so +those nodes run again. ## Replay A Linked Run diff --git a/docs/reference/topology-yaml.md b/docs/reference/topology-yaml.md index 57d822937..7c26f1c01 100644 --- a/docs/reference/topology-yaml.md +++ b/docs/reference/topology-yaml.md @@ -99,10 +99,13 @@ that post-agent contract failure is terminal. `failure_policy: continue` marks every node in that group as nonblocking. The generated workflow waits for such a node to settle and consumes its artifacts only when its verifier succeeded. When it fails or is skipped, unrelated work -continues and nodes in other groups run with the results that did succeed. A -node of the same group never runs without it: a node that depends on it -directly, or through nodes of the same group, is skipped, and one that reaches -it only through another group fails its input admission instead. The packaged +continues, and every node outside the group, including a node with no `group` +and the nodes a dynamic group generates, runs with the results that did +succeed. A node of the same group never runs without it: a node that depends on +it directly, or through nodes of the same group, is skipped. One that reaches it +only through another group fails its input admission instead, which counts as +an `artifact-contract` failure, so the run's best-effort report is published +unverified. Keep a chain that must stay strict inside one group. The packaged topologies use this for the property lenses, goals, strategies, and specialists; ordinary groups retain fail-closed `halt` semantics. diff --git a/packages/artifacts/test/smithers-task-manifest.test.ts b/packages/artifacts/test/smithers-task-manifest.test.ts index 98945df29..3dd031fae 100644 --- a/packages/artifacts/test/smithers-task-manifest.test.ts +++ b/packages/artifacts/test/smithers-task-manifest.test.ts @@ -493,8 +493,9 @@ test("optional task inputs must come from producers that continue on failure", ( const producer = "/runs/run-1/artifacts/producer"; // Which consumers opt in is compiler policy, so the gate accepts every shape it has emitted: only - // the review task opts in (#1120), every consumer opts in (before #1120), or a review task keeps the - // input required (dynamic lowering of an empty expansion keeps its source required). + // the consumer outside the producer's group opts in (#1198; #1120 allowed only review), every + // consumer opts in (before #1120), or a review task keeps the input required (dynamic lowering of + // an empty expansion keeps its source required). const emittedShapes: Array<[consumerOptional: string[], reviewerOptional: string[]]> = [ [[], [producer]], [[producer], [producer]],