From 2a179c945be89f5d1e30804a9f3987a3ab23568f Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:14 +0200 Subject: [PATCH 1/6] Add runnable public reference package --- package.json | 8 ++++++++ 1 file changed, 8 insertions(+) create mode 100644 package.json diff --git a/package.json b/package.json new file mode 100644 index 0000000..d9de8ee --- /dev/null +++ b/package.json @@ -0,0 +1,8 @@ +{ + "name": "machineoutcome-reference", + "version": "1.0.0", + "private": false, + "type": "module", + "scripts": { "test": "node --test", "verify": "node --test" }, + "engines": { "node": ">=20" } +} From 88defcf3db25a9b721d252e844cc4d6b2424b530 Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:22 +0200 Subject: [PATCH 2/6] Add fail-closed public outcome verifier --- src/reference-outcome-verifier.mjs | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 src/reference-outcome-verifier.mjs diff --git a/src/reference-outcome-verifier.mjs b/src/reference-outcome-verifier.mjs new file mode 100644 index 0000000..21e8121 --- /dev/null +++ b/src/reference-outcome-verifier.mjs @@ -0,0 +1,24 @@ +export function verifyMutationOutcome(input) { + const required = ['expectedPreState', 'expectedPostState', 'observedPreState', 'readbackComplete', 'providerAcknowledgement']; + for (const key of required) { + if (!(key in input)) throw new TypeError(`missing ${key}`); + } + + if (input.observedPreState !== input.expectedPreState) { + return { status: 'FAILED', reason: 'PRESTATE_MISMATCH', retry: 'DO_NOT_MUTATE' }; + } + + if (!input.readbackComplete) { + return { status: 'UNKNOWN', reason: 'READBACK_INCOMPLETE', retry: 'RECONCILE_BEFORE_RETRY' }; + } + + if (input.observedPostState === input.expectedPostState) { + return { status: 'VERIFIED', reason: 'POSTSTATE_CONFIRMED', retry: 'DO_NOT_RETRY' }; + } + + if (input.providerAcknowledgement === 'ambiguous' && input.observedPostState === input.observedPreState) { + return { status: 'UNKNOWN', reason: 'AMBIGUOUS_MUTATION_STATE', retry: 'RECONCILE_BEFORE_RETRY' }; + } + + return { status: 'FAILED', reason: 'POSTSTATE_MISMATCH', retry: 'DO_NOT_RETRY' }; +} From 69a239ce76dc57ada0d59f10d92ca9a854296c31 Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:33 +0200 Subject: [PATCH 3/6] Add adversarial outcome verification tests --- test/reference-outcome-verifier.test.mjs | 49 ++++++++++++++++++++++++ 1 file changed, 49 insertions(+) create mode 100644 test/reference-outcome-verifier.test.mjs diff --git a/test/reference-outcome-verifier.test.mjs b/test/reference-outcome-verifier.test.mjs new file mode 100644 index 0000000..160e32a --- /dev/null +++ b/test/reference-outcome-verifier.test.mjs @@ -0,0 +1,49 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { verifyMutationOutcome } from '../src/reference-outcome-verifier.mjs'; + +function base(overrides = {}) { + return { + expectedPreState: 'sha:pre', + expectedPostState: 'sha:post', + observedPreState: 'sha:pre', + observedPostState: 'sha:post', + readbackComplete: true, + providerAcknowledgement: 'accepted', + ...overrides, + }; +} + +test('exact precondition plus exact provider readback is VERIFIED', () => { + assert.deepEqual(verifyMutationOutcome(base()), { + status: 'VERIFIED', reason: 'POSTSTATE_CONFIRMED', retry: 'DO_NOT_RETRY' + }); +}); + +test('stale pre-state fails before mutation authority', () => { + assert.deepEqual(verifyMutationOutcome(base({ observedPreState: 'sha:other' })), { + status: 'FAILED', reason: 'PRESTATE_MISMATCH', retry: 'DO_NOT_MUTATE' + }); +}); + +test('missing provider readback preserves UNKNOWN and blocks retry', () => { + assert.deepEqual(verifyMutationOutcome(base({ readbackComplete: false, observedPostState: null })), { + status: 'UNKNOWN', reason: 'READBACK_INCOMPLETE', retry: 'RECONCILE_BEFORE_RETRY' + }); +}); + +test('ambiguous transport response can still verify through exact readback', () => { + assert.equal(verifyMutationOutcome(base({ providerAcknowledgement: 'ambiguous' })).status, 'VERIFIED'); +}); + +test('ambiguous transport plus unchanged state remains UNKNOWN', () => { + assert.deepEqual(verifyMutationOutcome(base({ providerAcknowledgement: 'ambiguous', observedPostState: 'sha:pre' })), { + status: 'UNKNOWN', reason: 'AMBIGUOUS_MUTATION_STATE', retry: 'RECONCILE_BEFORE_RETRY' + }); +}); + +test('complete readback of a wrong post-state is FAILED', () => { + assert.deepEqual(verifyMutationOutcome(base({ observedPostState: 'sha:unexpected' })), { + status: 'FAILED', reason: 'POSTSTATE_MISMATCH', retry: 'DO_NOT_RETRY' + }); +}); From d17784be9795f395e1420acb50e61dc05510fcd2 Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:41 +0200 Subject: [PATCH 4/6] Add synthetic outcome fixture --- fixtures/verified-after-ambiguous-transport.json | 8 ++++++++ 1 file changed, 8 insertions(+) create mode 100644 fixtures/verified-after-ambiguous-transport.json diff --git a/fixtures/verified-after-ambiguous-transport.json b/fixtures/verified-after-ambiguous-transport.json new file mode 100644 index 0000000..1cd696c --- /dev/null +++ b/fixtures/verified-after-ambiguous-transport.json @@ -0,0 +1,8 @@ +{ + "expectedPreState": "sha:pre", + "expectedPostState": "sha:post", + "observedPreState": "sha:pre", + "observedPostState": "sha:post", + "readbackComplete": true, + "providerAcknowledgement": "ambiguous" +} From 05672101729cb8f6295e7a82fcb17f9f268a2468 Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:38:46 +0200 Subject: [PATCH 5/6] Add public reference CI --- .github/workflows/verify-reference.yml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 .github/workflows/verify-reference.yml diff --git a/.github/workflows/verify-reference.yml b/.github/workflows/verify-reference.yml new file mode 100644 index 0000000..499da9f --- /dev/null +++ b/.github/workflows/verify-reference.yml @@ -0,0 +1,19 @@ +name: verify-reference + +on: + push: + branches: [main] + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 24 + - run: npm test From 226c77ce2055d8d489cb9749fc085dedf1d700ac Mon Sep 17 00:00:00 2001 From: Sam <125739417+SamCT86@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:39:05 +0200 Subject: [PATCH 6/6] Reframe MachineOutcome around executable verification proof --- README.md | 119 +++++++++++++++++++++++------------------------------- 1 file changed, 50 insertions(+), 69 deletions(-) diff --git a/README.md b/README.md index 54158d1..c04a449 100644 --- a/README.md +++ b/README.md @@ -1,97 +1,78 @@ # MachineOutcome — verified outcomes before agent trust -**Sarmad Tawfeek · AI systems · technical implementation · automation** -**Status:** Building -**Portfolio:** https://sarmadtawfeek.se/ +A public, executable engineering reference for one MachineOutcome principle: **do not infer success from an attempted mutation; verify the exact resulting state.** The production system remains private. -## My role in this build +## Five-minute technical evaluation -I researched the product/system problem, chose the direction, defined the high-level blueprint and quality expectations, and used specialist AI personas/agents to drive implementation and iteration. - -The implementation is heavily AI-assisted. I do **not** claim that I personally hand-wrote every line of code or independently selected every low-level technical mechanism. My direct ownership is the product direction, system requirements, expert/persona orchestration, acceptance criteria and quality gates. - -MachineOutcome starts from a narrower question than “how trustworthy is this agent?” - -> **What actually happened after this specific agent attempted this specific task, and what evidence supports that conclusion?** - -The current first task class is **software repository change** rather than a universal agent score. - -## What exists today - -Current private-source evidence includes: +```bash +git clone https://github.com/SamCT86/machineoutcome-case-study.git +cd machineoutcome-case-study +npm test +``` -- a concrete first task class: `software.repository_change.v1`; -- task / attempt / evidence / outcome identity; -- an initial operational utility called **Agent Change Outcome Guard**; -- append-oriented evidence and correction semantics; -- structured control/evidence records for review and same-artifact verification workflows; -- a dependency order where reliability, delegation and routing remain downstream of verified outcome evidence. +Then inspect: -That means this repo is not presenting a future reliability score as if it already existed. +- `src/reference-outcome-verifier.mjs` — bounded state-verification reference; +- `test/reference-outcome-verifier.test.mjs` — adversarial retry/readback tests; +- `fixtures/verified-after-ambiguous-transport.json` — synthetic provider-state example; +- `PROOF.md` — broader implementation evidence; +- `PUBLIC_BOUNDARY.md` — what intentionally stays private. -**Start with the evidence layer:** [PROOF.md](PROOF.md) +## What this proves -## System boundary +The public reference encodes a narrow operational contract: ```text -Task - ↓ -Agent attempt - ↓ -Inspectable evidence - ↓ -VERIFIED / FAILED / UNKNOWN - ↓ -Outcome receipt / history - ↓ -Only with comparable evidence: -reliability / delegation support +expected pre-state +→ mutation attempt +→ provider/system readback +→ VERIFIED | FAILED | UNKNOWN +→ retry only when the observed state makes retry safe ``` -These are current system requirements/behaviors; they are not a claim that I personally originated every low-level mechanism used to implement them. +It demonstrates that: -## A failure boundary in the system +- stale pre-state blocks mutation authority; +- incomplete readback preserves `UNKNOWN`; +- an ambiguous transport response is not automatically a failure; +- exact post-state readback can verify an ambiguously acknowledged mutation; +- ambiguous state blocks blind retry; +- a complete but incorrect post-state is `FAILED`. -A coding agent may evaluate repositories containing prompt-like text or instructions. If that content can redefine the evaluator’s rules, the thing being measured can influence the measurement process. +The reference deliberately treats **observed state as stronger evidence than transport optimism**. -The current system boundary therefore treats repository/task content as **untrusted data, not instruction authority**. +## Why this matters for AI agents -## How AI fits +Agent systems fail in a dangerous way when “the call returned strangely” becomes “retry it” without reconciling what actually happened. A duplicated repository write, deployment, payment, migration or external action can be worse than an explicit failure. -AI agents/models are used heavily for implementation, system exploration, edge-case generation, review and iteration. +The production MachineOutcome system is materially broader than this sample and includes task/attempt/evidence identity, provenance, append-oriented history, recovery and downstream reliability/delegation work. That private implementation is not published here. -My role is to define the outcome-verification problem, blueprint the required system behavior, structure the expert/persona workflow, set the quality bar and require evidence/quality gates before accepting stronger claims. +This repository is a **reference edition**, not a source release of the production runtime. -More detail: [docs/HOW_I_BUILD_WITH_AI.md](docs/HOW_I_BUILD_WITH_AI.md) +## How I build -## Technical context +I use AI agents heavily for implementation, investigation, testing and adversarial review. My ownership is the product problem, evidence doctrine, architecture constraints, acceptance gates, falsifiers and the decision to accept or reject the resulting system. -`AI-agent workflows` · `Git / GitHub evidence` · `structured verification` · `provenance` · `deterministic outcome states` +I do not claim to have hand-written every line. The intended engineering signal is the ability to direct AI-native implementation toward explicit state, falsifiable claims, deterministic verification and safe recovery boundaries. -This is implementation context, not a claim that I personally selected or authored every technical mechanism. +## Public/private boundary -## Inspect the case study +Public here: -- [Observable proof](PROOF.md) -- [Sanitized outcome example](examples/sanitized-outcome.json) -- [System view](docs/SYSTEM_VIEW.md) -- [System requirements & trade-offs](docs/DECISIONS.md) -- [Verification approach](docs/VERIFICATION.md) -- [Public / private boundary](PUBLIC_BOUNDARY.md) +- a bounded state-verification reference; +- synthetic states; +- executable tests; +- CI; +- non-proprietary system/evidence documentation. -## Not claimed +Private: -- a universal agent trust score; -- broad task coverage; -- proven commercial demand; -- product-market fit; -- completion of every planned reliability, delegation or routing layer; -- personal authorship of every implementation detail. +- production providers and credentials; +- internal repository IDs, incident details and operator authority; +- production schemas, storage and recovery implementation; +- proprietary evaluator/runtime logic; +- unreleased reliability, delegation and routing systems. -The case study is strongest when read as an AI-native product/system build that I direct and quality-gate, with low-level implementation performed heavily through AI-assisted workflows. - -## Related engineering case studies +## Not claimed -- [Agent Cashflow OS](https://github.com/SamCT86/agent-cashflow-os-case-study) — forecast provenance, calibration and held-out evaluation discipline. -- [ReleaseProof](https://github.com/SamCT86/releaseproof-case-study) — exact-artifact verification and reproducible evidence. -- [Billable Meetings OS](https://github.com/SamCT86/billable-meetings-os-case-study) — deterministic business-rule verification with explicit review states. +This repository does not claim universal agent reliability, broad task coverage, commercial demand, product-market fit, or that this small reference implementation is the production MachineOutcome runtime.