diff --git a/.github/workflows/ucf-foundation.yml b/.github/workflows/ucf-foundation.yml new file mode 100644 index 0000000..bf3dff5 --- /dev/null +++ b/.github/workflows/ucf-foundation.yml @@ -0,0 +1,44 @@ +name: UCF foundation + +on: + pull_request: + paths: + - "foundry/contracts/transition_models.py" + - "foundry/environments/**" + - "foundry/providers/base.py" + - "foundry/runtime/**" + - "foundry/orchestration/agent_runner.py" + - "foundry/orchestration/run_engine.py" + - "foundry/contracts/task_types.py" + - "tests/unit/runtime/**" + - "tests/unit/test_transition_models.py" + - "tests/unit/orchestration/test_agent_runner.py" + - ".github/workflows/ucf-foundation.yml" + push: + paths: + - "foundry/contracts/transition_models.py" + - "foundry/environments/**" + - "foundry/providers/base.py" + - "foundry/runtime/**" + +jobs: + foundation: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: pip + - name: Install + run: python -m pip install -e ".[dev]" + - name: Compile + run: python -m compileall -q foundry tests + - name: Lint changed foundation + run: | + ruff check foundry/contracts/transition_models.py foundry/contracts/task_types.py foundry/environments foundry/providers/base.py foundry/runtime foundry/orchestration/agent_runner.py foundry/orchestration/run_engine.py tests/unit/test_transition_models.py tests/unit/runtime/test_transition_engine.py tests/unit/orchestration/test_agent_runner.py + - name: Test foundation + run: | + pytest -q tests/unit/test_transition_models.py tests/unit/runtime/test_transition_engine.py tests/unit/orchestration/test_agent_runner.py + - name: Test full suite + run: pytest -q diff --git a/CLAUDE.md b/CLAUDE.md index b8cbba8..8f68a2c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,27 +1,45 @@ -# Unicorn Foundry — Project Guidance +# UCF — Project Guidance ## What This Repo Is -unicorn-foundry is the internal Claude orchestration system for Unicorn Protocol. -It plans, builds, reviews, extracts, evaluates, and improves the Unicorn system. +UCF is an experiment in persistent machine operation across changing state. -It is NOT a chatbot. It is a controlled run engine that produces artifacts, diffs, PRs, and structured data. +The repository now contains two layers: -There are two repos: -- `unicorn-app` — the product (Next.js + Go + Postgres). Users touch this. -- `unicorn-foundry` — this repo. The build system. Humans and Claude touch this. +- **UCF foundation** — provider-neutral contracts and a minimal transition runtime built around explicit state, evidence, action, verification, outcomes, and continuity. +- **Historical Foundry runtime** — the original Claude/Git/PR orchestration system built while working on the now-discontinued Unicorn project. -Foundry writes code INTO unicorn-app via git worktrees and PRs. It never writes to unicorn-app's database directly. It never deploys anything. +The historical implementation is evidence of how the experiment emerged. Preserve it, but do not treat Unicorn, Claude, GitHub, source code, or pull requests as architectural invariants of UCF. + +The general loop is: + +**State(t) → Reason/Plan → Controlled Action → Observation → Verification → Outcome → State(t+1)** + +The important object is the state transition. The model is a participant in the loop, not the loop itself. ## Core Thesis -Unicorn Protocol makes startup reality computationally legible. +UCF explores the boundary between **intelligence at an instant** and **intelligence through time**. + +New foundation work should preserve these invariants: + +1. state is explicit; +2. evidence remains attached to claims about state and outcomes; +3. actions occur through controlled environment boundaries; +4. verification is separate from generation; +5. meaningful transitions leave durable history; +6. intelligence providers are replaceable; +7. resulting state can seed the next transition. + +The original Unicorn chain — **Signals → Evidence → State → Legibility** — remains historical context, not the active product objective. + +## Compatibility Rule -The chain: **Signals → Evidence → State → Legibility** +Do not mass-rename or delete historical Foundry code merely to make terminology look generic. Generalization must be earned through exercised interfaces and tests. -A startup emits signals. Those signals become evidence. Evidence is used to infer state. State becomes legible to humans and software. +When touching new UCF foundation code, prefer the contracts under `foundry/contracts/transition_models.py`, `foundry/runtime/`, `foundry/environments/`, and `foundry/providers/base.py`. -Everything Foundry builds must serve that chain. +When touching historical Foundry code, preserve existing behavior unless the task explicitly migrates that behavior onto the new transition interfaces. ## Non-Negotiable Rules @@ -43,7 +61,7 @@ Everything Foundry builds must serve that chain. 9. **Structured output over prose.** Plans, reviews, extractions, evals, and verification results must return validated JSON matching their defined schemas. Never return prose where structured output is expected. -10. **No Claude in the hot path.** unicorn-app serves precomputed truth. Claude lives in Foundry, before the read models, not inside user requests. +10. **Do not introduce new provider coupling into the UCF foundation.** Claude remains the default provider for the historical Foundry runtime, but new transition/runtime interfaces must depend on capabilities rather than a model vendor. ## Repo Layout @@ -71,9 +89,9 @@ unicorn-foundry/ └── tests/ ← unit + integration ``` -## Canon +## Historical Foundry Canon -The source of truth for Unicorn's domain model lives in `canon/`. +The source of truth for the historical Unicorn domain model lives in `canon/`. It is not the general UCF state model. Before doing any extraction, schema, or domain work, always read the relevant canon document: - `canon/docs/event_taxonomy.md` — what counts as an event, event types, required fields @@ -93,7 +111,7 @@ Use the right model for the right job: Default routing is defined in `foundry/orchestration/model_router.py`. Override via `model_override` in task requests only when justified. -## Language Boundaries +## Historical Foundry Language Boundaries - **Go** — unicorn-app backend (API, workers). When implementing Go code, follow the patterns already in `services/api/`. - **TypeScript** — unicorn-app frontend (Next.js). Follow patterns in `apps/web/`. @@ -101,7 +119,7 @@ Default routing is defined in `foundry/orchestration/model_router.py`. Override Never mix these. A task that touches Go code uses the backend-implementer subagent. A task that touches TS uses the frontend-implementer. Foundry itself is always Python. -## How Runs Work +## Historical Foundry Run Lifecycle 1. A task is submitted via the control plane API. 2. A worktree is created for the target repo + branch. diff --git a/README.md b/README.md index 56b8166..67a28e2 100644 --- a/README.md +++ b/README.md @@ -1,210 +1,213 @@ -# Unicorn Foundry - -Internal Claude orchestration system for [Unicorn Protocol](https://github.com/sinethxyz/ucf). Plans, builds, reviews, extracts, evaluates, and improves the Unicorn system through controlled, artifact-producing runs. - -Foundry is not a chatbot. It is a **controlled run engine** that produces artifacts, diffs, PRs, and structured data. +# UCF + +**An experiment in persistent machine operation across changing state.** + +UCF began as Unicorn Foundry, the execution side of a discontinued project called Unicorn. Unicorn explored how a changing environment could be made machine-legible through **signals → evidence → state → legibility**. UCF explored the complementary problem: once a machine has a representation of its current environment, how can it change that environment deliberately, determine what actually happened, and continue from the resulting state? + +The original implementation was built around Claude and software-engineering workflows. The underlying systems question was broader: + +> **What must exist around an intelligent model for it to remain coherent while the environment it operates in changes?** + +UCF treats the model as a participant in the system, not the system itself. + +## The Loop + +```text +WORLD + │ + ▼ +OBSERVATION + │ + ▼ +EVIDENCE + │ + ▼ +STATE(t) + │ + ▼ +REASON / DECIDE + │ + ▼ +INTENT + │ + ▼ +PLAN + │ + ▼ +CONTROLLED EXECUTION + │ + ▼ +VERIFICATION / REVIEW + │ + ▼ +OUTCOME + │ + ▼ +STATE(t+1) + └──────────────↻ +``` -## Core Thesis +The important object is the **state transition**. -Unicorn Protocol makes startup reality computationally legible. +A model may reason, plan, classify, review, or propose an action. UCF surrounds those capabilities with explicit state, typed transitions, isolated execution, deterministic verification, independent evaluation, durable artifacts, and event history. -**Signals → Evidence → State → Legibility** +That distinction can be summarized as: -A startup emits signals. Those signals become evidence. Evidence is used to infer state. State becomes legible to humans and software. Everything Foundry builds serves that chain. +**intelligence at an instant ≠ intelligence through time** -## How It Works +## Origin -1. A task is submitted via the control plane API (`POST /v1/runs`). -2. A git worktree is created for the target repo and branch. -3. A **planner** subagent produces a structured `PlanArtifact` (JSON). -4. An **implementer** subagent executes the plan in the worktree. -5. Deterministic **verification** runs (build, test, lint, schema validation). -6. A **reviewer** subagent independently reviews the diff (without seeing the plan). -7. If approved, a PR is opened. All artifacts are stored. +UCF was originally implemented as **Unicorn Foundry**, an internal control plane for Unicorn Protocol. -Every operation is isolated, logged, and reproducible. Failed runs can be retried. Any run can be cancelled. +The two projects explored opposite sides of one loop: -## Architecture +```text +Unicorn +signals → evidence → state → legibility +UCF / Unicorn Foundry +state → intent → plan → action → verification → transition ``` - ┌─────────────────────────────────────────┐ - │ EXTERNAL INPUTS │ - │ human specs raw sources bug reports │ - │ eval datasets canon updates │ - └──────────────────┬──────────────────────┘ - │ - ▼ -┌──────────────────────────────────────────────────────────────────────────┐ -│ unicorn-foundry │ -│ │ -│ ┌────────────────────────────────────────────────────────────────────┐ │ -│ │ FastAPI Control Plane │ │ -│ │ │ │ -│ │ POST /v1/runs GET /v1/runs/{id} │ │ -│ │ POST /v1/runs/{id}/cancel POST /v1/runs/{id}/retry │ │ -│ │ GET /v1/runs/{id}/events GET /v1/runs/{id}/artifacts │ │ -│ │ POST /v1/reviews POST /v1/specs/plan │ │ -│ │ POST /v1/patches/apply POST /v1/batches/extract │ │ -│ │ GET /v1/batches/{id} GET /v1/batches/{id}/results │ │ -│ │ POST /v1/evals/run GET /v1/evals/{id} │ │ -│ │ POST /v1/worktrees/cleanup GET /v1/health │ │ -│ └────────────────────────┬───────────────────────────────────────────┘ │ -│ │ │ -│ ┌──────────────┼──────────────┐ │ -│ ▼ ▼ ▼ │ -│ ┌──────────────┐ ┌─────────────┐ ┌──────────────┐ │ -│ │ Orchestrator │ │ Batch │ │ Eval │ │ -│ │ (Agent SDK) │ │ Processor │ │ Runner │ │ -│ └──────┬───────┘ └──────┬──────┘ └──────┬───────┘ │ -│ │ │ │ │ -│ ▼ ▼ ▼ │ -│ ┌──────────────────────────────────────────────────────────────────┐ │ -│ │ Claude Layer │ │ -│ │ CLAUDE.md agents skills hooks rules MCP profiles │ │ -│ └──────────────────────────────────────────────────────────────────┘ │ -│ │ -│ ┌──────────────────────────────────────────────────────────────────┐ │ -│ │ Storage Layer │ │ -│ │ PostgreSQL: runs, events, artifacts, worktrees, batches, evals │ │ -│ │ Redis: task queue, run status pub/sub │ │ -│ │ Object storage: plans, diffs, patches, logs │ │ -│ └──────────────────────────────────────────────────────────────────┘ │ -└────────────────────────────────┬─────────────────────────────────────────┘ - │ git worktrees / PRs / artifacts - ▼ -┌──────────────────────────────────────────────────────────────────────────┐ -│ unicorn-app │ -│ Next.js (TS) → Go API → PostgreSQL + pgvector │ -│ Domains: companies, events, evidence, state, scoring, search │ -└──────────────────────────────────────────────────────────────────────────┘ + +Together, the intended system was: + +```text +world + ↓ +representation + ↓ +intelligence + ↓ +controlled action + ↓ +world' + ↓ +representation' + ↻ ``` -## Repo Structure +Unicorn is no longer an active project. UCF is being preserved and generalized because the infrastructure problem it exposed is not specific to Unicorn. + +This repository does **not** claim to implement a learned world model or a complete architecture for general intelligence. It is a software-systems experiment in maintaining coherent machine operation across observation, action, consequence, and time. + +## What The Existing Implementation Actually Contains + +The historical Foundry implementation already provides concrete machinery for this experiment: + +- a typed run-state machine; +- isolated git worktrees for actions; +- structured planning artifacts; +- model/provider routing; +- deterministic build, test, lint, and schema verification; +- independent diff review; +- persistent run events and artifacts; +- PostgreSQL-backed run state; +- Redis-backed work queues; +- evidence and extraction contracts; +- evaluation infrastructure; +- deterministic hooks for policy enforcement; +- retry, cancellation, and failure states. +Some originally planned paths remain incomplete. In particular, parts of the extraction/batch pipeline are still Phase 1 stubs. The repository should therefore be read as a working experimental system with unfinished surfaces, not as a completed general architecture. + +## Historical Implementation + +The codebase still uses its original **Foundry** terminology and contains Unicorn-specific adapters, schemas, task types, and documentation. Those are retained for now because they are evidence of how the experiment emerged. + +The current restructuring deliberately starts with the **conceptual boundary** before rewriting the runtime: + +```text +historical implementation generalized interpretation + +Unicorn state → environment state +Unicorn canon → state/evidence contracts +Claude → intelligence provider +Foundry run → controlled state transition +git worktree → isolated action environment +verification → transition validation +review → independent evaluation +run artifacts → durable transition evidence +PR → one possible action outcome ``` -unicorn-foundry/ -├── CLAUDE.md # repo-wide guidance and non-negotiable rules -├── .claude/ -│ ├── settings.json # permissions, hooks, deny rules -│ ├── agents/ # subagent definitions -│ │ ├── planner.md # structured implementation planning -│ │ ├── backend-implementer.md # Go implementation specialist -│ │ ├── frontend-implementer.md # TypeScript/Next.js specialist -│ │ ├── reviewer.md # independent diff reviewer -│ │ ├── extractor.md # signal-to-event extraction -│ │ ├── migration-guard.md # high-scrutiny infra/migration review -│ │ └── repo-explorer.md # read-only reconnaissance -│ ├── rules/ # enforced policy documents -│ │ ├── repo-safety.md # secret blocking, protected paths -│ │ ├── api-contracts.md # OpenAPI-first, schema validation -│ │ ├── testing.md # test requirements per change type -│ │ ├── pr-standards.md # PR format, labels, artifact links -│ │ └── migrations.md # migration safety, forbidden operations -│ └── skills/ # reusable Claude Code skills -│ ├── spec-to-plan/ # generate plans from specs -│ ├── endpoint-generator/ # scaffold API endpoints -│ ├── safe-refactor/ # refactor with verification -│ ├── review-diff/ # review any diff -│ ├── issue-to-pr/ # end-to-end issue resolution -│ ├── extract-signals/ # signal extraction pipeline -│ └── run-eval/ # run evaluation suites -├── .mcp.json # MCP server connections (GitHub, Postgres) -├── app/ # FastAPI control plane -│ ├── main.py # app entry, middleware, router registration -│ ├── config.py # env-based settings (FOUNDRY_ prefix) -│ ├── deps.py # dependency injection -│ └── routes/ -│ ├── runs.py # run CRUD, cancel, retry -│ ├── reviews.py # independent review requests -│ ├── specs.py # spec-to-plan generation -│ ├── patches.py # patch application to worktrees -│ ├── batches.py # batch extraction jobs -│ ├── evals.py # evaluation suite runs -│ ├── worktrees.py # worktree cleanup -│ └── health.py # health check -├── foundry/ # orchestration core -│ ├── contracts/ # Pydantic models (strict, typed) -│ │ ├── shared.py # TaskType, RunState, MCPProfile, enums -│ │ ├── task_types.py # TaskRequest, PlanStep, PlanArtifact -│ │ ├── run_models.py # RunEvent, RunArtifact, RunResponse -│ │ ├── review_models.py # ReviewIssue, ReviewVerdict -│ │ ├── extraction_models.py # Evidence, ExtractionEvent, ExtractionResult -│ │ └── eval_models.py # EvalDefinition, EvalItemResult, EvalResult -│ ├── db/ -│ │ ├── engine.py # async SQLAlchemy engine setup -│ │ ├── models.py # ORM: Run, RunEvent, RunArtifact, Worktree, -│ │ │ # BatchJob, BatchItem, EvalRun, -│ │ │ # VerificationResult -│ │ └── queries/ # data access layer -│ │ ├── runs.py # run CRUD queries -│ │ ├── artifacts.py # artifact storage queries -│ │ ├── batches.py # batch job queries -│ │ └── evals.py # eval run queries -│ ├── orchestration/ -│ │ ├── run_engine.py # core state machine (14 states, transitions) -│ │ ├── agent_runner.py # Agent SDK wrapper for subagents -│ │ ├── model_router.py # task-type → agent-role → model mapping -│ │ └── prompt_templates.py # system prompts per agent/task -│ ├── git/ -│ │ ├── worktree.py # create, list, cleanup worktrees -│ │ ├── branch.py # foundry/{task-type}-{description} naming -│ │ └── pr.py # PR creation via GitHub API -│ ├── providers/ -│ │ ├── claude_agent.py # Claude Agent SDK integration -│ │ ├── claude_messages.py # Claude Messages API integration -│ │ ├── claude_batch.py # Claude Message Batches API (bulk extraction) -│ │ └── github.py # GitHub REST client (PRs, comments, labels) -│ ├── tasks/ # task type implementations -│ │ ├── endpoint_build.py # build new API endpoints -│ │ ├── feature_slice.py # implement feature slices -│ │ ├── bug_fix.py # diagnose and fix bugs -│ │ ├── refactor.py # code refactoring -│ │ ├── migration_plan.py # database migration planning -│ │ ├── extraction_batch.py # batch signal extraction -│ │ ├── eval_run.py # evaluation suite execution -│ │ └── review_diff.py # standalone diff review -│ ├── verification/ -│ │ ├── runner.py # dispatch verification by file type -│ │ ├── go_verify.py # go build, go vet, go test -│ │ ├── ts_verify.py # tsc, eslint, next build -│ │ └── schema_verify.py # OpenAPI + JSON Schema validation -│ └── storage/ -│ ├── artifact_store.py # write/read artifacts to object storage -│ └── log_store.py # structured run event logging -├── workers/ # background task consumers -│ ├── run_worker.py # picks tasks from Redis, executes runs -│ ├── batch_worker.py # polls Anthropic Batch API, stores results -│ └── cleanup_worker.py # periodic worktree/artifact cleanup -├── hooks/ # deterministic enforcement scripts -│ ├── pre_tool_use/ -│ │ ├── block_secrets.sh # deny read/write to secret files -│ │ ├── block_protected_paths.sh # guard migrations/, auth/, infra/ -│ │ └── require_plan.sh # block edits without a stored plan -│ └── post_tool_use/ -│ ├── verify_after_edit.sh # run verification after file edits -│ └── log_tool_call.sh # log every tool invocation -├── evals/ # evaluation framework -│ ├── runner.py # eval orchestration -│ └── scorers/ -│ ├── extraction_scorer.py # score extraction accuracy -│ ├── evidence_scorer.py # score evidence quality -│ └── state_scorer.py # score state inference -├── canon/ # source of truth (shared with unicorn-app) -│ ├── docs/ # domain documentation -│ └── schemas/ # JSON Schemas for domain objects -├── scripts/ -│ ├── run_task.py # CLI task submission -│ ├── seed_db.py # seed database with test data -│ └── export_artifacts.py # export artifacts for inspection -├── tests/ # unit + integration tests + +The long-term architecture should not require Claude, GitHub, source code, or Unicorn. Those are properties of the first implementation, not invariants of UCF. + +## Architectural Invariants + +1. **State is explicit.** The system should not depend on a model reconstructing its entire operating reality from a prompt. +2. **Actions produce transitions.** Work is understood as movement from a known state to a resulting state. +3. **Evidence survives inference.** Claims about what happened should remain traceable to observations and artifacts. +4. **Execution is controlled.** Intelligence proposes or performs actions inside explicit boundaries. +5. **Verification is separate from generation.** Producing an action and establishing that it worked are different operations. +6. **History is durable.** Meaningful transitions leave events and artifacts behind. +7. **Models are replaceable participants.** UCF should not depend conceptually on one provider, model family, or reasoning architecture. +8. **Continuity is a systems property.** Long-horizon coherence comes from the loop around intelligence as well as from intelligence itself. + +## Current Direction + +This repository is being reopened as UCF rather than maintained as an active Unicorn Foundry product. + +A provider-neutral vNext foundation now lives alongside the historical Foundry runtime: + +- `IntelligenceProvider` separates orchestration from a concrete model vendor; +- `ExecutionEnvironment` separates isolated execution from Git worktrees; +- provider-neutral transition contracts represent state, evidence, actions, observations, verification, and outcomes; +- `TransitionEngine` closes a minimal state → action → observation → verification → outcome loop; +- `TransitionJournal` requires the verified outcome to survive the call; +- tests exercise that loop with a non-Unicorn environment and fake components. + +This does **not** mean the legacy Foundry runtime has already been generalized. `RunEngine`, verification, prompts, PR handling, persistence names, and parts of configuration remain software/Git/Claude-shaped. They will be migrated incrementally after the generic boundary is proven. + +See [RETROSPECTIVE.md](RETROSPECTIVE.md) for the present-day interpretation, [docs/runtime-decoupling-audit.md](docs/runtime-decoupling-audit.md) for the migration map, and [docs/architecture.md](docs/architecture.md) for the original Foundry architecture specification. + +## Repository Map + +```text +UCf/ +├── README.md # current UCF thesis and status +├── RETROSPECTIVE.md # historical interpretation boundary +├── CLAUDE.md # guidance for future agents/engineering ├── docs/ -│ └── architecture.md # full architecture specification -├── Dockerfile # Python 3.12-slim production image -├── docker-compose.yml # Postgres, Redis, app, workers -└── pyproject.toml # dependencies, ruff, mypy, pytest config +│ ├── ucf-architecture.md # generalized architecture mapping +│ ├── runtime-decoupling-audit.md # migration map and remaining coupling +│ └── architecture.md # original Unicorn Foundry specification +├── foundry/ +│ ├── contracts/ +│ │ ├── transition_models.py # state, evidence, action, outcome contracts +│ │ └── ... # historical Foundry contracts +│ ├── runtime/ +│ │ ├── interfaces.py # observer/planner/executor/verifier/journal +│ │ └── transition_engine.py # provider-neutral state-transition loop +│ ├── environments/ +│ │ ├── base.py # ExecutionEnvironment contract +│ │ └── git_worktree.py # first concrete environment adapter +│ ├── providers/ +│ │ ├── base.py # IntelligenceProvider contract +│ │ └── claude_*.py # historical/default Claude adapters +│ ├── orchestration/ # historical Foundry run machinery +│ ├── verification/ # historical deterministic code verification +│ ├── git/ # historical Git/PR action surface +│ ├── tasks/ # historical Foundry task implementations +│ ├── db/ # historical run persistence +│ └── storage/ # historical artifact persistence +├── app/ # historical FastAPI control plane +├── workers/ # historical background workers +├── canon/ # historical Unicorn domain contracts +├── hooks/ # historical deterministic safeguards +├── tests/ +│ └── unit/runtime/ # provider-neutral transition-loop tests +└── .github/workflows/ + └── ucf-foundation.yml # compile, lint, foundation + regression tests ``` -## Task Types +The repository intentionally contains both the generalized UCF foundation and the historical Foundry implementation. The historical directories are not being renamed away until their behavior has been migrated through exercised UCF interfaces. + +## Historical Foundry Runtime + +The sections below document the original concrete runtime. They are retained because they show how the systems problem was first implemented; they should not be read as requirements of the generalized UCF architecture. + +## Historical Foundry Task Types + | Task Type | Description | Model Routing | |-----------|-------------|---------------| @@ -220,7 +223,7 @@ unicorn-foundry/ | `eval_run` | Run evaluation suites against model outputs | Sonnet (evaluate) | | `canon_update` | Update shared schemas and domain docs | Opus (plan/review), Sonnet (impl) | -## Run Lifecycle +## Historical Foundry Run Lifecycle ``` queued → creating_worktree → planning → implementing → verifying @@ -231,7 +234,7 @@ Failure states: `plan_failed`, `verification_failed`, `review_failed`, `cancelle Failed runs in `plan_failed`, `verification_failed`, or `review_failed` can be retried (transitions back to `queued`). Any non-terminal run can be cancelled. -## API Endpoints +## Historical Foundry API Endpoints | Method | Path | Description | |--------|------|-------------| @@ -252,7 +255,7 @@ Failed runs in `plan_failed`, `verification_failed`, or `review_failed` can be r | `GET` | `/v1/evals/{id}` | Get eval results | | `POST` | `/v1/worktrees/cleanup` | Clean up stale worktrees | -## Subagents +## Historical Foundry Subagents | Agent | Role | Model | |-------|------|-------| @@ -264,7 +267,7 @@ Failed runs in `plan_failed`, `verification_failed`, or `review_failed` can be r | **Migration Guard** | High-scrutiny review for migrations/auth/infra | Opus | | **Repo Explorer** | Read-only codebase reconnaissance | Haiku | -## Claude Code Skills +## Historical Foundry Claude Code Skills | Skill | Description | |-------|-------------| @@ -276,7 +279,7 @@ Failed runs in `plan_failed`, `verification_failed`, or `review_failed` can be r | `extract-signals` | Run signal extraction pipeline | | `run-eval` | Execute evaluation suites | -## Hooks (Deterministic Enforcement) +## Historical Foundry Hooks (Deterministic Enforcement) | Hook | Trigger | Purpose | |------|---------|---------| @@ -286,7 +289,7 @@ Failed runs in `plan_failed`, `verification_failed`, or `review_failed` can be r | `verify_after_edit.sh` | Post: Edit, Write | Run verification after file modifications | | `log_tool_call.sh` | Post: all tools | Log every tool invocation for auditability | -## MCP Profiles +## Historical Foundry MCP Profiles Runs can be scoped to specific MCP server access: @@ -298,7 +301,7 @@ Runs can be scoped to specific MCP server access: | `research_full` | GitHub + Postgres (read-only) | Full research capabilities | | `app_build_minimal` | GitHub | Minimal build access | -## Model Routing +## Historical Foundry Model Routing | Model | Use Case | |-------|----------| @@ -308,7 +311,7 @@ Runs can be scoped to specific MCP server access: Routing is defined in `foundry/orchestration/model_router.py`. Override via `model_override` in task requests when justified. -## Database Schema +## Historical Foundry Database Schema PostgreSQL tables managed via Alembic: @@ -411,7 +414,7 @@ python scripts/export_artifacts.py --run-id | Type Checking | mypy (strict mode) | | Testing | pytest + pytest-asyncio | -## Language Boundaries +## Historical Foundry Language Boundaries - **Python** — this repo (Foundry). All orchestration, extraction, eval code. - **Go** — unicorn-app backend. Foundry writes Go code into unicorn-app via PRs. diff --git a/RETROSPECTIVE.md b/RETROSPECTIVE.md new file mode 100644 index 0000000..a72528b --- /dev/null +++ b/RETROSPECTIVE.md @@ -0,0 +1,120 @@ +# UCF Retrospective + +## Why this document exists + +UCF did not begin as an attempt to propose a general theory of machine intelligence. + +It began as an engineering response to a practical problem. + +While building Unicorn, I was trying to make a changing environment legible to software through signals, evidence, and explicit state. Once that representation existed, another problem appeared: a capable model could still behave discontinuously across time. + +It could produce a good plan without preserving why the plan existed. It could make a change without establishing whether the environment actually reached the intended state. It could observe an outcome without turning that outcome into durable state for the next operation. + +UCF, then called Unicorn Foundry, was an attempt to build the infrastructure around that boundary. + +## The original split + +The original conceptual split was approximately: + +```text +Unicorn +world → signals → evidence → state → legibility + +UCF +state → intent → plan → execution → verification → outcome +``` + +The useful unit was therefore larger than a model call. + +```text +STATE(t) + ↓ +reason + ↓ +intent + ↓ +action + ↓ +consequence + ↓ +observation + ↓ +evidence + ↓ +STATE(t+1) +``` + +The implementation expressed only part of this larger loop, and it expressed that part through the concrete environment available at the time: software repositories, Claude, Git worktrees, pull requests, tests, schemas, artifacts, PostgreSQL, Redis, and deterministic hooks. + +## What was actually built + +The repository contains real implementations of several mechanisms that matter to the original question: + +- explicit run states and legal transitions; +- persistent events and run metadata; +- isolated worktrees; +- structured plan artifacts; +- implementation and provider boundaries; +- deterministic verification; +- independent review; +- durable artifacts; +- cancellation, retry, and failure paths; +- evidence/extraction contracts; +- evaluation scaffolding. + +It also contains planned or partial surfaces. Some task paths remain unimplemented, and the architecture document describes more than the runtime currently completes. + +That distinction matters. This retrospective is not intended to make the historical implementation appear more complete than it was. + +## What I understand differently now + +The most useful way I now understand the experiment is as a distinction between **intelligence at an instant** and **intelligence through time**. + +A capable model can map context to an impressive output. Persistent operation asks for more: + +- What is currently believed to be true? +- What evidence supports that state? +- What changed since the previous operation? +- What is the intended transition? +- What action was actually taken? +- What consequence followed? +- How was the consequence verified? +- What should become durable state for the next operation? + +Those questions are systems questions even when an intelligent model participates in answering them. + +This is why UCF should not be architecturally synonymous with Claude, agents, coding, or GitHub. Those were components of its first environment. + +## What UCF is not + +UCF is not presented as a learned world model. + +It is not a claim that software orchestration solves intelligence. + +It is not a claim that explicit state can replace learned representations, perception, planning, or adaptation. + +It is not a retrospective claim to ideas that were not present in the original work. + +The narrower claim is historical and architectural: + +> While trying to build a system that could operate coherently for longer than a single model interaction, I independently encountered the problem that model capability alone did not provide state, consequence, verification, or continuity. UCF was an attempt to build some of that surrounding infrastructure. + +## Why reopen it + +Unicorn has been discontinued, but the problem that produced UCF remains useful. + +The next phase is therefore not to revive Unicorn. It is to extract the general mechanisms from the historical Foundry implementation and ask which of them survive when the original assumptions are removed. + +The working question is: + +> **What must exist around an intelligent model for it to remain coherent while the environment it operates in changes?** + +The implementation should earn increasingly general answers to that question rather than assuming them in advance. + +## Preservation rule + +The existing Git history is part of the evidence. + +Historical commits should remain intact. New documentation should distinguish original implementation, later interpretation, and future direction rather than rewriting one as another. + +UCF should become more general by evolution, not by pretending it always was. diff --git a/docs/runtime-decoupling-audit.md b/docs/runtime-decoupling-audit.md new file mode 100644 index 0000000..1698e23 --- /dev/null +++ b/docs/runtime-decoupling-audit.md @@ -0,0 +1,76 @@ +# Runtime Decoupling Audit + +This audit separates the general UCF mechanism from assumptions inherited from Unicorn Foundry. + +## Already generalized in this branch + +### Intelligence provider boundary + +`AgentRunner` now depends on an `IntelligenceProvider` protocol. Claude remains the default historical implementation, but the orchestration boundary no longer requires the concrete Claude provider type. + +### Execution target boundary + +`TaskRequest.repo` remains named for backwards compatibility, but it is no longer restricted to `unicorn-app` or `unicorn-foundry`. + +### Environment boundary + +`ExecutionEnvironment` defines prepare, observe-changes, and cleanup operations. `GitWorktreeEnvironment` adapts the historical worktree implementation to that interface. + +### Transition vocabulary + +`foundry/contracts/transition_models.py` defines provider-neutral state snapshots, evidence references, action proposals, transition requests, observations, verification decisions, and outcomes without assuming GitHub or Unicorn. + +### Generic transition loop + +`TransitionEngine` now coordinates a minimal provider-neutral loop through injected observer, planner, executor, verifier, environment, and journal capabilities. The engine does not know about Claude, GitHub, Go, TypeScript, or Unicorn. + +The historical `RunEngine` is still unchanged and Git/PR-shaped. The new engine is a parallel foundation, not a claim that migration is complete. + +## Remaining historical couplings + +| Coupling | Current form | General form | Migration priority | +| --- | --- | --- | --- | +| Lifecycle terminal | `PR_OPENED -> COMPLETED` | outcome recorded / accepted | P0 | +| Action environment | Git worktree | ExecutionEnvironment | P0 | +| Action result | Git diff + PR | environment-specific action artifact | P0 | +| Implementer role | Go / TypeScript literal | executor capability | P1 | +| Verification | Go/TS/schema commands | verifier plugins | P1 | +| Model routing | Claude model IDs | provider + capability routing | P1 | +| Prompt layer | coding-specific planner/implementer prompts | transition-role prompts | P1 | +| Canon | Unicorn/startup schemas | environment-specific state contracts | P2 | +| Extraction | startup signal extraction | observer adapters | P2 | +| Config names | FOUNDRY, unicorn_app_* | UCF + legacy aliases | P2 | +| Persistence names | runs, PR URL, worktrees | transitions, outcomes, workspaces | P3 | +| API routes | /runs, /patches, /worktrees | /transitions, /actions, /environments | P3 | + +## P0 foundation status + +The generic loop now exists alongside Foundry: + +1. an execution environment prepares and cleans up an isolated workspace; +2. an observer establishes explicit before-state; +3. a planner proposes a provider-neutral action; +4. an executor applies the action; +5. the observer establishes after-state; +6. an independent verifier accepts or rejects the transition; +7. a journal records the verified outcome; +8. the resulting state can seed the next transition. + +`tests/unit/runtime/test_transition_engine.py` exercises this with a non-Unicorn environment and fake capabilities. + +The next P0 task is **adapter migration**: make the historical Git/Claude workflow exercise these interfaces rather than maintaining a separate architectural path. In particular, `PR_OPENED -> COMPLETED` must stop being the general definition of a successful transition. + +## What should not be renamed yet + +Do not mass-rename Foundry classes, database tables, routes, or artifact types merely to match the new vocabulary. Renaming before the generalized loop works would create churn without increasing capability. + +Keep the historical runtime operational while new interfaces are introduced alongside it. Once a non-Unicorn closed loop passes end to end, migrate internals incrementally. + +## Foundation boundary + +The repository now contains the code-level boundary required to test UCF independently from Unicorn. Because the historical runtime is not yet routed through `TransitionEngine`, the project is currently in a dual state: + +- **general UCF foundation:** provider/environment/transition interfaces plus a minimal loop; +- **historical Foundry runtime:** the working Git/PR orchestration implementation. + +The next milestone is to make those two paths converge without erasing the original implementation history. diff --git a/docs/ucf-architecture.md b/docs/ucf-architecture.md new file mode 100644 index 0000000..acd2204 --- /dev/null +++ b/docs/ucf-architecture.md @@ -0,0 +1,40 @@ +# Historical Architecture Note + +The current runtime and architecture specification were written for Unicorn Foundry, the original implementation of UCF. They intentionally retain the original terminology. + +## Generalized boundary + +Environment -> Evidence + State(t) -> Intelligence -> Controlled Transition -> Outcome + Evidence -> State(t+1) -> repeat. + +## Mapping from Foundry + +| Historical Foundry concept | General UCF concept | +| --- | --- | +| TaskRequest | requested transition / intent | +| RunState | execution-state machine | +| PlanArtifact | proposed transition plan | +| Git worktree | isolated action environment | +| implementer agent | action executor | +| verification runner | deterministic transition validator | +| blind reviewer | independent evaluator | +| RunEvent | transition event | +| RunArtifact | durable evidence | +| PR | accepted/publishable action outcome | +| Claude provider | intelligence provider | +| Unicorn canon | environment/state contracts | + +This mapping is interpretive. It describes the direction of the reopened project; it does not claim the historical runtime already provides a domain-independent implementation. + +## Migration principle + +Generalization should proceed from the outside inward: + +1. establish the UCF thesis and vocabulary; +2. preserve the historical implementation; +3. identify Unicorn-specific assumptions; +4. introduce provider/environment interfaces where evidence justifies them; +5. migrate one closed-loop example end to end; +6. evaluate continuity, transition correctness, recovery, and state accuracy; +7. remove historical coupling only when replacement abstractions are exercised. + +The goal is not abstraction for its own sake. The goal is to discover which mechanisms are genuinely necessary for coherent machine operation through time. diff --git a/foundry/contracts/task_types.py b/foundry/contracts/task_types.py index 1e6c65c..30866c8 100644 --- a/foundry/contracts/task_types.py +++ b/foundry/contracts/task_types.py @@ -14,10 +14,17 @@ class TaskRequest(FoundryBaseModel): - """A task submitted to Foundry for execution.""" + """A requested controlled transition. + + "repo" is retained for backwards compatibility with the Git-based first + implementation, but is no longer restricted to Unicorn repositories. + """ task_type: TaskType - repo: Literal["unicorn-app", "unicorn-foundry"] + repo: str = Field( + min_length=1, + description="Execution target identifier; historically a Git repository", + ) base_branch: str = "main" title: str prompt: str @@ -32,7 +39,6 @@ class TaskRequest(FoundryBaseModel): priority: int = 0 metadata: dict = Field(default_factory=dict) - class PlanStep(FoundryBaseModel): """A single step in an implementation plan.""" diff --git a/foundry/contracts/transition_models.py b/foundry/contracts/transition_models.py new file mode 100644 index 0000000..369bca4 --- /dev/null +++ b/foundry/contracts/transition_models.py @@ -0,0 +1,81 @@ +"""Provider-neutral contracts for stateful UCF transitions. + +These models describe the conceptual boundary of UCF without assuming Git, +source code, pull requests, or a particular intelligence provider. +""" + +from __future__ import annotations + +from datetime import datetime +from uuid import UUID, uuid4 + +from pydantic import Field + +from foundry.contracts.shared import FoundryBaseModel + + +class EvidenceRef(FoundryBaseModel): + """Reference to evidence supporting a state or transition claim.""" + + kind: str + uri: str + description: str | None = None + checksum: str | None = None + + +class StateSnapshot(FoundryBaseModel): + """Explicit representation of an environment at a point in time.""" + + id: UUID = Field(default_factory=uuid4) + environment: str + observed_at: datetime + state: dict + evidence: list[EvidenceRef] = Field(default_factory=list) + + +class TransitionRequest(FoundryBaseModel): + """Request to move an environment from a known state toward an objective.""" + + id: UUID = Field(default_factory=uuid4) + environment: str + objective: str + before_state: StateSnapshot | None = None + constraints: dict = Field(default_factory=dict) + metadata: dict = Field(default_factory=dict) + + +class TransitionObservation(FoundryBaseModel): + """Observed consequence of an attempted transition.""" + + transition_id: UUID + observed_state: StateSnapshot + action_evidence: list[EvidenceRef] = Field(default_factory=list) + verification_evidence: list[EvidenceRef] = Field(default_factory=list) + + +class TransitionOutcome(FoundryBaseModel): + """Durable result of a transition after verification/evaluation.""" + + transition_id: UUID + accepted: bool + before_state: StateSnapshot | None = None + after_state: StateSnapshot | None = None + observation: TransitionObservation | None = None + reason: str | None = None + metadata: dict = Field(default_factory=dict) + + +class ActionProposal(FoundryBaseModel): + """Provider-neutral action proposed for an environment.""" + + kind: str + description: str + payload: dict = Field(default_factory=dict) + + +class VerificationDecision(FoundryBaseModel): + """Independent judgment of whether the requested transition succeeded.""" + + accepted: bool + reason: str | None = None + evidence: list[EvidenceRef] = Field(default_factory=list) diff --git a/foundry/environments/__init__.py b/foundry/environments/__init__.py new file mode 100644 index 0000000..f4ddabd --- /dev/null +++ b/foundry/environments/__init__.py @@ -0,0 +1 @@ +"""Execution-environment abstractions for UCF.""" diff --git a/foundry/environments/base.py b/foundry/environments/base.py new file mode 100644 index 0000000..9710088 --- /dev/null +++ b/foundry/environments/base.py @@ -0,0 +1,33 @@ +"""Provider-neutral execution environment contracts. + +An execution environment is the boundary in which a requested state transition +is prepared, performed, observed, and cleaned up. Git worktrees are the first +implementation, not the definition of the abstraction. +""" + +from __future__ import annotations + +from typing import Protocol +from uuid import UUID + + +class ExecutionEnvironment(Protocol): + """Minimal environment lifecycle required by a controlled transition.""" + + async def prepare( + self, + target: str, + base_ref: str, + transition_id: UUID, + transition_name: str, + ) -> str: + """Prepare an isolated workspace and return its location/identifier.""" + ... + + async def observe_changes(self, workspace: str) -> str: + """Return a durable representation of changes produced in the workspace.""" + ... + + async def cleanup(self, workspace: str) -> None: + """Release resources associated with the workspace.""" + ... diff --git a/foundry/environments/git_worktree.py b/foundry/environments/git_worktree.py new file mode 100644 index 0000000..5060f12 --- /dev/null +++ b/foundry/environments/git_worktree.py @@ -0,0 +1,42 @@ +"""Git worktree implementation of the UCF execution-environment contract.""" + +from __future__ import annotations + +import asyncio +from uuid import UUID + +from foundry.git.worktree import WorktreeManager + + +class GitWorktreeEnvironment: + """Adapt the historical WorktreeManager to the general environment boundary.""" + + def __init__(self, manager: WorktreeManager) -> None: + self.manager = manager + + async def prepare( + self, + target: str, + base_ref: str, + transition_id: UUID, + transition_name: str, + ) -> str: + # WorktreeManager currently creates from HEAD; base_ref remains explicit + # in the interface so a future implementation can honor arbitrary refs. + del base_ref + return await self.manager.create(target, transition_name, transition_id) + + async def observe_changes(self, workspace: str) -> str: + proc = await asyncio.create_subprocess_exec( + "git", + "diff", + "HEAD", + cwd=workspace, + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + ) + stdout, _ = await proc.communicate() + return stdout.decode(errors="replace") + + async def cleanup(self, workspace: str) -> None: + await self.manager.cleanup(workspace) diff --git a/foundry/orchestration/agent_runner.py b/foundry/orchestration/agent_runner.py index e073703..a3bf045 100644 --- a/foundry/orchestration/agent_runner.py +++ b/foundry/orchestration/agent_runner.py @@ -1,20 +1,21 @@ -"""Agent SDK wrapper for running subagents. +"""Provider-neutral runner for planning, implementation, review, and exploration. -Wraps the Claude Agent SDK to execute planner, implementer, reviewer, -and extractor subagents with appropriate tool access and system prompts. +Claude remains the default backend for the historical implementation, but the +runner depends on the IntelligenceProvider contract rather than a vendor type. """ from __future__ import annotations import asyncio import logging -from typing import Any, Literal +from typing import Literal from foundry.contracts.review_models import ReviewVerdict from foundry.contracts.shared import TaskType from foundry.contracts.task_types import PlanArtifact, TaskRequest from foundry.orchestration import prompt_templates from foundry.orchestration.model_router import resolve_model +from foundry.providers.base import IntelligenceProvider from foundry.providers.claude_agent import ClaudeAgentProvider logger = logging.getLogger(__name__) @@ -33,15 +34,20 @@ class AgentRunner: - """Wraps the Anthropic Agent SDK to run subagents with specific roles. + """Run role-specific operations through a replaceable intelligence provider. - Each subagent gets its own context window, tool access list, and system - prompt. The runner handles serialization of structured outputs and - validation against expected schemas. + Each role gets its own context, tool access, and system prompt. Claude is + the backwards-compatible default, not an architectural requirement. """ - def __init__(self, api_key: str | None = None) -> None: - self.provider = ClaudeAgentProvider(api_key=api_key) + def __init__( + self, + api_key: str | None = None, + provider: IntelligenceProvider | None = None, + ) -> None: + self.provider: IntelligenceProvider = ( + provider if provider is not None else ClaudeAgentProvider(api_key=api_key) + ) async def run_agent( self, @@ -52,9 +58,9 @@ async def run_agent( output_schema: type | None = None, worktree_path: str | None = None, ) -> dict: - """Execute a Claude subagent with the given configuration. + """Execute a role through the configured intelligence provider. - Delegates to the ClaudeAgentProvider. When output_schema is provided, + When output_schema is provided, uses run_with_structured_output for validated JSON responses. Otherwise uses run for free-form text/JSON responses. diff --git a/foundry/orchestration/run_engine.py b/foundry/orchestration/run_engine.py index 54658a4..6a6e90e 100644 --- a/foundry/orchestration/run_engine.py +++ b/foundry/orchestration/run_engine.py @@ -17,9 +17,8 @@ from typing import TYPE_CHECKING from uuid import UUID -from sqlalchemy.ext.asyncio import AsyncSession - import redis.asyncio as aioredis +from sqlalchemy.ext.asyncio import AsyncSession from foundry.contracts.review_models import ReviewIssue, ReviewVerdict from foundry.contracts.run_models import RunResponse @@ -343,7 +342,11 @@ async def execute_run(self, task_request: TaskRequest) -> RunResponse: state_at_failure = run.state current_state = RunState(run.state) # Abbreviate traceback for metadata (last 1500 chars) - abbreviated_tb = full_traceback[-1500:] if len(full_traceback) > 1500 else full_traceback + abbreviated_tb = ( + full_traceback[-1500:] + if len(full_traceback) > 1500 + else full_traceback + ) if RunState.ERRORED in VALID_TRANSITIONS.get(current_state, set()): await self._transition( run_id, current_state, RunState.ERRORED, @@ -396,7 +399,11 @@ async def execute_run(self, task_request: TaskRequest) -> RunResponse: try: await self.worktree_manager.cleanup(worktree_path) except Exception: - logger.warning("Failed to clean up worktree at %s for run %s", worktree_path, run_id) + logger.warning( + "Failed to clean up worktree at %s for run %s", + worktree_path, + run_id, + ) async def cancel_run(self, run_id: UUID) -> RunResponse: """Cancel an in-progress run. @@ -457,9 +464,13 @@ async def retry_run(self, run_id: UUID) -> RunResponse: current_state = RunState(run.state) if current_state not in _RETRYABLE_STATES: + retryable_states = ", ".join( + state.value + for state in sorted(_RETRYABLE_STATES, key=lambda state: state.value) + ) raise ValueError( f"Run {run_id} in state {current_state.value} is not retryable. " - f"Only runs in {', '.join(s.value for s in sorted(_RETRYABLE_STATES, key=lambda s: s.value))} can be retried." + f"Only runs in {retryable_states} can be retried." ) await self._transition( @@ -768,7 +779,11 @@ async def _run_implementation( """ # Determine model and language for implementation from foundry.orchestration.model_router import resolve_model - impl_model = resolve_model(task_request.task_type, "implementer", task_request.model_override) + impl_model = resolve_model( + task_request.task_type, + "implementer", + task_request.model_override, + ) language = "go" # Go-only for Phase 1 # Transition to IMPLEMENTING @@ -1062,7 +1077,11 @@ async def _run_review( """ from foundry.orchestration.model_router import resolve_model - review_model = resolve_model(task_request.task_type, "reviewer", task_request.model_override) + review_model = resolve_model( + task_request.task_type, + "reviewer", + task_request.model_override, + ) # 1. Transition to REVIEWING await self._transition( @@ -1164,7 +1183,8 @@ async def _run_review( issue_count = len(review.issues) await self._add_event( run_id, RunState.REVIEWING, - f"Review verdict: {review.verdict.value}, {issue_count} issues", + f"Review complete: {review.verdict.value}, issues: {issue_count}, " + f"confidence: {review.confidence}", metadata={ "artifact": "review.json", "verdict": review.verdict.value, diff --git a/foundry/providers/base.py b/foundry/providers/base.py new file mode 100644 index 0000000..166d80c --- /dev/null +++ b/foundry/providers/base.py @@ -0,0 +1,37 @@ +"""Provider-neutral contract for UCF intelligence backends. + +The historical implementation uses Claude, but orchestration should depend on +capabilities rather than a concrete model vendor. +""" + +from __future__ import annotations + +from typing import Protocol, runtime_checkable + + +@runtime_checkable +class IntelligenceProvider(Protocol): + """Minimal capability contract required by AgentRunner.""" + + async def run( + self, + system_prompt: str, + user_message: str, + model: str, + tools: list[str] | None = None, + working_directory: str | None = None, + ) -> dict: + """Execute an unstructured reasoning/action request.""" + ... + + async def run_with_structured_output( + self, + system_prompt: str, + user_message: str, + model: str, + output_schema: type, + tools: list[str] | None = None, + working_directory: str | None = None, + ) -> dict: + """Execute a request whose response must validate against a schema.""" + ... diff --git a/foundry/runtime/__init__.py b/foundry/runtime/__init__.py new file mode 100644 index 0000000..91a93de --- /dev/null +++ b/foundry/runtime/__init__.py @@ -0,0 +1 @@ +"""Provider-neutral UCF transition runtime.""" diff --git a/foundry/runtime/interfaces.py b/foundry/runtime/interfaces.py new file mode 100644 index 0000000..9b12ac7 --- /dev/null +++ b/foundry/runtime/interfaces.py @@ -0,0 +1,66 @@ +"""Capability interfaces for the provider-neutral UCF transition loop.""" + +from __future__ import annotations + +from typing import Protocol + +from foundry.contracts.transition_models import ( + ActionProposal, + EvidenceRef, + StateSnapshot, + TransitionOutcome, + TransitionRequest, + VerificationDecision, +) + + +class StateObserver(Protocol): + """Observe an environment and return explicit state.""" + + async def observe(self, request: TransitionRequest, workspace: str) -> StateSnapshot: + ... + + +class TransitionPlanner(Protocol): + """Propose an action from objective plus current state.""" + + async def plan( + self, + request: TransitionRequest, + current_state: StateSnapshot, + workspace: str, + ) -> ActionProposal: + ... + + +class ActionExecutor(Protocol): + """Apply an action proposal inside an execution environment.""" + + async def execute( + self, + request: TransitionRequest, + action: ActionProposal, + workspace: str, + ) -> list[EvidenceRef]: + ... + + +class TransitionVerifier(Protocol): + """Judge the observed consequence independently from action generation.""" + + async def verify( + self, + request: TransitionRequest, + before: StateSnapshot, + after: StateSnapshot, + action: ActionProposal, + action_evidence: list[EvidenceRef], + ) -> VerificationDecision: + ... + + +class TransitionJournal(Protocol): + """Persist the verified outcome so continuity does not depend on call context.""" + + async def record(self, outcome: TransitionOutcome) -> None: + ... diff --git a/foundry/runtime/transition_engine.py b/foundry/runtime/transition_engine.py new file mode 100644 index 0000000..b68957d --- /dev/null +++ b/foundry/runtime/transition_engine.py @@ -0,0 +1,86 @@ +"""Minimal provider-neutral state transition engine. + +This runs alongside the historical Foundry RunEngine. It exists to prove the +general UCF loop before legacy Git/PR terminology is migrated. +""" + +from __future__ import annotations + +from foundry.contracts.transition_models import ( + TransitionObservation, + TransitionOutcome, + TransitionRequest, +) +from foundry.environments.base import ExecutionEnvironment +from foundry.runtime.interfaces import ( + ActionExecutor, + StateObserver, + TransitionJournal, + TransitionPlanner, + TransitionVerifier, +) + + +class TransitionEngine: + """Coordinate one explicit state transition from observation to outcome.""" + + def __init__( + self, + environment: ExecutionEnvironment, + observer: StateObserver, + planner: TransitionPlanner, + executor: ActionExecutor, + verifier: TransitionVerifier, + journal: TransitionJournal, + ) -> None: + self.environment = environment + self.observer = observer + self.planner = planner + self.executor = executor + self.verifier = verifier + self.journal = journal + + async def execute(self, request: TransitionRequest) -> TransitionOutcome: + workspace = await self.environment.prepare( + target=request.environment, + base_ref=str(request.metadata.get("base_ref", "current")), + transition_id=request.id, + transition_name=str(request.metadata.get("transition_name", request.id)), + ) + + try: + before = request.before_state + if before is None: + before = await self.observer.observe(request, workspace) + + action = await self.planner.plan(request, before, workspace) + action_evidence = await self.executor.execute(request, action, workspace) + after = await self.observer.observe(request, workspace) + + decision = await self.verifier.verify( + request=request, + before=before, + after=after, + action=action, + action_evidence=action_evidence, + ) + + observation = TransitionObservation( + transition_id=request.id, + observed_state=after, + action_evidence=action_evidence, + verification_evidence=decision.evidence, + ) + outcome = TransitionOutcome( + transition_id=request.id, + accepted=decision.accepted, + before_state=before, + after_state=after, + observation=observation, + reason=decision.reason, + metadata={"action_kind": action.kind}, + ) + await self.journal.record(outcome) + return outcome + finally: + await self.environment.cleanup(workspace) diff --git a/pyproject.toml b/pyproject.toml index a10be66..2dc7071 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "unicorn-foundry" version = "0.1.0" -description = "Claude-powered orchestration system for Unicorn Protocol" +description = "UCF experiment in persistent machine operation across changing state" requires-python = ">=3.12" dependencies = [ "fastapi>=0.115.0", @@ -21,6 +21,7 @@ dependencies = [ [project.optional-dependencies] dev = [ + "aiosqlite>=0.20.0", "pytest>=8.3.0", "pytest-asyncio>=0.24.0", "pytest-cov>=5.0.0", @@ -32,6 +33,9 @@ dev = [ requires = ["hatchling"] build-backend = "hatchling.build" +[tool.hatch.build.targets.wheel] +packages = ["foundry", "app", "workers"] + [tool.ruff] target-version = "py312" line-length = 100 diff --git a/tests/unit/orchestration/test_agent_runner.py b/tests/unit/orchestration/test_agent_runner.py index 4257a7c..78d6fef 100644 --- a/tests/unit/orchestration/test_agent_runner.py +++ b/tests/unit/orchestration/test_agent_runner.py @@ -1,8 +1,4 @@ -"""Tests for AgentRunner — mock Claude client. - -Tests verify that run_agent and run_planner correctly delegate to the -ClaudeAgentProvider with expected arguments and handle responses properly. -""" +"""Tests for AgentRunner provider delegation and role behavior.""" from unittest.mock import AsyncMock, MagicMock from uuid import uuid4 @@ -11,8 +7,7 @@ from foundry.contracts.shared import Complexity, MCPProfile, TaskType from foundry.contracts.task_types import PlanArtifact, PlanStep, TaskRequest -from foundry.orchestration.agent_runner import AgentRunner, PLANNER_TOOLS - +from foundry.orchestration.agent_runner import PLANNER_TOOLS, AgentRunner # --------------------------------------------------------------------------- # Fixtures @@ -60,6 +55,12 @@ def runner() -> AgentRunner: return AgentRunner(api_key="test-key") +def test_runner_accepts_injected_provider() -> None: + provider = MagicMock() + runner = AgentRunner(provider=provider) + assert runner.provider is provider + + # --------------------------------------------------------------------------- # run_agent tests # --------------------------------------------------------------------------- diff --git a/tests/unit/orchestration/test_event_logging.py b/tests/unit/orchestration/test_event_logging.py index 1846eef..adacbb1 100644 --- a/tests/unit/orchestration/test_event_logging.py +++ b/tests/unit/orchestration/test_event_logging.py @@ -204,7 +204,7 @@ async def test_all_lifecycle_events_present( assert any("go_test passed" in m for m in messages) assert any("Verification passed" in m for m in messages) assert any("Blind review started" in m for m in messages) - assert any("Review verdict:" in m for m in messages) + assert any("Review complete:" in m for m in messages) assert any("PR #99 opened" in m for m in messages) assert any("Run completed successfully" in m for m in messages) @@ -357,7 +357,7 @@ async def test_review_events_have_model_and_verdict( assert review_started.model_used is not None assert "model" in review_started.metadata_ - review_done = next(e for e in events if "Review verdict:" in e.message) + review_done = next(e for e in events if "Review complete:" in e.message) assert "approve" in review_done.metadata_["verdict"] assert review_done.metadata_["issue_count"] == 0 assert review_done.duration_ms is not None @@ -519,7 +519,7 @@ async def test_review_rejection_emits_review_events( messages = [e.message for e in events] assert any("Blind review started" in m for m in messages) - assert any("Review verdict: reject" in m for m in messages) + assert any("Review complete: reject" in m for m in messages) # The transition event to review_failed should have verdict metadata fail_events = [e for e in events if e.state == "review_failed"] diff --git a/tests/unit/runtime/test_transition_engine.py b/tests/unit/runtime/test_transition_engine.py new file mode 100644 index 0000000..f4f13e9 --- /dev/null +++ b/tests/unit/runtime/test_transition_engine.py @@ -0,0 +1,154 @@ +"""Tests for the provider-neutral UCF transition engine.""" + +from datetime import UTC, datetime +from uuid import UUID + +from foundry.contracts.transition_models import ( + ActionProposal, + EvidenceRef, + StateSnapshot, + TransitionOutcome, + TransitionRequest, + VerificationDecision, +) +from foundry.runtime.transition_engine import TransitionEngine + + +class FakeEnvironment: + def __init__(self) -> None: + self.cleaned = False + + async def prepare( + self, + target: str, + base_ref: str, + transition_id: UUID, + transition_name: str, + ) -> str: + assert target == "warehouse-robot" + assert base_ref == "current" + assert transition_name + return f"memory://{transition_id}" + + async def observe_changes(self, workspace: str) -> str: + return "" + + async def cleanup(self, workspace: str) -> None: + self.cleaned = True + + +class FakeObserver: + def __init__(self) -> None: + self.calls = 0 + + async def observe(self, request: TransitionRequest, workspace: str) -> StateSnapshot: + self.calls += 1 + location = "dock-a" if self.calls == 1 else "dock-b" + return StateSnapshot( + environment=request.environment, + observed_at=datetime.now(UTC), + state={"location": location}, + ) + + +class FakePlanner: + async def plan( + self, + request: TransitionRequest, + current_state: StateSnapshot, + workspace: str, + ) -> ActionProposal: + assert current_state.state["location"] == "dock-a" + return ActionProposal( + kind="move", + description="Move the robot to dock-b", + payload={"destination": "dock-b"}, + ) + + +class FakeExecutor: + async def execute( + self, + request: TransitionRequest, + action: ActionProposal, + workspace: str, + ) -> list[EvidenceRef]: + assert action.payload["destination"] == "dock-b" + return [ + EvidenceRef( + kind="action-log", + uri=f"{workspace}/actions/1", + description="Move command issued", + ) + ] + + +class FakeJournal: + def __init__(self) -> None: + self.outcomes: list[TransitionOutcome] = [] + + async def record(self, outcome: TransitionOutcome) -> None: + self.outcomes.append(outcome) + + +class FakeVerifier: + async def verify( + self, + request: TransitionRequest, + before: StateSnapshot, + after: StateSnapshot, + action: ActionProposal, + action_evidence: list[EvidenceRef], + ) -> VerificationDecision: + accepted = ( + before.state["location"] == "dock-a" + and after.state["location"] == "dock-b" + and bool(action_evidence) + ) + return VerificationDecision( + accepted=accepted, + reason="Destination observed" if accepted else "Destination not observed", + evidence=[ + EvidenceRef( + kind="verification", + uri="memory://verification/1", + ) + ], + ) + + +async def test_transition_engine_closes_loop_without_unicorn_git_or_claude() -> None: + environment = FakeEnvironment() + journal = FakeJournal() + engine = TransitionEngine( + environment=environment, + observer=FakeObserver(), + planner=FakePlanner(), + executor=FakeExecutor(), + verifier=FakeVerifier(), + journal=journal, + ) + request = TransitionRequest( + environment="warehouse-robot", + objective="move to dock-b", + ) + + outcome = await engine.execute(request) + + assert outcome.accepted is True + assert outcome.before_state is not None + assert outcome.after_state is not None + assert outcome.before_state.state["location"] == "dock-a" + assert outcome.after_state.state["location"] == "dock-b" + assert outcome.observation is not None + assert len(outcome.observation.action_evidence) == 1 + assert len(outcome.observation.verification_evidence) == 1 + assert journal.outcomes == [outcome] + assert environment.cleaned is True + + next_request = TransitionRequest( + environment=request.environment, + objective="return to dock-a", + before_state=outcome.after_state, + ) + assert next_request.before_state is outcome.after_state diff --git a/tests/unit/test_transition_models.py b/tests/unit/test_transition_models.py new file mode 100644 index 0000000..30240ec --- /dev/null +++ b/tests/unit/test_transition_models.py @@ -0,0 +1,52 @@ +"""Tests for provider-neutral UCF transition contracts.""" + +from datetime import UTC, datetime + +from foundry.contracts.transition_models import ( + EvidenceRef, + StateSnapshot, + TransitionObservation, + TransitionOutcome, + TransitionRequest, +) + + +def test_transition_contract_does_not_assume_unicorn_or_git() -> None: + before = StateSnapshot( + environment="warehouse-robot", + observed_at=datetime.now(UTC), + state={"location": "dock-a", "carrying": False}, + evidence=[ + EvidenceRef( + kind="sensor", + uri="sensor://robot/location", + description="Position observation", + ) + ], + ) + request = TransitionRequest( + environment="warehouse-robot", + objective="move to dock-b", + before_state=before, + ) + + after = StateSnapshot( + environment=request.environment, + observed_at=datetime.now(UTC), + state={"location": "dock-b", "carrying": False}, + ) + observation = TransitionObservation( + transition_id=request.id, + observed_state=after, + ) + outcome = TransitionOutcome( + transition_id=request.id, + accepted=True, + before_state=before, + after_state=after, + observation=observation, + ) + + assert outcome.accepted is True + assert outcome.after_state is not None + assert outcome.after_state.state["location"] == "dock-b"