From cf90c4db974367d24272487da389f42c105972d3 Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:07:35 +0200
Subject: [PATCH 01/28] local runtime
---
.../pNNN-local-runtime-without-temporal.md | 257 ++++++++++++++++++
1 file changed, 257 insertions(+)
create mode 100644 docs/roadmap/later/pNNN-local-runtime-without-temporal.md
diff --git a/docs/roadmap/later/pNNN-local-runtime-without-temporal.md b/docs/roadmap/later/pNNN-local-runtime-without-temporal.md
new file mode 100644
index 000000000..371cfbb71
--- /dev/null
+++ b/docs/roadmap/later/pNNN-local-runtime-without-temporal.md
@@ -0,0 +1,257 @@
+# PNNN: Local runtime without Temporal
+
+**Status**
+- Later / exploratory. Written 2026-10-01 as a review for roadmap discussion,
+ not a decision.
+- Effort figures are estimates from reading the code, not from a prototype.
+
+A local Lightspeed with no Temporal, Postgres or Docker looks achievable in
+four to five months with two engineers, including a validation spike. The
+reason is that Temporal only orchestrates our work: every durable fact already
+lives in Postgres and CAS, and the agent loop already runs in-process for
+evals.
+
+Suggested direction to explore:
+
+- Keep Temporal for the hosted runtime.
+- Add a local runtime for interactive sessions: SQLite, a filesystem CAS and an
+ embedded envd, behind the existing public API.
+- Have both runtimes share one session orchestration core rather than forking
+ it.
+- Leave bots, channels and schedules as hosted-only features.
+
+## How much we rely on Temporal
+
+Temporal is our orchestrator, not our database. The session event log,
+checkpoints, CAS and all domain records already live in Postgres, and the bot
+controller treats its Postgres row as authoritative. What only Temporal holds
+is in-flight orchestration: queued admissions, pending emissions and their
+retry backoffs, workflow-start dedupe, the bot inbox and coalescing buffers,
+chat delivery state, and Schedules.
+
+The orchestration surface is broad, though:
+
+- **7 workflow types:** session, sub-agent execution, environment job,
+ transcription, bot controller, bot trigger fire, chat conversation.
+- **About 20k lines** in `temporal-workflow`, written directly against the
+ Temporal SDK (`&mut WorkflowContext` everywhere), with no trait in between.
+ About 40% is pure logic.
+- **About 36 client call sites** in the server: starts, signals, queries,
+ describes and terminates.
+- **About 60 activities** across sessions, bots and channels, with about 32
+ retry policies.
+- **External workers:** the TypeScript chat connectors are Temporal activity
+ workers on their own task queues.
+
+The server crate is about 40k lines of production code. Most of it (gateway,
+environments, MCP, secrets, `SessionTools`) has little or no Temporal coupling.
+
+| Temporal capability | What we use it for | Local equivalent |
+| --- | --- | --- |
+| Durable workflow per session, replay, continue-as-new | `AgentSessionWorkflow` drives `CoreAgentDrive`, races admissions against running activities, runs preparation; rolls over at 10k history events | A tokio task per active session, rebuilt from the log and checkpoint on start (`create_or_load_session` already does this) |
+| Signals | About 13 API mutations funnel into one `submit_admissions` signal; `deliver_emission` carries workflow-to-workflow results | A channel into the session task; a persisted inbox only if an admission must survive a crash before it is committed |
+| Queries + poll loops | The gateway polls `status` every 500 ms in about 15 wait-until-accepted loops; bot, chat and job snapshots | A direct reply from the session task, which is simpler than today |
+| Activities with retry, timeouts, heartbeats | LLM, tools, storage, MCP, environment jobs; heartbeats are how cancellation reaches LLM and tool futures | Plain async calls with a retry helper and a cancellation token. Backoff state is lost on restart, which is acceptable locally. |
+| Durable timers | Await deadlines, promise hard deadlines, cancel watchdog, emission and start retries, bot idle-close | Timers recomputed from state on start; the engine's `await_wake` already derives the next wake from state |
+| Workflow-id dedupe, signal-with-start | Session start, environment jobs, workflow-tool executions, bot controllers, conversations | Unique keys in the store plus an in-process registry of live session tasks |
+| Workflow-tool protocol | Sub-agents, environment-job tools, external plugin workflows (contract is Temporal signals, queries and task queues) | In-process spawn for sub-agents and jobs. External plugins need another transport or stay hosted-only. |
+| Schedules | Bot cron and poll triggers | An in-process scheduler, or hosted-only |
+| Task queues to external workers | Telegram and WhatsApp connectors | Hosted-only |
+
+Several parts were already built without Temporal and would carry over
+unchanged:
+
+- The reapers, the environment reconciler and the CAS sweeper are plain tokio
+ loops.
+- Clients follow progress by long-polling events from Postgres; there is no SSE
+ or WebSocket.
+- The expected-head check on event appends and the promise reaper already
+ assume that signals can be lost.
+
+## What already runs without Temporal
+
+Most of the agent already runs without Temporal: an in-process agent loop
+exists today and is used by `crates/eval` against real providers. The
+architecture work done so far (sans-IO engine, CAS references, store traits) is
+what makes a second runtime plausible.
+
+| Piece | State today | Reusable as-is for local? |
+| --- | --- | --- |
+| `engine` (30k lines) incl. `CoreAgentDrive` | Deterministic, zero Temporal dependency. Emits `AppendEvents`, `GenerateLlm`, `CompactContext`, `InvokeTools`, `Idle`, `Closed`. | Yes |
+| `SessionRunner` in `test-support` (2.5k lines) | Substrate-neutral loop: `drive_until_quiescent` fulfils LLM, compaction and tool actions in-process. Used by `eval` and replay tests. | Yes, after promoting it out of test-support |
+| `llm-runtime` + `llm-clients` (31k lines) | `impl CoreAgentLlm for LlmRuntime`; Anthropic, OpenAI Responses, Completions. | Yes |
+| `tools` (27k lines) | `InlineToolRuntime`, local and scoped filesystem tools, VFS, skills. | Mostly. The local process executor is a one-line placeholder, so there is no in-process shell tool. |
+| `store-fs` (1.8k lines) | Real filesystem CAS (`.lightspeed/cas/sha256/...`), VFS catalog, partial `FsSessionStore` (no listing, metadata, or cross-process lock). Unused in production. | CAS yes; session store needs finishing |
+| In-memory stores | Session, blob, environments, auth, bots, channels, MCP registry. | Tests and ephemeral runs only |
+| `environment-daemon` (envd, 12k lines) | Runs on a laptop today (`./dev.sh runtime` starts one on `127.0.0.1:19091`). | Yes, embedded or as a sidecar |
+| `bots`, `channels` domain crates | Pure state and policy; `controller/state.rs` (2.5k lines) has zero Temporal references. | Logic yes; the workflow shells around it no |
+| `cli` (24k lines) | Pure HTTP client of the hosted gateway (`HttpAgentApi`). | The TUI yes; needs a local backend behind it |
+| `platform/web` | Talks to the runtime only through the generated API client; the browser demo already swaps in an in-browser backend. | Yes, if a local runtime serves the same API |
+
+Two dependencies are not Temporal but still block a local install: PostgreSQL
+(`store-pg`, 18k lines, 10 migrations) and S3-compatible object storage
+(optional; small blobs are already inlined in Postgres). There is no SQLite
+anywhere in the tree. The [first-class runtime CLI](../p183-first-class-runtime-cli.md)
+work explicitly put "no embedded database or runtime alternative" out of scope.
+
+## Four ways to get there
+
+The options differ mainly in where the session orchestration lives: today it is
+about 20k lines of Rust written directly against the Temporal SDK, with no
+abstraction between it and the SDK.
+
+| Option | What it means | Hosted runtime | Local install | Rough effort | Main risk |
+| --- | --- | --- | --- | --- | --- |
+| **A. Bundle the current stack** | A launcher starts the Temporal dev server (SQLite-backed), an embedded or managed Postgres, envd and `lightspeed-server` as subprocesses behind one command. | Unchanged | One command, but Go + Postgres binaries, several processes, and slow startup | 2–4 weeks | Feels like a server install, not a CLI tool, so it may not fix adoption |
+| **B. Two runtimes** | Keep the Temporal runtime. Build a separate local runtime around `SessionRunner`, SQLite and the filesystem CAS that serves the same public API. | Unchanged | Single binary, no services | 2–3 months for interactive sessions | Orchestration semantics (admission, steering, cancel, promises, sub-agents) are re-implemented and drift from the hosted behaviour |
+| **C. One orchestration core, two substrates** | Do for the session workflow what the engine did for the agent loop: move admission racing, preparation, promise polling, emission delivery and the watchdog into a sans-IO orchestrator. Temporal and a local tokio/SQLite substrate each interpret it. | Temporal stays, behind a thinner shell | Single binary, no services | 3–5 months, mostly refactoring the hosted path first | A large refactor of working, live-validated code; Temporal's determinism rules (e.g. no custom wakers) constrain the shared design |
+| **D. Drop Temporal everywhere** | Option C, plus a Postgres-backed durable substrate (inbox, outbox, timers, leases) replaces Temporal in hosted too. | Postgres-only; we own scheduling, leases and failover | Same binary with SQLite | 6+ months | We take on the distributed-systems work Temporal does today: worker leases, failover, timer sweeps, at-least-once delivery, schedules |
+
+Option B is the fastest route to a real local product, but every orchestration
+feature then has to be built twice. Option C costs more up front and leaves a
+single definition of session behaviour; it is also the only path that keeps
+Option D open later without committing to it now. Option A is worth a short
+spike only to test whether "one command" alone moves adoption.
+
+## Where a local runtime plugs in
+
+```mermaid
+flowchart TD
+ subgraph Clients
+ CLI[lightspeed CLI TUI]
+ Web[Web UI]
+ SDK[API clients and SDKs]
+ end
+ subgraph Shared["Shared, runtime-neutral"]
+ API["Public API AgentApiService + JSON-RPC"]
+ Orch["Session orchestrator (new) sans-IO: admissions, awaits, promises, sub-agents"]
+ Engine["Engine and adapters CoreAgentDrive, llm-runtime, tools, MCP"]
+ end
+ subgraph Hosted["Hosted substrate (today)"]
+ Temporal["Temporal durable workflows"]
+ PG[("PostgreSQL store-pg")]
+ S3[("S3 CAS object storage")]
+ RemoteEnvd["Remote envd via environment gateway"]
+ HostedOnly["Hosted only: bots, channels, schedules"]
+ end
+ subgraph Local["Local substrate (new)"]
+ Tasks["Session tasks (new) tokio, per session"]
+ SQLite[("SQLite (new) store-sqlite")]
+ FsCas[("Filesystem CAS store-fs, exists")]
+ EmbeddedEnvd["Embedded envd (new) local shell, files"]
+ OneBinary["One binary, data in ~/.lightspeed"]
+ end
+ Clients --> Shared
+ Shared --> Hosted
+ Shared --> Local
+```
+
+The local runtime keeps the public API, the engine and the adapters as they
+are. The new work is the extracted orchestrator and the substrate under it. The
+CLI and web UI need no changes to talk to either runtime.
+
+## What a local version would take
+
+A useful local v1 is about eight work items, assuming its scope is interactive
+sessions only. Bots, chat channels, schedules, Platform login and
+multi-universe tenancy stay hosted features. Those are always-on, multi-user
+concerns, and they account for most of the Temporal surface we would otherwise
+have to replace (bot controller, trigger fires, conversations, Schedules,
+connector task queues).
+
+In scope for local v1: sessions and runs; steer, cancel and approvals; local
+files and shell; MCP; skills and profiles; sub-agents; resuming a session days
+later. The public API stays identical, so the CLI TUI and the web UI work
+unchanged.
+
+The effort figures are engineer-weeks for someone who knows the codebase, at
+review-level confidence. The "Effort" column assumes Option C; the last column
+notes where Option B differs.
+
+| # | Work item | What exists | What is new | Effort | Option B instead |
+| --- | --- | --- | --- | --- | --- |
+| 1 | Local backend behind the public API | `AgentApiService` trait (about 119 methods, many with "unavailable" defaults) and a generic `dispatch_json_rpc` in `crates/api` | `LocalAgentApi` implementing the session, run, context, events, VFS and models subset. The CLI calls it in-process; `lightspeed serve` exposes it to the web UI. | 3–4 | Same |
+| 2 | Session orchestrator | `CoreAgentDrive`; `SessionRunner` (synchronous drive-until-quiescent); the Temporal session workflow | Admissions handled while a model or tool call runs (steer, cancel, approvals), awaits and timers, promises, queued runs, sub-agents spawned in-process | 8–12, which includes reshaping the hosted workflow | 5–7 on its own, then every later feature built twice |
+| 3 | SQLite store | Store traits in `engine`, `vfs`, `environments`, `auth`, `mcp`, `profiles`; `store-pg` as the reference | A `store-sqlite` crate with one-file migrations. jsonb containment becomes `json_each` or filtering in the app; `text[]` becomes JSON; advisory locks become a single-writer process lock. | 3–5 | Same |
+| 4 | Filesystem CAS + collection | `FsBlobStore` (sha256 layout) | Wire it in; reference roots and sweeps without Postgres | 1 | Same |
+| 5 | Local shell and process tools | envd (12k lines) runs on laptops today; the in-process `ProcessExecutor` is a placeholder | Embed envd as a library over an in-memory transport, or spawn it as a sidecar. Default the active environment to the working directory. | 2–3 | Same |
+| 6 | Permission model for a user's own machine | MCP approvals (`AwaitingApproval`, parked runs) | Approvals for shell and writes outside the workspace, with allow rules per session and per directory. Codex and Claude Code users expect this. | 2–3 | Same |
+| 7 | Identity and configuration | Single-user auth mode; model defaults; `connect` profiles in the CLI | An implicit local universe and actor; provider keys from environment variables or the OS keychain; a data directory such as `~/.lightspeed` | 1–2 | Same |
+| 8 | Packaging and tests | Release pipeline for envd (musl builds); replay vectors | One `lightspeed` binary (TUI by default, plus `serve`); the substrate-neutral test suite run against both substrates | 2–3 | Tests per runtime |
+
+Total: about 22–33 engineer-weeks for Option C, or 19–28 for Option B before
+the duplication cost. With two people in parallel, that is roughly three to
+four months of calendar time, plus the spike in the sequence below.
+
+Two items are mostly mechanical. The gateway's Temporal calls are concentrated
+in `workflow.rs` and `session_lifecycle.rs` (start, `submit_admissions`,
+`status` polling, describe, terminate). Putting them behind a small
+`SessionControl` trait would let the hosted gateway and the local backend share
+most of the 19k-line service layer. The activity helpers return Temporal error
+types (`ActivityError`, `ApplicationFailure`), so they need a neutral error
+type before the local substrate can call them.
+
+## Risks and open questions
+
+The biggest risk is not the database swap. It is that two runtimes slowly
+disagree about what a session does.
+
+**Risks**
+
+- **Semantic drift.** Steering, cancellation, promise deadlines and sub-agent
+ budgets have been hardened against live Temporal suites. A second runtime
+ needs the same conformance suite, written against the substrate-neutral API
+ and run against both substrates in CI.
+- **Crash and sleep semantics.** Today Temporal retries an activity interrupted
+ by a worker restart. Locally, a closed laptop lid or a killed process leaves
+ an LLM call or shell command half-done. We need an explicit rule, probably:
+ retry model calls, and mark interrupted shell commands as interrupted rather
+ than re-running them. This overlaps the parked idempotency work for tools.
+- **Two writers on one data directory.** Two CLI windows on the same session
+ need either a file lock per session or one local daemon that owns the store.
+ Codex and Claude Code avoid this by being one process per conversation.
+- **Every schema change twice.** Each `store-pg` migration needs a SQLite twin.
+ The release metadata currently pins one schema revision.
+- **Shared orchestrator under Temporal's rules.** For Option C, the shared code
+ must stay deterministic and avoid custom wakers (the TMPRL1100 constraint).
+ That suggests a sans-IO state machine, as `bots::controller::state` already
+ is, rather than shared async code.
+- **Plugin contract.** The workflow-tool contract is defined in Temporal terms
+ (signals, queries, task queues). External plugin workflows would not run
+ locally unless the contract gets a second, non-Temporal binding.
+
+**Open questions**
+
+- [ ] Is the adoption blocker really infrastructure, or also the first-run
+ experience (keys, environments, profiles)? Option A answers this cheaply.
+- [ ] Should a local session be movable to hosted, e.g. `lightspeed push`?
+ Sessions are event logs plus CAS, so export and import look plausible, and
+ that would make local an on-ramp to hosted rather than a fork.
+- [ ] What is Lightspeed's differentiator against Codex, Claude Code and Pi on
+ a laptop? Candidates: provider-native multi-model sessions, durable
+ resumable sessions, VFS workspaces, sub-agents, and the same agent later
+ running as a hosted bot.
+- [ ] Should local sandbox shell commands (macOS seatbelt, Linux namespaces),
+ or rely on approvals alone at first?
+- [ ] Would hosted ever drop Temporal (Option D)? If not, Option C's main
+ payoff is a single definition of behaviour, not portability.
+
+## Suggested sequence
+
+A short spike decides whether the refactor starts. Durations are calendar weeks
+for two engineers; each gate sits between phases, and the first one is the real
+decision.
+
+| Phase | Duration | Work | Gate after |
+| --- | --- | --- | --- |
+| 0 · Spike | 2–4 weeks | CLI over `SessionRunner`; in-memory or SQLite store; embedded envd; try with design partners | **Go / no-go:** spike used on real tasks; pick Option B or C |
+| 1 · Extract the core | 5–7 weeks | Sans-IO orchestrator; thin Temporal shell; `SessionControl` trait; neutral activity errors | **Hosted unchanged:** live Temporal suites green on the thin shell |
+| 2 · Local substrate | 5–7 weeks | SQLite store; local session tasks; shell approvals; one-binary packaging | **Parity:** conformance suite green on both substrates |
+| 3 · Beta and bridge | 2–3 weeks | Public local release; push session to hosted; docs and onboarding | Then revisit Option D |
+
+Start with a throwaway-tolerant spike. Wire the existing CLI to an in-process
+`SessionRunner` with an embedded envd, then put it in front of a few design
+partners. That tests the adoption hypothesis for 2–4 weeks of work, before we
+commit to refactoring hosted orchestration. If the spike lands, extract the
+orchestrator while hosted is the only consumer, so live suites prove nothing
+changed. Only then build the local substrate on top of it.
From 3bbf5e9dce88dba4aa9716ddef606bbcc5c162b3 Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:21:11 +0200
Subject: [PATCH 02/28] local runtime
---
.../pNNN-local-runtime-without-temporal.md | 166 ++++++++++++++----
1 file changed, 136 insertions(+), 30 deletions(-)
diff --git a/docs/roadmap/later/pNNN-local-runtime-without-temporal.md b/docs/roadmap/later/pNNN-local-runtime-without-temporal.md
index 371cfbb71..1b553cf7e 100644
--- a/docs/roadmap/later/pNNN-local-runtime-without-temporal.md
+++ b/docs/roadmap/later/pNNN-local-runtime-without-temporal.md
@@ -4,20 +4,26 @@
- Later / exploratory. Written 2026-10-01 as a review for roadmap discussion,
not a decision.
- Effort figures are estimates from reading the code, not from a prototype.
+- Direction preference: if we pursue a local runtime, it is Option C (one
+ orchestration core, two substrates). A separate, forked local runtime
+ (Option B) is not on the table.
A local Lightspeed with no Temporal, Postgres or Docker looks achievable in
-four to five months with two engineers, including a validation spike. The
+three to four months with two engineers, with a validation spike running in
+parallel to the first phase. The
reason is that Temporal only orchestrates our work: every durable fact already
lives in Postgres and CAS, and the agent loop already runs in-process for
evals.
-Suggested direction to explore:
+Direction:
- Keep Temporal for the hosted runtime.
-- Add a local runtime for interactive sessions: SQLite, a filesystem CAS and an
- embedded envd, behind the existing public API.
-- Have both runtimes share one session orchestration core rather than forking
- it.
+- Lift session orchestration out of the Temporal workflow into a sans-IO
+ `sessions` crate. This is worth doing on its own, before and without a local
+ runtime.
+- Add a local runtime for interactive sessions on top of the same `sessions`
+ crate: SQLite, a filesystem CAS and an embedded envd, behind the existing
+ public API.
- Leave bots, channels and schedules as hosted-only features.
## How much we rely on Temporal
@@ -107,11 +113,36 @@ abstraction between it and the SDK.
| **C. One orchestration core, two substrates** | Do for the session workflow what the engine did for the agent loop: move admission racing, preparation, promise polling, emission delivery and the watchdog into a sans-IO orchestrator. Temporal and a local tokio/SQLite substrate each interpret it. | Temporal stays, behind a thinner shell | Single binary, no services | 3–5 months, mostly refactoring the hosted path first | A large refactor of working, live-validated code; Temporal's determinism rules (e.g. no custom wakers) constrain the shared design |
| **D. Drop Temporal everywhere** | Option C, plus a Postgres-backed durable substrate (inbox, outbox, timers, leases) replaces Temporal in hosted too. | Postgres-only; we own scheduling, leases and failover | Same binary with SQLite | 6+ months | We take on the distributed-systems work Temporal does today: worker leases, failover, timer sweeps, at-least-once delivery, schedules |
-Option B is the fastest route to a real local product, but every orchestration
-feature then has to be built twice. Option C costs more up front and leaves a
-single definition of session behaviour; it is also the only path that keeps
-Option D open later without committing to it now. Option A is worth a short
-spike only to test whether "one command" alone moves adoption.
+Option C is the preferred direction. Option B is the fastest route to a real
+local product, but every orchestration feature then has to be built twice and
+the two runtimes drift. Option C costs more up front and leaves a single
+definition of session behaviour; it is also the only path that keeps Option D
+open later without committing to it now. Option A is worth a short spike only
+to test whether "one command" alone moves adoption.
+
+### Option C pays off without a local runtime
+
+Lifting orchestration out of the Temporal workflow improves the hosted runtime
+even if no local runtime follows:
+
+- **Testability.** Admission racing, preparation, promise polling, emission
+ retry, the cancel watchdog and continue-as-new gating are today testable only
+ through Temporal, and some failures (such as the custom-waker restriction,
+ TMPRL1100) surface only in live suites. As a plain state machine they get
+ fast unit tests and replay vectors, as the engine already has.
+- **Smaller determinism surface.** Only the thin interpreter has to obey
+ Temporal's workflow rules, not about 20k lines of orchestration.
+- **Less SDK exposure.** The Temporal Rust SDK is at 0.4.0. A thinner shell
+ limits how much code each SDK upgrade touches.
+- **One copy of session behaviour.** `test-support`'s `SessionRunner` already
+ duplicates hosted behaviour by hand (its prompt-refresh fallback "mirrors the
+ hosted product"). With a shared `sessions` crate, eval and tests run the
+ production orchestration.
+- **Proven pattern.** `bots::controller::state` is already a pure state machine
+ inside a Temporal shell; sessions would follow the same pattern.
+
+The cost is a refactor of live-validated code. It pays back because session
+orchestration keeps changing: most recent roadmap items touched it.
## Where a local runtime plugs in
@@ -124,8 +155,8 @@ flowchart TD
end
subgraph Shared["Shared, runtime-neutral"]
API["Public API AgentApiService + JSON-RPC"]
- Orch["Session orchestrator (new) sans-IO: admissions, awaits, promises, sub-agents"]
- Engine["Engine and adapters CoreAgentDrive, llm-runtime, tools, MCP"]
+ Orch["sessions (new) sans-IO orchestration: admissions, awaits, promises, sub-agents"]
+ Engine["harness and adapters CoreAgentDrive, llm-runtime, tools, MCP"]
end
subgraph Hosted["Hosted substrate (today)"]
Temporal["Temporal durable workflows"]
@@ -150,6 +181,79 @@ The local runtime keeps the public API, the engine and the adapters as they
are. The new work is the extracted orchestrator and the substrate under it. The
CLI and web UI need no changes to talk to either runtime.
+### Crate layout
+
+The orchestration gets its own crate rather than living in the engine:
+
+| Crate | Role |
+| --- | --- |
+| `harness` (renamed from `engine`) | Lightspeed's native agent loop: events, session state, context, tool planning, `CoreAgentDrive`. Deterministic and event-sourced. |
+| `sessions` (new) | Sans-IO session orchestration: admission inbox, run slot, preparation steps, promise sources, emission outbox, workflow-start dedupe, wake computation, cancel watchdog. Depends on `harness`. |
+| `temporal-workflow` | Thin interpreters that run `sessions`, `bots` and `channels` state machines on Temporal. |
+| `temporal-runtime` (renamed from `temporal-server`) | Activities, roles and Temporal wiring. |
+| `local-runtime` (later) | Tokio interpreter of `sessions` over SQLite, filesystem CAS and embedded envd. |
+
+Why `sessions` is separate from `harness`:
+
+- **Different state models.** Harness state is reduced from the event log;
+ replaying the log reconstructs it. Orchestration state is in-flight
+ bookkeeping (pending admissions, undelivered emissions, start dedupe, timers)
+ that is carried across continue-as-new and partly derived from harness state.
+ Mixing them blurs the "replay the log, get the state" invariant.
+- **Vocabulary.** [External harness sessions](../p185-external-harness-sessions.md)
+ uses "harness" for what owns model calls, context, tools and the inner loop,
+ and gives Lightspeed admission, orchestration, access policy and supervision.
+ That maps onto `harness` and `sessions` respectively. Keeping orchestration
+ above the harness also leaves room to drive an external harness through the
+ same `sessions` machinery later.
+- **Naming convention.** `bots`, `channels` and `environments` already hold
+ their domain's pure state machines and policy; `sessions` matches.
+
+A later split could also move the gateway's service layer (about 23k lines,
+little Temporal coupling) out of `temporal-runtime` into its own crate once a
+`SessionControl` trait exists, so both runtimes serve the API from the same
+code. That is independent of the renames.
+
+### A sync core with async interpreters
+
+`sessions` should be a synchronous state machine: inputs such as "admission
+arrived", "activity completed" or "timer fired"; outputs such as start or
+cancel an activity, set a timer, signal or start a workflow, roll over. Each
+runtime provides a small async interpreter that owns the racing and the
+plumbing. Shared async code generic over a host trait (`start_activity`,
+`timer`, `next_admission`, `select`) is a real alternative, but the sync core
+is preferred because:
+
+- **Temporal's rules stay out of shared code.** Under Temporal, await order,
+ `select` and combinators must be deterministic on its executor. Shared async
+ code would have to obey that even on tokio, where nothing enforces it, and
+ violations surface only under Temporal. A machine without futures cannot
+ violate them.
+- **Racing becomes explicit input.** The hard behaviour is what happens when
+ an admission, cancel or approval arrives during a model or tool call. In
+ async code that is a `select` whose semantics differ between executors
+ (branch order; dropping a future versus Temporal's explicit cancellation and
+ waiting for its result). As ordered inputs, interleavings are testable,
+ including the awkward ones.
+- **State is already a value.** Continue-as-new carry, and a local restart,
+ need the orchestration state as a serializable struct. Async code keeps it in
+ future stack frames and needs hand-extracted carry state, which is what
+ `AgentSessionContinuationState` does today.
+- **Cheaper tests.** Feed input sequences, assert emitted commands; no fake
+ executor or timing.
+
+Costs: state machines invert control, so linear multi-step flows (such as the
+preparation retry loop) read worse than top-to-bottom async code. Versioning
+is not avoided either: changing what the machine decides still changes the
+commands in Temporal history. The bot controller's split is the working
+precedent: decisions in a sync core, racing in a small async shell. Linear
+steps that never race can stay async in the interpreter rather than being
+forced into states.
+
+To settle it with evidence rather than preference, the first step of the
+extraction ports one slice both ways (the wait loop plus admission racing
+against a running activity) and compares the code and its tests.
+
## What a local version would take
A useful local v1 is about eight work items, assuming its scope is interactive
@@ -181,7 +285,8 @@ notes where Option B differs.
Total: about 22–33 engineer-weeks for Option C, or 19–28 for Option B before
the duplication cost. With two people in parallel, that is roughly three to
-four months of calendar time, plus the spike in the sequence below.
+four months of calendar time; the spike in the sequence below runs alongside
+the first phase.
Two items are mostly mechanical. The gateway's Temporal calls are concentrated
in `workflow.rs` and `session_lifecycle.rs` (start, `submit_admissions`,
@@ -212,10 +317,10 @@ disagree about what a session does.
Codex and Claude Code avoid this by being one process per conversation.
- **Every schema change twice.** Each `store-pg` migration needs a SQLite twin.
The release metadata currently pins one schema revision.
-- **Shared orchestrator under Temporal's rules.** For Option C, the shared code
- must stay deterministic and avoid custom wakers (the TMPRL1100 constraint).
- That suggests a sans-IO state machine, as `bots::controller::state` already
- is, rather than shared async code.
+- **Shared orchestrator under Temporal's rules.** The shared code must stay
+ deterministic and avoid custom wakers (the TMPRL1100 constraint). The sync
+ core described under [A sync core with async interpreters](#a-sync-core-with-async-interpreters)
+ is the mitigation; the risk is that awkward flows get forced into states.
- **Plugin contract.** The workflow-tool contract is defined in Temporal terms
(signals, queries, task queues). External plugin workflows would not run
locally unless the contract gets a second, non-Temporal binding.
@@ -238,20 +343,21 @@ disagree about what a session does.
## Suggested sequence
-A short spike decides whether the refactor starts. Durations are calendar weeks
-for two engineers; each gate sits between phases, and the first one is the real
-decision.
+Because the extraction is worth doing on its own, it does not have to wait for
+the spike; the spike gates only the local substrate. Durations are calendar
+weeks for two engineers.
| Phase | Duration | Work | Gate after |
| --- | --- | --- | --- |
-| 0 · Spike | 2–4 weeks | CLI over `SessionRunner`; in-memory or SQLite store; embedded envd; try with design partners | **Go / no-go:** spike used on real tasks; pick Option B or C |
-| 1 · Extract the core | 5–7 weeks | Sans-IO orchestrator; thin Temporal shell; `SessionControl` trait; neutral activity errors | **Hosted unchanged:** live Temporal suites green on the thin shell |
-| 2 · Local substrate | 5–7 weeks | SQLite store; local session tasks; shell approvals; one-binary packaging | **Parity:** conformance suite green on both substrates |
+| 0 · Spike | 2–4 weeks, in parallel with phase 1 | CLI over `SessionRunner`; in-memory or SQLite store; embedded envd; try with design partners | **Go / no-go on local:** spike used on real tasks |
+| 1 · Extract `sessions` | 5–7 weeks | Port one slice both ways and pick sync core or async host trait; sans-IO orchestrator; thin Temporal interpreter; `SessionControl` trait; neutral activity errors; `engine` → `harness` and `temporal-server` → `temporal-runtime` renames | **Hosted unchanged:** live Temporal suites green on the thin interpreter |
+| 2 · Local substrate | 5–7 weeks | SQLite store; tokio interpreter of `sessions`; shell approvals; one-binary packaging | **Parity:** conformance suite green on both substrates |
| 3 · Beta and bridge | 2–3 weeks | Public local release; push session to hosted; docs and onboarding | Then revisit Option D |
-Start with a throwaway-tolerant spike. Wire the existing CLI to an in-process
-`SessionRunner` with an embedded envd, then put it in front of a few design
-partners. That tests the adoption hypothesis for 2–4 weeks of work, before we
-commit to refactoring hosted orchestration. If the spike lands, extract the
-orchestrator while hosted is the only consumer, so live suites prove nothing
-changed. Only then build the local substrate on top of it.
+The spike is throwaway-tolerant: wire the existing CLI to an in-process
+`SessionRunner` with an embedded envd and put it in front of a few design
+partners to test the adoption hypothesis. Meanwhile, extract `sessions` while
+hosted is its only consumer, so live suites prove nothing changed. If the spike
+does not land, phase 1 still stands on its own and phases 2 and 3 wait. If it
+does, the local substrate is built on the extracted core, and the spike's
+`SessionRunner` is replaced by the production orchestration.
From 8326ae481182fe41e07b324e6505840d198fa41c Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:31:47 +0200
Subject: [PATCH 03/28] platform universe icons
---
.../db/migrations/0003_brief_purple_man.sql | 2 +
.../db/migrations/meta/0003_snapshot.json | 1105 +++++++++++++++++
platform/db/migrations/meta/_journal.json | 7 +
platform/db/scripts/check-migrations.ts | 2 +
platform/db/src/schema/platform.ts | 2 +
.../src/routes/universe-appearance.test.ts | 92 ++
platform/shared/src/index.ts | 4 +
platform/shared/src/universe-appearance.ts | 16 +
platform/web/src/api.ts | 4 +-
platform/web/src/components/bot/face.tsx | 3 +-
.../universe-appearance-card.test.tsx | 114 ++
.../components/universe-appearance-card.tsx | 75 ++
platform/web/src/components/universe-icon.tsx | 34 +
.../web/src/components/universe-switcher.tsx | 8 +-
.../src/demo/fixtures/personal-assistant.ts | 2 +
.../web/src/demo/fixtures/software-factory.ts | 2 +
.../src/demo/fixtures/technical-support.ts | 2 +
platform/web/src/demo/router.test.ts | 29 +
platform/web/src/demo/routes/platform.ts | 6 +-
platform/web/src/demo/store.ts | 6 +-
platform/web/src/lib/identity-colors.ts | 19 +
.../web/src/pages/GeneralSettingsPage.tsx | 4 +-
release/metadata.env | 2 +-
23 files changed, 1530 insertions(+), 10 deletions(-)
create mode 100644 platform/db/migrations/0003_brief_purple_man.sql
create mode 100644 platform/db/migrations/meta/0003_snapshot.json
create mode 100644 platform/server/src/routes/universe-appearance.test.ts
create mode 100644 platform/shared/src/universe-appearance.ts
create mode 100644 platform/web/src/components/universe-appearance-card.test.tsx
create mode 100644 platform/web/src/components/universe-appearance-card.tsx
create mode 100644 platform/web/src/components/universe-icon.tsx
create mode 100644 platform/web/src/lib/identity-colors.ts
diff --git a/platform/db/migrations/0003_brief_purple_man.sql b/platform/db/migrations/0003_brief_purple_man.sql
new file mode 100644
index 000000000..e66d1cf0a
--- /dev/null
+++ b/platform/db/migrations/0003_brief_purple_man.sql
@@ -0,0 +1,2 @@
+ALTER TABLE "universes" ADD COLUMN "icon" text DEFAULT 'orbit' NOT NULL;--> statement-breakpoint
+ALTER TABLE "universes" ADD COLUMN "icon_color" text DEFAULT 'default' NOT NULL;
\ No newline at end of file
diff --git a/platform/db/migrations/meta/0003_snapshot.json b/platform/db/migrations/meta/0003_snapshot.json
new file mode 100644
index 000000000..f26eb29c3
--- /dev/null
+++ b/platform/db/migrations/meta/0003_snapshot.json
@@ -0,0 +1,1105 @@
+{
+ "id": "9435a284-56db-4883-ac83-db8f678c6bec",
+ "prevId": "79232b64-c0bb-4714-a45d-2ee458a3aeb1",
+ "version": "7",
+ "dialect": "postgresql",
+ "tables": {
+ "public.account": {
+ "name": "account",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "account_id": {
+ "name": "account_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "provider_id": {
+ "name": "provider_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "user_id": {
+ "name": "user_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "access_token": {
+ "name": "access_token",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "refresh_token": {
+ "name": "refresh_token",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "id_token": {
+ "name": "id_token",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "access_token_expires_at": {
+ "name": "access_token_expires_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "refresh_token_expires_at": {
+ "name": "refresh_token_expires_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "scope": {
+ "name": "scope",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "password": {
+ "name": "password",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ }
+ },
+ "indexes": {
+ "account_userId_idx": {
+ "name": "account_userId_idx",
+ "columns": [
+ {
+ "expression": "user_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {
+ "account_user_id_user_id_fk": {
+ "name": "account_user_id_user_id_fk",
+ "tableFrom": "account",
+ "tableTo": "user",
+ "columnsFrom": [
+ "user_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.invitation": {
+ "name": "invitation",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "organization_id": {
+ "name": "organization_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "email": {
+ "name": "email",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "role": {
+ "name": "role",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "status": {
+ "name": "status",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'pending'"
+ },
+ "expires_at": {
+ "name": "expires_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "inviter_id": {
+ "name": "inviter_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ }
+ },
+ "indexes": {
+ "invitation_organizationId_idx": {
+ "name": "invitation_organizationId_idx",
+ "columns": [
+ {
+ "expression": "organization_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ },
+ "invitation_email_idx": {
+ "name": "invitation_email_idx",
+ "columns": [
+ {
+ "expression": "email",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {
+ "invitation_organization_id_organization_id_fk": {
+ "name": "invitation_organization_id_organization_id_fk",
+ "tableFrom": "invitation",
+ "tableTo": "organization",
+ "columnsFrom": [
+ "organization_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ },
+ "invitation_inviter_id_user_id_fk": {
+ "name": "invitation_inviter_id_user_id_fk",
+ "tableFrom": "invitation",
+ "tableTo": "user",
+ "columnsFrom": [
+ "inviter_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.member": {
+ "name": "member",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "organization_id": {
+ "name": "organization_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "user_id": {
+ "name": "user_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "role": {
+ "name": "role",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'member'"
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ }
+ },
+ "indexes": {
+ "member_organizationId_idx": {
+ "name": "member_organizationId_idx",
+ "columns": [
+ {
+ "expression": "organization_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ },
+ "member_userId_idx": {
+ "name": "member_userId_idx",
+ "columns": [
+ {
+ "expression": "user_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {
+ "member_organization_id_organization_id_fk": {
+ "name": "member_organization_id_organization_id_fk",
+ "tableFrom": "member",
+ "tableTo": "organization",
+ "columnsFrom": [
+ "organization_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ },
+ "member_user_id_user_id_fk": {
+ "name": "member_user_id_user_id_fk",
+ "tableFrom": "member",
+ "tableTo": "user",
+ "columnsFrom": [
+ "user_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.organization": {
+ "name": "organization",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "name": {
+ "name": "name",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "slug": {
+ "name": "slug",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "logo": {
+ "name": "logo",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "metadata": {
+ "name": "metadata",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ }
+ },
+ "indexes": {
+ "organization_slug_uidx": {
+ "name": "organization_slug_uidx",
+ "columns": [
+ {
+ "expression": "slug",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": true,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {},
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {
+ "organization_slug_unique": {
+ "name": "organization_slug_unique",
+ "nullsNotDistinct": false,
+ "columns": [
+ "slug"
+ ]
+ }
+ },
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.session": {
+ "name": "session",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "expires_at": {
+ "name": "expires_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "token": {
+ "name": "token",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "ip_address": {
+ "name": "ip_address",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "user_agent": {
+ "name": "user_agent",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "user_id": {
+ "name": "user_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "active_organization_id": {
+ "name": "active_organization_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "impersonated_by": {
+ "name": "impersonated_by",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "access_version": {
+ "name": "access_version",
+ "type": "integer",
+ "primaryKey": false,
+ "notNull": true,
+ "default": 0
+ }
+ },
+ "indexes": {
+ "session_userId_idx": {
+ "name": "session_userId_idx",
+ "columns": [
+ {
+ "expression": "user_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {
+ "session_user_id_user_id_fk": {
+ "name": "session_user_id_user_id_fk",
+ "tableFrom": "session",
+ "tableTo": "user",
+ "columnsFrom": [
+ "user_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {
+ "session_token_unique": {
+ "name": "session_token_unique",
+ "nullsNotDistinct": false,
+ "columns": [
+ "token"
+ ]
+ }
+ },
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.user": {
+ "name": "user",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "name": {
+ "name": "name",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "email": {
+ "name": "email",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "email_verified": {
+ "name": "email_verified",
+ "type": "boolean",
+ "primaryKey": false,
+ "notNull": true,
+ "default": false
+ },
+ "image": {
+ "name": "image",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "role": {
+ "name": "role",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "banned": {
+ "name": "banned",
+ "type": "boolean",
+ "primaryKey": false,
+ "notNull": false,
+ "default": false
+ },
+ "ban_reason": {
+ "name": "ban_reason",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "ban_expires": {
+ "name": "ban_expires",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "identity_source": {
+ "name": "identity_source",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'local'"
+ },
+ "oidc_issuer": {
+ "name": "oidc_issuer",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "oidc_subject": {
+ "name": "oidc_subject",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "company_admitted": {
+ "name": "company_admitted",
+ "type": "boolean",
+ "primaryKey": false,
+ "notNull": true,
+ "default": false
+ },
+ "provider_checked_at": {
+ "name": "provider_checked_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "emergency_admin": {
+ "name": "emergency_admin",
+ "type": "boolean",
+ "primaryKey": false,
+ "notNull": true,
+ "default": false
+ },
+ "access_version": {
+ "name": "access_version",
+ "type": "integer",
+ "primaryKey": false,
+ "notNull": true,
+ "default": 0
+ }
+ },
+ "indexes": {
+ "user_oidc_identity_idx": {
+ "name": "user_oidc_identity_idx",
+ "columns": [
+ {
+ "expression": "oidc_issuer",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ },
+ {
+ "expression": "oidc_subject",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": true,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {},
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {
+ "user_email_unique": {
+ "name": "user_email_unique",
+ "nullsNotDistinct": false,
+ "columns": [
+ "email"
+ ]
+ }
+ },
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.verification": {
+ "name": "verification",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "text",
+ "primaryKey": true,
+ "notNull": true
+ },
+ "identifier": {
+ "name": "identifier",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "value": {
+ "name": "value",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "expires_at": {
+ "name": "expires_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ }
+ },
+ "indexes": {
+ "verification_identifier_idx": {
+ "name": "verification_identifier_idx",
+ "columns": [
+ {
+ "expression": "identifier",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {},
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.identity_audit": {
+ "name": "identity_audit",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "uuid",
+ "primaryKey": true,
+ "notNull": true,
+ "default": "gen_random_uuid()"
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "actor_id": {
+ "name": "actor_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "action": {
+ "name": "action",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "target_id": {
+ "name": "target_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "universe_id": {
+ "name": "universe_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "outcome": {
+ "name": "outcome",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "details": {
+ "name": "details",
+ "type": "jsonb",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'{}'::jsonb"
+ }
+ },
+ "indexes": {
+ "identity_audit_created_idx": {
+ "name": "identity_audit_created_idx",
+ "columns": [
+ {
+ "expression": "created_at",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": false,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {},
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.universe_setup_installations": {
+ "name": "universe_setup_installations",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "uuid",
+ "primaryKey": true,
+ "notNull": true,
+ "default": "gen_random_uuid()"
+ },
+ "universe_id": {
+ "name": "universe_id",
+ "type": "uuid",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "setup_id": {
+ "name": "setup_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "installed_version": {
+ "name": "installed_version",
+ "type": "integer",
+ "primaryKey": false,
+ "notNull": true,
+ "default": 0
+ },
+ "status": {
+ "name": "status",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'installing'"
+ },
+ "state": {
+ "name": "state",
+ "type": "jsonb",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'{}'::jsonb"
+ },
+ "error": {
+ "name": "error",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "installed_by_user_id": {
+ "name": "installed_by_user_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ }
+ },
+ "indexes": {
+ "universe_setup_installations_universe_setup_idx": {
+ "name": "universe_setup_installations_universe_setup_idx",
+ "columns": [
+ {
+ "expression": "universe_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ },
+ {
+ "expression": "setup_id",
+ "isExpression": false,
+ "asc": true,
+ "nulls": "last"
+ }
+ ],
+ "isUnique": true,
+ "concurrently": false,
+ "method": "btree",
+ "with": {}
+ }
+ },
+ "foreignKeys": {
+ "universe_setup_installations_universe_id_universes_id_fk": {
+ "name": "universe_setup_installations_universe_id_universes_id_fk",
+ "tableFrom": "universe_setup_installations",
+ "tableTo": "universes",
+ "columnsFrom": [
+ "universe_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ },
+ "universe_setup_installations_installed_by_user_id_user_id_fk": {
+ "name": "universe_setup_installations_installed_by_user_id_user_id_fk",
+ "tableFrom": "universe_setup_installations",
+ "tableTo": "user",
+ "columnsFrom": [
+ "installed_by_user_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "set null",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {},
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ },
+ "public.universes": {
+ "name": "universes",
+ "schema": "",
+ "columns": {
+ "id": {
+ "name": "id",
+ "type": "uuid",
+ "primaryKey": true,
+ "notNull": true,
+ "default": "gen_random_uuid()"
+ },
+ "organization_id": {
+ "name": "organization_id",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "lightspeed_universe_id": {
+ "name": "lightspeed_universe_id",
+ "type": "uuid",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "name": {
+ "name": "name",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true
+ },
+ "icon": {
+ "name": "icon",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'orbit'"
+ },
+ "icon_color": {
+ "name": "icon_color",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'default'"
+ },
+ "gateway_url": {
+ "name": "gateway_url",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": false
+ },
+ "status": {
+ "name": "status",
+ "type": "text",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'active'"
+ },
+ "features": {
+ "name": "features",
+ "type": "jsonb",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "'{}'::jsonb"
+ },
+ "created_at": {
+ "name": "created_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ },
+ "updated_at": {
+ "name": "updated_at",
+ "type": "timestamp with time zone",
+ "primaryKey": false,
+ "notNull": true,
+ "default": "now()"
+ }
+ },
+ "indexes": {},
+ "foreignKeys": {
+ "universes_organization_id_organization_id_fk": {
+ "name": "universes_organization_id_organization_id_fk",
+ "tableFrom": "universes",
+ "tableTo": "organization",
+ "columnsFrom": [
+ "organization_id"
+ ],
+ "columnsTo": [
+ "id"
+ ],
+ "onDelete": "cascade",
+ "onUpdate": "no action"
+ }
+ },
+ "compositePrimaryKeys": {},
+ "uniqueConstraints": {
+ "universes_organization_id_unique": {
+ "name": "universes_organization_id_unique",
+ "nullsNotDistinct": false,
+ "columns": [
+ "organization_id"
+ ]
+ },
+ "universes_lightspeed_universe_id_unique": {
+ "name": "universes_lightspeed_universe_id_unique",
+ "nullsNotDistinct": false,
+ "columns": [
+ "lightspeed_universe_id"
+ ]
+ }
+ },
+ "policies": {},
+ "checkConstraints": {},
+ "isRLSEnabled": false
+ }
+ },
+ "enums": {},
+ "schemas": {},
+ "sequences": {},
+ "roles": {},
+ "policies": {},
+ "views": {},
+ "_meta": {
+ "columns": {},
+ "schemas": {},
+ "tables": {}
+ }
+}
\ No newline at end of file
diff --git a/platform/db/migrations/meta/_journal.json b/platform/db/migrations/meta/_journal.json
index 25df5e58c..2db71ac33 100644
--- a/platform/db/migrations/meta/_journal.json
+++ b/platform/db/migrations/meta/_journal.json
@@ -22,6 +22,13 @@
"when": 1790531942445,
"tag": "0002_company_identity",
"breakpoints": true
+ },
+ {
+ "idx": 3,
+ "version": "7",
+ "when": 1790863719262,
+ "tag": "0003_brief_purple_man",
+ "breakpoints": true
}
]
}
\ No newline at end of file
diff --git a/platform/db/scripts/check-migrations.ts b/platform/db/scripts/check-migrations.ts
index 54279798f..2b46a59be 100644
--- a/platform/db/scripts/check-migrations.ts
+++ b/platform/db/scripts/check-migrations.ts
@@ -89,6 +89,8 @@ async function requirePlatformShape(pool: pg.Pool): Promise {
await requireColumn(pool, "user", "oidc_subject");
await requireColumn(pool, "session", "access_version");
await requireColumn(pool, "universes", "lightspeed_universe_id");
+ await requireColumn(pool, "universes", "icon");
+ await requireColumn(pool, "universes", "icon_color");
for (const table of [
"bots",
"bot_triggers",
diff --git a/platform/db/src/schema/platform.ts b/platform/db/src/schema/platform.ts
index 00299bc24..3b170b40e 100644
--- a/platform/db/src/schema/platform.ts
+++ b/platform/db/src/schema/platform.ts
@@ -32,6 +32,8 @@ export const universes = pgTable("universes", {
.references(() => organization.id, { onDelete: "cascade" }),
lightspeedUniverseId: uuid("lightspeed_universe_id").notNull().unique(),
name: text("name").notNull(),
+ icon: text("icon").default("orbit").notNull(),
+ iconColor: text("icon_color").default("default").notNull(),
/// Gateway RPC endpoint; null = the deployment default from env.
gatewayUrl: text("gateway_url"),
status: text("status", { enum: ["active", "archived"] })
diff --git a/platform/server/src/routes/universe-appearance.test.ts b/platform/server/src/routes/universe-appearance.test.ts
new file mode 100644
index 000000000..d2ff0f31c
--- /dev/null
+++ b/platform/server/src/routes/universe-appearance.test.ts
@@ -0,0 +1,92 @@
+import { readFile } from "node:fs/promises";
+import { PGlite } from "@electric-sql/pglite";
+import { drizzle } from "drizzle-orm/pglite";
+import { Hono } from "hono";
+import { UNIVERSE_ICONS } from "@lightspeed/platform-shared";
+import { afterAll, beforeAll, expect, it } from "vitest";
+import type { ApiVariables, AppContext } from "../context.js";
+import { universeRoutes } from "./universes.js";
+
+const database = new PGlite();
+const id = "33333333-3333-4333-8333-333333333333";
+const migrations = new URL("../../../db/migrations/", import.meta.url);
+
+beforeAll(async () => {
+ const journal = JSON.parse(await readFile(new URL("meta/_journal.json", migrations), "utf8")) as { entries: { tag: string }[] };
+ await database.exec(await readFile(new URL(`${journal.entries[0]!.tag}.sql`, migrations), "utf8"));
+ // A retained universe must receive the same defaults as a fresh one.
+ await database.exec(`
+ INSERT INTO "user" (id, name, email) VALUES
+ ('admin', 'Admin', 'admin@example.test'), ('viewer', 'Viewer', 'viewer@example.test'),
+ ('contributor', 'Contributor', 'contributor@example.test'), ('operator', 'Operator', 'operator@example.test');
+ INSERT INTO organization (id, name, slug, created_at) VALUES ('org', 'Test', 'test', now());
+ INSERT INTO member (id, organization_id, user_id, role, created_at)
+ SELECT id, 'org', id, id, now() FROM "user";
+ INSERT INTO universes (id, organization_id, lightspeed_universe_id, name)
+ VALUES ('${id}', 'org', '${id}', 'Test');
+ `);
+ for (const entry of journal.entries.slice(1)) {
+ await database.exec(await readFile(new URL(`${entry.tag}.sql`, migrations), "utf8"));
+ }
+});
+afterAll(async () => database.close());
+
+function request(userId: string, method = "GET", body?: unknown, platformAdmin = false, path = `/${id}`) {
+ const app = new Hono<{ Variables: ApiVariables }>();
+ app.use("*", async (c, next) => {
+ c.set("session", { user: { id: userId, role: platformAdmin ? "admin" : "user" } } as ApiVariables["session"]);
+ await next();
+ });
+ app.route("/", universeRoutes({ db: drizzle(database) } as unknown as AppContext));
+ return app.request(path, { method, headers: { "content-type": "application/json" },
+ ...(body !== undefined ? { body: JSON.stringify(body) } : {}),
+ });
+}
+
+it("gives retained and newly created universes the default appearance", async () => {
+ expect(await (await request("viewer")).json()).toMatchObject({ icon: "orbit", iconColor: "default" });
+ await database.exec(`
+ INSERT INTO organization (id, name, slug, created_at) VALUES ('new-org', 'New', 'new', now());
+ INSERT INTO universes (organization_id, lightspeed_universe_id, name)
+ VALUES ('new-org', '44444444-4444-4444-8444-444444444444', 'New');
+ `);
+ const { rows } = await database.query("SELECT icon, icon_color FROM universes WHERE organization_id = 'new-org'");
+ expect(rows).toEqual([{ icon: "orbit", icon_color: "default" }]);
+});
+
+it.each(["viewer", "contributor", "operator"])("refuses appearance changes by a %s", async role => {
+ expect((await request(role, "PATCH", { icon: "rocket", iconColor: "blue" })).status).toBe(403);
+});
+
+it("persists an admin's selection for every member and list read", async () => {
+ const response = await request("admin", "PATCH", { icon: "rocket", iconColor: "blue" });
+ expect(response.status).toBe(200);
+ expect(await response.json()).toMatchObject({ icon: "rocket", iconColor: "blue" });
+ for (const role of ["viewer", "contributor", "operator", "admin"]) {
+ expect(await (await request(role)).json()).toMatchObject({ icon: "rocket", iconColor: "blue" });
+ expect(await (await request(role, "GET", undefined, false, "/")).json())
+ .toEqual(expect.arrayContaining([expect.objectContaining({ id, icon: "rocket", iconColor: "blue" })]));
+ }
+});
+
+it("allows a platform admin without membership and preserves omitted appearance fields", async () => {
+ expect((await request("platform-admin", "PATCH", { iconColor: "violet" }, true)).status).toBe(200);
+ expect((await request("admin", "PATCH", { name: "Renamed" })).status).toBe(200);
+ expect(await (await request("viewer")).json()).toMatchObject({ name: "Renamed", icon: "rocket", iconColor: "violet" });
+});
+
+it.each([{ icon: "unknown" }, { iconColor: "#ffffff" }, { icon: null }, { iconColor: null }])("rejects unsupported appearance values (%j)", async body => {
+ const before = await (await request("viewer")).json();
+ expect((await request("admin", "PATCH", body)).status).toBe(400);
+ expect(await (await request("viewer")).json()).toEqual(before);
+});
+
+it("lets admins restore the default appearance", async () => {
+ expect((await request("admin", "PATCH", { icon: "orbit", iconColor: "default" })).status).toBe(200);
+ expect(await (await request("viewer")).json()).toMatchObject({ icon: "orbit", iconColor: "default" });
+});
+
+it.each(UNIVERSE_ICONS)("persists the %s icon for other members", async icon => {
+ expect((await request("admin", "PATCH", { icon })).status).toBe(200);
+ expect(await (await request("viewer")).json()).toMatchObject({ icon });
+});
diff --git a/platform/shared/src/index.ts b/platform/shared/src/index.ts
index 1a25d6bfc..f0af9faca 100644
--- a/platform/shared/src/index.ts
+++ b/platform/shared/src/index.ts
@@ -1,4 +1,6 @@
import { z } from "zod";
+import { universeIconSchema, universeIconColorSchema } from "./universe-appearance.js";
+export * from "./universe-appearance.js";
/// Input shapes shared by the API (validation) and the CLI (request typing).
@@ -14,6 +16,8 @@ export type UniverseCreateInput = z.infer;
export const universeUpdateSchema = z.object({
name: z.string().min(1).max(100).optional(),
+ icon: universeIconSchema.optional(),
+ iconColor: universeIconColorSchema.optional(),
gatewayUrl: z.union([z.url(), z.null()]).optional(),
status: z.enum(["active", "archived"]).optional(),
/// Switches to change; others keep what they were.
diff --git a/platform/shared/src/universe-appearance.ts b/platform/shared/src/universe-appearance.ts
new file mode 100644
index 000000000..0fb432197
--- /dev/null
+++ b/platform/shared/src/universe-appearance.ts
@@ -0,0 +1,16 @@
+import { z } from "zod";
+
+export const UNIVERSE_ICONS = [
+ "orbit", "globe", "rocket", "star", "sparkles", "zap", "sun", "moon",
+ "atom", "compass", "mountain", "leaf", "flame", "heart", "code", "briefcase",
+ "house", "music", "shield", "bot",
+ "anchor", "book", "camera", "coffee", "crown", "gem", "palette", "puzzle", "telescope", "cpu",
+] as const;
+export const UNIVERSE_ICON_COLORS = [
+ "default", "slate", "red", "orange", "amber", "green", "teal", "blue", "violet", "pink",
+] as const;
+
+export const universeIconSchema = z.enum(UNIVERSE_ICONS);
+export const universeIconColorSchema = z.enum(UNIVERSE_ICON_COLORS);
+export type UniverseIconName = z.infer;
+export type UniverseIconColor = z.infer;
diff --git a/platform/web/src/api.ts b/platform/web/src/api.ts
index ed48239da..09d902830 100644
--- a/platform/web/src/api.ts
+++ b/platform/web/src/api.ts
@@ -1,4 +1,4 @@
-import type { FeatureStates, UniverseRole } from "@lightspeed/platform-shared";
+import type { FeatureStates, UniverseRole, UniverseIconName, UniverseIconColor } from "@lightspeed/platform-shared";
export type { ModelConfig, ModelDefaults, ModelDefaultsPutParams } from "@lightspeed-ai/agent-client";
import type {
Attribution,
@@ -77,6 +77,8 @@ export async function api(method: string, path: string, body?: unknown, signa
}
export interface Universe {
+ icon?: UniverseIconName;
+ iconColor?: UniverseIconColor;
id: string;
lightspeedUniverseId: string;
name: string;
diff --git a/platform/web/src/components/bot/face.tsx b/platform/web/src/components/bot/face.tsx
index 6b4b3bf0e..a28fa6053 100644
--- a/platform/web/src/components/bot/face.tsx
+++ b/platform/web/src/components/bot/face.tsx
@@ -1,5 +1,6 @@
import { BotFace } from "@/components/icons/bot";
import { cn } from "@/lib/utils";
+import { identityColor } from "@/lib/identity-colors";
/// A bot keeps one colour everywhere it appears — roster, header, threads —
/// derived from its immutable id, so renaming never changes the face.
@@ -10,7 +11,7 @@ export function botHue(botId: string): number {
}
export function botColor(botId: string): string {
- return `oklch(0.58 0.11 ${botHue(botId)})`;
+ return identityColor(botHue(botId));
}
export function BotAvatar({
diff --git a/platform/web/src/components/universe-appearance-card.test.tsx b/platform/web/src/components/universe-appearance-card.test.tsx
new file mode 100644
index 000000000..81501d56d
--- /dev/null
+++ b/platform/web/src/components/universe-appearance-card.test.tsx
@@ -0,0 +1,114 @@
+// @vitest-environment jsdom
+import { act, type ReactNode } from "react";
+import { createRoot, type Root } from "react-dom/client";
+import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
+import { MemoryRouter } from "react-router-dom";
+import { afterEach, beforeEach, expect, it, vi } from "vitest";
+import type { Universe } from "@/api";
+import { UNIVERSE_ICONS } from "@lightspeed/platform-shared";
+import { PermissionIdentityProvider } from "@/lib/permissions";
+import { GeneralSettingsPage } from "@/pages/GeneralSettingsPage";
+import { UniverseAppearanceCard } from "./universe-appearance-card";
+import { UniverseIcon } from "./universe-icon";
+
+const mocks = vi.hoisted(() => ({ api: vi.fn(), universe: {} as Universe }));
+vi.mock("@/api", async original => ({ ...await original(), api: mocks.api }));
+vi.mock("@/lib/universes", async original => ({
+ ...await original(),
+ useActiveUniverse: () => ({ universe: mocks.universe, slug: mocks.universe.slug, isLoading: false }),
+}));
+
+let root: Root;
+let container: HTMLDivElement;
+let client: QueryClient;
+beforeEach(() => {
+ vi.stubGlobal("IS_REACT_ACT_ENVIRONMENT", true);
+ mocks.universe = { id: "universe", lightspeedUniverseId: "runtime", name: "Test", slug: "test",
+ gatewayUrl: null, status: "active", createdAt: "2026-01-01", role: "admin",
+ features: { bots: true, channels: true }, icon: "orbit", iconColor: "default" };
+ mocks.api.mockReset().mockImplementation(async (_method, _path, fields) => ({ ...mocks.universe, ...fields }));
+ client = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: Infinity } } });
+ client.setQueryData(["universes"], [mocks.universe]);
+ container = document.createElement("div");
+ document.body.append(container);
+ root = createRoot(container);
+});
+afterEach(async () => {
+ await act(async () => root.unmount());
+ client.clear(); container.remove(); vi.unstubAllGlobals();
+});
+async function render(node: ReactNode) {
+ await act(async () => root.render({node}));
+}
+function button(label: string) {
+ const result = Array.from(container.querySelectorAll("button"))
+ .find(button => button.getAttribute("aria-label") === label || button.textContent === label);
+ expect(result, `Missing button: ${label}`).toBeDefined();
+ return result!;
+}
+async function click(label: string) { await act(async () => button(label).click()); }
+
+it.each(UNIVERSE_ICONS)("previews the %s icon and saves it to the shared universe cache", async icon => {
+ const iconLabel = `${icon.charAt(0).toUpperCase() + icon.slice(1)} icon`;
+ function CachedIcon() {
+ const universe = client.getQueryData(["universes"])![0]!;
+ return ;
+ }
+ await render();
+ expect(button("Save appearance").disabled).toBe(true);
+ await click(iconLabel); await click("Blue color");
+ expect(button(iconLabel).getAttribute("aria-pressed")).toBe("true");
+ expect(button("Blue color").getAttribute("aria-pressed")).toBe("true");
+ const preview = container.querySelector('[aria-label="Universe appearance preview"]')!;
+ expect(preview.querySelector(`.lucide-${icon}`)).not.toBeNull();
+ expect(preview.querySelector("span[style]")?.getAttribute("style")).toContain("oklch(0.58 0.11 255)");
+ expect(mocks.api).not.toHaveBeenCalled();
+ await click("Save appearance");
+ await vi.waitFor(() => expect(client.getQueryData(["universes"])![0]).toMatchObject({ icon, iconColor: "blue" }));
+ expect(mocks.api).toHaveBeenCalledWith("PATCH", "/api/v1/universes/universe", { icon, iconColor: "blue" });
+ await render();
+ expect(container.querySelector(`.lucide-${icon}`)).not.toBeNull();
+ expect(container.querySelector("span[style]")?.getAttribute("style")).toContain("oklch(0.58 0.11 255)");
+});
+
+it("keeps the saved appearance when a save fails and allows retry", async () => {
+ mocks.api.mockRejectedValueOnce(new Error("Unable to save"));
+ await render();
+ await click("Star icon"); await click("Save appearance");
+ await vi.waitFor(() => expect(container.querySelector('[role="alert"]')?.textContent).toBe("Unable to save"));
+ expect(client.getQueryData(["universes"])![0]).toMatchObject({ icon: "orbit", iconColor: "default" });
+ expect(button("Star icon").getAttribute("aria-pressed")).toBe("true");
+ await click("Save appearance");
+ await vi.waitFor(() => expect(client.getQueryData(["universes"])![0]).toMatchObject({ icon: "star" }));
+});
+
+it("restores defaults only after saving", async () => {
+ mocks.universe = { ...mocks.universe, icon: "heart", iconColor: "pink" };
+ await render();
+ await click("Reset to default");
+ expect(mocks.api).not.toHaveBeenCalled();
+ expect(button("Orbit icon").getAttribute("aria-pressed")).toBe("true");
+ expect(button("Default color").getAttribute("aria-pressed")).toBe("true");
+ await click("Save appearance");
+ expect(mocks.api).toHaveBeenCalledWith("PATCH", "/api/v1/universes/universe", { icon: "orbit", iconColor: "default" });
+});
+
+it.each(["viewer", "contributor", "operator"] as const)("hides General settings from a %s", async role => {
+ mocks.universe.role = role;
+ await render();
+ expect(container.textContent).not.toContain("Appearance");
+ expect(container.querySelector("form")).toBeNull();
+});
+
+it.each([false, true])("offers appearance on General settings for an authorized admin (platform admin: %s)", async platformAdmin => {
+ mocks.universe.role = platformAdmin ? null : "admin";
+ await render();
+ expect(container.textContent).toContain("Appearance");
+ expect(button("Save appearance")).toBeDefined();
+});
+
+it("renders the default badge when appearance is absent", async () => {
+ await render();
+ expect(container.querySelector(".lucide-orbit")).not.toBeNull();
+ expect(container.querySelector(".bg-sidebar-primary")).not.toBeNull();
+});
diff --git a/platform/web/src/components/universe-appearance-card.tsx b/platform/web/src/components/universe-appearance-card.tsx
new file mode 100644
index 000000000..680559c58
--- /dev/null
+++ b/platform/web/src/components/universe-appearance-card.tsx
@@ -0,0 +1,75 @@
+import { useState, type FormEvent } from "react";
+import { useMutation, useQueryClient } from "@tanstack/react-query";
+import { UNIVERSE_ICONS, UNIVERSE_ICON_COLORS, type UniverseIconName, type UniverseIconColor } from "@lightspeed/platform-shared";
+import { api, type Universe } from "@/api";
+import { Button } from "@/components/ui/button";
+import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
+import { UniverseIcon } from "@/components/universe-icon";
+
+export function UniverseAppearanceCard({ universe }: { universe: Universe }) {
+ const queryClient = useQueryClient();
+ const [icon, setIcon] = useState(universe.icon ?? "orbit");
+ const [iconColor, setIconColor] = useState(universe.iconColor ?? "default");
+ const changed = icon !== (universe.icon ?? "orbit") || iconColor !== (universe.iconColor ?? "default");
+ const save = useMutation({
+ mutationFn: () => api("PATCH", `/api/v1/universes/${universe.id}`, { icon, iconColor }),
+ onSuccess: (updated) => {
+ queryClient.setQueryData(["universes"], rows => rows?.map(row => row.id === updated.id ? updated : row));
+ void queryClient.invalidateQueries({ queryKey: ["universes"] });
+ },
+ });
+ function submit(event: FormEvent) {
+ event.preventDefault();
+ if (changed && !save.isPending) save.mutate();
+ }
+ return (
+
+
+ Appearance
+ The icon and color shown in the universe switcher for everyone.
+
+
+
+
+
+ );
+}
+
+function label(value: string) { return value.charAt(0).toUpperCase() + value.slice(1); }
diff --git a/platform/web/src/components/universe-icon.tsx b/platform/web/src/components/universe-icon.tsx
new file mode 100644
index 000000000..f8df3429d
--- /dev/null
+++ b/platform/web/src/components/universe-icon.tsx
@@ -0,0 +1,34 @@
+import {
+ Orbit, Globe, Rocket, Star, Sparkles, Zap, Sun, Moon, Atom, Compass,
+ Mountain, Leaf, Flame, Heart, Code, Briefcase, House, Music, Shield, Bot, type LucideIcon,
+ Anchor, Book, Camera, Coffee, Crown, Gem, Palette, Puzzle, Telescope, Cpu,
+} from "lucide-react";
+import type { UniverseIconName, UniverseIconColor } from "@lightspeed/platform-shared";
+import { cn } from "@/lib/utils";
+import { UNIVERSE_ICON_BACKGROUNDS } from "@/lib/identity-colors";
+
+const ICONS: Record = {
+ orbit: Orbit, globe: Globe, rocket: Rocket, star: Star, sparkles: Sparkles,
+ zap: Zap, sun: Sun, moon: Moon, atom: Atom, compass: Compass,
+ mountain: Mountain, leaf: Leaf, flame: Flame, heart: Heart, code: Code, briefcase: Briefcase,
+ house: House, music: Music, shield: Shield, bot: Bot,
+ anchor: Anchor, book: Book, camera: Camera, coffee: Coffee, crown: Crown,
+ gem: Gem, palette: Palette, puzzle: Puzzle, telescope: Telescope, cpu: Cpu,
+};
+
+export function UniverseIcon({ icon = "orbit", iconColor = "default", className }: {
+ icon?: UniverseIconName;
+ iconColor?: UniverseIconColor;
+ className?: string;
+}) {
+ const Icon = ICONS[icon] ?? Orbit;
+ const background = UNIVERSE_ICON_BACKGROUNDS[iconColor] ?? UNIVERSE_ICON_BACKGROUNDS.default;
+ const foreground = background === UNIVERSE_ICON_BACKGROUNDS.default
+ ? "bg-sidebar-primary text-sidebar-primary-foreground"
+ : "text-white";
+ return (
+
+
+
+ );
+}
diff --git a/platform/web/src/components/universe-switcher.tsx b/platform/web/src/components/universe-switcher.tsx
index 8e76c2c93..b6683f1e5 100644
--- a/platform/web/src/components/universe-switcher.tsx
+++ b/platform/web/src/components/universe-switcher.tsx
@@ -2,7 +2,8 @@ import { slugify, universeSlugSchema } from "@lightspeed/platform-shared";
import { useState, type FormEvent } from "react";
import { useMutation, useQueryClient } from "@tanstack/react-query";
import { useNavigate } from "react-router-dom";
-import { Check, ChevronsUpDown, Orbit, Plus } from "lucide-react";
+import { Check, ChevronsUpDown, Plus } from "lucide-react";
+import { UniverseIcon } from "@/components/universe-icon";
import { api, type Universe } from "@/api";
import { Button } from "@/components/ui/button";
import {
@@ -61,9 +62,7 @@ export function UniverseSwitcher({
/>
}
>
-
-
-
+
{active?.name ?? "Select universe"}
@@ -87,6 +86,7 @@ export function UniverseSwitcher({
key={universe.id}
onClick={() => navigate(universeHome(universe))}
>
+ {universe.name}
{universe.id === active?.id && }
diff --git a/platform/web/src/demo/fixtures/personal-assistant.ts b/platform/web/src/demo/fixtures/personal-assistant.ts
index 54080040c..28c8755a0 100644
--- a/platform/web/src/demo/fixtures/personal-assistant.ts
+++ b/platform/web/src/demo/fixtures/personal-assistant.ts
@@ -2693,6 +2693,8 @@ export function seedPersonalAssistant(store: DemoStore): void {
id: PERSONAL_ASSISTANT_UNIVERSE_ID,
slug: PERSONAL_ASSISTANT_SLUG,
name: "Personal Assistant",
+ icon: "sparkles",
+ iconColor: "violet",
lightspeedUniverseId: LIGHTSPEED_UNIVERSE_ID,
role: "admin",
createdAt: agoIso(5 * 7 * DAY_MS),
diff --git a/platform/web/src/demo/fixtures/software-factory.ts b/platform/web/src/demo/fixtures/software-factory.ts
index 45b13d97a..24bab6d2b 100644
--- a/platform/web/src/demo/fixtures/software-factory.ts
+++ b/platform/web/src/demo/fixtures/software-factory.ts
@@ -4429,6 +4429,8 @@ export function seedSoftwareFactory(store: DemoStore): void {
id: SOFTWARE_FACTORY_UNIVERSE_ID,
slug: SOFTWARE_FACTORY_SLUG,
name: "Software Factory",
+ icon: "code",
+ iconColor: "blue",
lightspeedUniverseId: ENGINE_UNIVERSE_ID,
role: "admin",
createdAt: agoIso(70 * DAY_MS),
diff --git a/platform/web/src/demo/fixtures/technical-support.ts b/platform/web/src/demo/fixtures/technical-support.ts
index b72407991..f83b8f51f 100644
--- a/platform/web/src/demo/fixtures/technical-support.ts
+++ b/platform/web/src/demo/fixtures/technical-support.ts
@@ -2330,6 +2330,8 @@ export function seedTechnicalSupport(store: DemoStore): void {
id: TECHNICAL_SUPPORT_UNIVERSE_ID,
slug: TECHNICAL_SUPPORT_SLUG,
name: "Technical Support",
+ icon: "shield",
+ iconColor: "teal",
lightspeedUniverseId: LIGHTSPEED_UNIVERSE_ID,
role: "admin",
createdAt: agoIso(49 * DAY_MS),
diff --git a/platform/web/src/demo/router.test.ts b/platform/web/src/demo/router.test.ts
index 352c68fbc..b248022cc 100644
--- a/platform/web/src/demo/router.test.ts
+++ b/platform/web/src/demo/router.test.ts
@@ -63,6 +63,35 @@ const universeReads = [
];
describe("demo router", () => {
+ it("starts demo universes with distinct icons and colors while new universes use defaults", async () => {
+ const { call } = await boot();
+ const response = await call("GET", "/api/v1/universes");
+ expect(response.status).toBe(200);
+ const universes = response.json as Universe[];
+ expect(universes).toHaveLength(3);
+ expect(new Set(universes.map(universe => universe.icon)).size).toBe(universes.length);
+ expect(new Set(universes.map(universe => universe.iconColor)).size).toBe(universes.length);
+ for (const universe of universes) {
+ expect(universe.icon).toBeTruthy();
+ expect(universe.icon).not.toBe("orbit");
+ expect(universe.iconColor).toBeTruthy();
+ expect(universe.iconColor).not.toBe("default");
+ }
+ const created = await call("POST", "/api/v1/universes", { name: "New universe" });
+ expect(created.status).toBe(201);
+ expect(created.json).toMatchObject({ icon: "orbit", iconColor: "default" });
+ });
+ it("saves shared universe appearance and validates choices", async () => {
+ const { call } = await boot();
+ const path = `/api/v1/universes/${SOFTWARE_FACTORY_UNIVERSE_ID}`;
+ expect((await call("PATCH", path, { icon: "rocket", iconColor: "violet" })).status).toBe(200);
+ expect((await call("GET", path)).json).toMatchObject({ icon: "rocket", iconColor: "violet" });
+ expect((await call("GET", "/api/v1/universes")).json).toEqual(expect.arrayContaining([
+ expect.objectContaining({ id: SOFTWARE_FACTORY_UNIVERSE_ID, icon: "rocket", iconColor: "violet" }),
+ ]));
+ expect((await call("PATCH", path, { icon: "star", iconColor: "invalid" })).status).toBe(400);
+ expect((await call("GET", path)).json).toMatchObject({ icon: "rocket", iconColor: "violet" });
+ });
it("keeps model defaults universe-scoped and revision-safe while preserving existing session models", async () => {
const { store, call } = await boot();
const base = `/api/v1/universes/${SOFTWARE_FACTORY_UNIVERSE_ID}`;
diff --git a/platform/web/src/demo/routes/platform.ts b/platform/web/src/demo/routes/platform.ts
index 4ed13ea23..2d71f45e7 100644
--- a/platform/web/src/demo/routes/platform.ts
+++ b/platform/web/src/demo/routes/platform.ts
@@ -2,7 +2,7 @@
/// universe API keys. The demo user is a platform admin, so every gate the
/// real server applies passes.
import { Hono } from "hono";
-import { effectiveFeatures, featureOverridesSchema, memberUpdateSchema, mergeFeatureOverrides, slugify, universeRoleSchema, universeSlugSchema } from "@lightspeed/platform-shared";
+import { effectiveFeatures, featureOverridesSchema, memberUpdateSchema, mergeFeatureOverrides, slugify, universeRoleSchema, universeSlugSchema, universeUpdateSchema } from "@lightspeed/platform-shared";
import type { MethodGroup } from "@lightspeed-ai/agent-client";
import type { EngineUniverse, Member, Universe } from "@/api";
import { universeApiKey, type DemoStore, type UniverseState } from "../store";
@@ -116,6 +116,10 @@ export function platformRoutes(store: DemoStore): Hono {
const state = universeFor(store, c);
if (!state) return notFound(c);
const body = await readBody & { features: unknown }>>(c);
+ const parsed = universeUpdateSchema.safeParse(body);
+ if (!parsed.success) return badRequest(c, "invalid universe settings");
+ if (parsed.data.icon !== undefined) state.universe.icon = parsed.data.icon;
+ if (parsed.data.iconColor !== undefined) state.universe.iconColor = parsed.data.iconColor;
if (body.features !== undefined) {
const features = featureOverridesSchema.safeParse(body.features);
if (!features.success) return c.json({ error: "unknown feature" }, 400);
diff --git a/platform/web/src/demo/store.ts b/platform/web/src/demo/store.ts
index cd62e6b62..4f62df132 100644
--- a/platform/web/src/demo/store.ts
+++ b/platform/web/src/demo/store.ts
@@ -1,6 +1,6 @@
/// In-memory state behind the browser demo. Fixtures fill it at boot, the
/// stub routes read and mutate it, and nothing survives a reload.
-import { effectiveFeatures, type FeatureOverrides, type MessageAttachment, type UniverseRole } from "@lightspeed/platform-shared";
+import { effectiveFeatures, type FeatureOverrides, type MessageAttachment, type UniverseRole, type UniverseIconName, type UniverseIconColor } from "@lightspeed/platform-shared";
import type {
BlobContent,
ChannelsStatus,
@@ -164,6 +164,8 @@ export interface UniverseInit {
id?: string;
slug: string;
name: string;
+ icon?: UniverseIconName;
+ iconColor?: UniverseIconColor;
lightspeedUniverseId?: string;
/// Membership role of the demo user; null = platform admin browsing.
role?: UniverseRole | null;
@@ -242,6 +244,8 @@ export class DemoStore {
id,
lightspeedUniverseId: init.lightspeedUniverseId ?? crypto.randomUUID(),
name: init.name,
+ icon: init.icon ?? "orbit",
+ iconColor: init.iconColor ?? "default",
slug: init.slug,
gatewayUrl: null,
status: "active",
diff --git a/platform/web/src/lib/identity-colors.ts b/platform/web/src/lib/identity-colors.ts
new file mode 100644
index 000000000..5cc4507c9
--- /dev/null
+++ b/platform/web/src/lib/identity-colors.ts
@@ -0,0 +1,19 @@
+import type { UniverseIconColor } from "@lightspeed/platform-shared";
+
+/// Shared lightness and saturation keep identity badges visually consistent.
+export function identityColor(hue: number, chroma = 0.11): string {
+ return `oklch(0.58 ${chroma} ${hue})`;
+}
+
+export const UNIVERSE_ICON_BACKGROUNDS: Record = {
+ default: "var(--sidebar-primary)",
+ slate: identityColor(260, 0.02),
+ red: identityColor(25),
+ orange: identityColor(55),
+ amber: identityColor(85),
+ green: identityColor(145),
+ teal: identityColor(185),
+ blue: identityColor(255),
+ violet: identityColor(300),
+ pink: identityColor(345),
+};
diff --git a/platform/web/src/pages/GeneralSettingsPage.tsx b/platform/web/src/pages/GeneralSettingsPage.tsx
index 029ef8e8d..70b4d15a5 100644
--- a/platform/web/src/pages/GeneralSettingsPage.tsx
+++ b/platform/web/src/pages/GeneralSettingsPage.tsx
@@ -33,6 +33,7 @@ import {
import { Switch } from "@/components/ui/switch";
import { universeSlugSchema, FEATURES, FEATURE_KEYS, type FeatureKey } from "@lightspeed/platform-shared";
import { useActiveUniverse } from "@/lib/universes";
+import { UniverseAppearanceCard } from "@/components/universe-appearance-card";
export function GeneralSettingsPage({ admin: _admin }: { admin: boolean }) {
const { universe, slug, isLoading } = useActiveUniverse();
@@ -47,11 +48,12 @@ export function GeneralSettingsPage({ admin: _admin }: { admin: boolean }) {
return (
<>
-
+
+
diff --git a/release/metadata.env b/release/metadata.env
index 86069c8f4..269721b7e 100644
--- a/release/metadata.env
+++ b/release/metadata.env
@@ -14,5 +14,5 @@ LIGHTSPEED_ENVIRONMENT_PROTOCOL_VERSION=2
LIGHTSPEED_SCHEMA_REVISION=10
# Drizzle journal length and the oldest platform migration boundary accepted by
# the current release's automated upgrade gate.
-LIGHTSPEED_PLATFORM_SCHEMA_REVISION=3
+LIGHTSPEED_PLATFORM_SCHEMA_REVISION=4
LIGHTSPEED_PLATFORM_UPGRADE_FROM=0000_platform_baseline
From f26ada27370efae0d1e922932d72beb8c130c018 Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:33:32 +0200
Subject: [PATCH 04/28] condext item fix doc
---
...-safe-media-and-context-entry-redaction.md | 278 ++++++++++++++++++
1 file changed, 278 insertions(+)
create mode 100644 docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
diff --git a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
new file mode 100644
index 000000000..15a36181f
--- /dev/null
+++ b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
@@ -0,0 +1,278 @@
+# P186 — Provider-safe media and context entry redaction
+
+**Status:** Proposed, 2026-10-01. Revises the request-time media rules of
+[tool result media](p171-tool-result-media.md).
+
+## Outcome
+
+A session's history can always be sent to its provider. Media that was
+admitted once never makes a later request invalid, however many images
+accumulate. When a provider still rejects a request for a reason the runtime
+cannot predict, the failure says so plainly. An operator then repairs the
+session with one public API call that appends an ordinary event, instead of
+rewriting the session log.
+
+Four changes deliver this, the same way for every provider API kind:
+
+1. Every image is sent as a normalized copy for the model, bounded in pixels
+ and bytes. Stored originals are unchanged.
+2. Each adapter checks the whole lowered request against its provider's
+ request limits before sending it, and omits the oldest media when it would
+ not fit.
+3. A provider rejecting a request is reported as a distinct run failure,
+ `RequestRejected`, carrying the provider's message word for word.
+4. `session/context/redact` replaces the content of chosen entries with a
+ fixed placeholder, in place, so an operator can neutralize the entry that
+ causes a rejection.
+
+## Incident
+
+A long-running agent session accumulated 32 images through tool results,
+one of them 2166 × 2464 px. Each image passed admission when it was produced.
+Once a request carried more than 20 images, Anthropic applied its many-image
+rule (every image at most 2000 px per side) and rejected the whole request
+with `invalid_request_error`. Every later run resent the same history and
+failed the same way, including the user's request to make the images smaller.
+Recovery required restoring the session from a backup and rewriting its event
+log by hand, because no public command could reach the offending entry.
+
+The cause was specific to one provider, but the failure mode is not: any
+provider can reject a history that grew past one of its limits or that holds
+one entry it no longer accepts, and the runtime has no supported way out.
+
+## Baseline
+
+- [Tool result media](p171-tool-result-media.md) assumes providers downscale
+ oversized images themselves and states that request failures are
+ forbidden. The first part is true only per image: the many-image rule
+ rejects rather than downscales, and it counts images from earlier turns.
+- Admission (`engine::media::admit_tool_media`, gateway run input) checks media
+ type, a 10 MB raw byte limit, and at most eight items per result or run.
+ Dimensions, image counts across the session, and request totals are never
+ checked.
+- All three adapters read the original blob and base64-encode it on every
+ request (`llm-runtime` `blob_io::read_base64`). The 10 MB raw admission
+ limit therefore admits images over Anthropic's 10 MB *encoded* limit, and a
+ few dozen ordinary images exceed the 32 MB request limit.
+- Compaction requests lower the same entries, images included, so compaction
+ fails on the same history.
+- Provider errors are classified provider-neutrally (`ProviderFailureKind` in
+ `llm-clients`), but `LlmRuntime` passes on only the retry decision. Every
+ terminal provider error becomes an untyped message, and the run fails as a
+ generic `ModelFailure`.
+- Context ordering is entry-ID ordering: active entries keep strictly
+ increasing `entry_id`, and a keyed upsert removes the old entry and appends
+ its replacement at the tail. External context commands (`UpsertContext`,
+ `ReplaceContextPrefix`, `RemoveContext`) address entries only by key.
+ Run-appended entries have no key and are unreachable.
+- Tool calls and tool results are separate entries (`ToolCall`, `ToolResult`),
+ one per call. Tool-produced media are further separate user-role entries
+ that follow the result.
+- Every provider requires each tool call to be answered: Anthropic a
+ `tool_result` per `tool_use`, OpenAI Responses a `function_call_output` per
+ `function_call`, Chat Completions a `tool` message per `tool_calls` id.
+ Removing calls is not safe either: an OpenAI Responses reasoning item must be
+ followed by the item it produced, and Anthropic signs thinking across the
+ assistant turn that holds the calls.
+
+### Provider limits
+
+The request limits check (Decision 2) keeps one row per provider API kind.
+Slice 2 is complete only when every API kind has its row.
+
+Anthropic Messages, first-party API, as of 2026-10-01:
+
+| Limit | Value |
+| --- | --- |
+| Per image, any request | 8000 × 8000 px |
+| Per image, request with more than 20 image blocks | 2000 px per side; earlier turns and `tool_result` images count |
+| Images per request | 600 (100 for 200k-context models) |
+| Per image, base64-encoded | 10 MB |
+| Request body | 32 MB |
+| Native resolution, high-resolution tier (Claude 4.7 and later) | 2576 px long edge, 4784 visual tokens; larger images are downscaled server-side |
+
+OpenAI Responses and Chat Completions rows are taken from current provider
+documentation during implementation.
+
+## Decisions
+
+### 1. Normalize every image once, with a fixed cap
+
+The adapter sends a normalized copy for the model instead of the original
+bytes. This is a property of lowering, not of admission or session state:
+the stored blob, the entry's `ContentRef`, its `media:` handle, and every
+projection keep referring to the original. A session may change models within
+its provider, so a copy computed for one request shape must not be baked into
+durable state.
+
+- **The cap is fixed and independent of the request.** No side may exceed
+ 2000 px. A cap that depended on the request (for example, 2576 px until the
+ 21st image) would change the bytes of earlier images partway through a
+ session, which invalidates the prompt cache and edits history as the
+ provider sees it. With a fixed cap, an image is lowered identically from its
+ first request to its last. On high-resolution models this gives up at most
+ the 2000–2576 px range, which the provider would otherwise downscale itself.
+- **A byte budget per image.** If a resized image is still above the
+ per-image budget (initially 3.75 MB raw, which is 5 MB encoded), it is
+ re-encoded as JPEG at a fixed quality, with any alpha channel flattened
+ onto white.
+- **Compliant images pass through byte-identical.** Dimensions come from the
+ image header without a full decode. An image within both the pixel cap and
+ the byte budget is sent unchanged, so existing sessions without oversized
+ images see no prompt-cache change.
+- **Output is deterministic.** For a given source and normalization spec, the
+ output bytes are always the same: fixed resampling filter, fixed encoder
+ settings, pinned codec crate version. Caching is therefore only an
+ optimization. A worker-local LRU keyed by (source blob ref, spec version)
+ avoids repeated decodes; a cache miss on another worker yields the same
+ bytes. A persistent copy store is added only if measurement calls for it.
+- **GIF and animated images** lower their first frame, which matches what
+ providers read.
+- **PDFs are not normalized.** They count toward request totals in Decision 2.
+- One shared function in `llm-runtime` serves all three adapters and the
+ compaction request path, which shares lowering. The engine is unchanged.
+
+### 2. Check the whole request against provider limits
+
+After lowering, each adapter compares the request with its row of the limits
+table: image block count, per-image encoded bytes, and total body bytes.
+
+- **When it fits, nothing changes.**
+- **When it does not fit, the oldest media is omitted.** Media entries are
+ replaced, oldest first, with the placeholder text the text-only path already
+ uses:
+ `[image · media:3f9a2c1d4e7b · omitted from this request to stay within provider limits]`.
+ Media from the current run's input and from the latest tool batch is never
+ omitted, because the model must see what it just asked for.
+- **The cut point moves in steps.** Which media is omitted is a pure function
+ of the context, so retries produce identical requests. The boundary moves
+ only when the request crosses the limit, and then it drops to a low-water
+ mark well below the limit, so a growing session invalidates its prompt
+ cache rarely rather than every turn.
+- **If the protected newest media alone does not fit**, the adapter fails the
+ turn with a typed error before any provider call. Admission bounds a single
+ result to eight items within the per-image budget, so this needs unusually
+ wide parallel tool batches.
+
+Omission rewrites earlier user content as the provider sees it, in sessions
+that are still healthy. Decision 1 keeps it rare, because pixel limits no
+longer trigger it; only request totals do.
+
+### 3. Report provider rejections as `RequestRejected`
+
+A terminal provider error classified as `InvalidRequest` or `ContextLength`
+fails the run with a new `RunFailureKind::RequestRejected`. Every other terminal
+provider error stays `ModelFailure`.
+
+- The classification comes from the existing provider-neutral
+ `ProviderFailureKind`. The LLM I/O boundary gains a rejected outcome beside
+ `Failed`, the turn failure records it, and the run failure carries it. The
+ engine stays deterministic: it records the classification it is given and
+ never inspects provider errors.
+- The failure record keeps the provider's message word for word.
+- Clients show that the provider rejected the request, with that message.
+ They do not suggest a fix or guess which entry caused it: the runtime cannot
+ know, and provider messages differ in how precisely they point at a cause.
+- The public run failure view gains the new kind; the API contract and the
+ TypeScript consumers are regenerated.
+
+### 4. Redact context entries in place
+
+`session/context/redact { sessionId, entryIds }` replaces the content of each
+named entry with a fixed placeholder chosen by the engine. The engine gains a
+`RedactContextEntries { expected_revision, entry_ids }` command and an
+`EntriesRedacted { base_revision, entry_ids, reason }` context event.
+
+**In place, not removal.** A redacted entry keeps its entry ID, position, kind,
+role, and `call_id`. Removing a tool result would leave its call unanswered,
+which every provider rejects; removing the call as well breaks reasoning and
+thinking that providers bind to it. Swapping content under the same entry ID
+keeps both the ordering invariant and call/result pairing. A keyed upsert
+cannot do this, because it appends a new entry at the tail.
+
+**The engine chooses the placeholder.** Clients name entries; they never supply
+replacement content, so this is a repair operation, not a general
+context-editing API.
+
+| Entry | Placeholder |
+| --- | --- |
+| Tool result | `[tool result removed by operator]` |
+| Media (image or document) | `[image · media:3f9a2c1d4e7b · removed by operator]` |
+| User message (run input, steering, context edit) | `[message removed by operator]` |
+
+Rejected, request-level:
+
+- tool calls, assistant output, reasoning, and provider-opaque entries, whose
+ content is provider-signed or provider-shaped;
+- any redaction while a run is active, or while compaction is pending;
+- unconsumed run input or steering, by the existing guard.
+
+An ID that is not in active context, or is already redacted, reports `absent`,
+so retries are idempotent. The response reports a result per ID.
+
+The original content stays in the event log. The web transcript resolves
+`media:` handles from transcript history, so a redacted image still renders
+there, beside the redaction event. A sub-agent hand-off resolves links against
+the child's active context, so a redacted image no longer travels with it.
+
+Redaction is available through the API and the runtime CLI. There is no web
+affordance and no automatic redaction.
+
+## How a failing session recovers
+
+| Failure | Handled by | Automatic |
+| --- | --- | --- |
+| Image over a provider's pixel or per-image byte limit | Normalization (Decision 1): the request is never built | Yes |
+| Request over a provider's image count or total size | Request limits check (Decision 2): oldest media omitted | Yes |
+| Any other rejection: a limit not yet in the table, an entry a provider no longer accepts, a provider change | `RequestRejected` (Decision 3), then redaction by an operator (Decision 4) | No |
+
+The runtime repairs automatically only what it can predict. For anything
+else it cannot know which entry is at fault, and removing context on a guess
+is worse than a visible failure. The provider's message, shown word for word,
+is the operator's starting point.
+
+## Slices
+
+1. **Normalization.** Shared lowering function in `llm-runtime`, header probe,
+ fixed cap, byte budget, deterministic encoding, worker-local cache, all three
+ adapters. Tests: oversized PNG and JPEG are downscaled within the cap;
+ compliant images pass through byte-identical; output is identical across
+ calls; a 32-image history with a 2166 × 2464 image lowers within the
+ many-image rule.
+2. **Request limits check.** Limits rows for every API kind, stepped omission,
+ typed failure before the provider call. Tests: omission order, protected
+ newest media, identical requests across retries, the cut point holds steady
+ while under the limit.
+3. **Rejection and redaction.** `RequestRejected` through the I/O boundary,
+ turn, and run failure; the redaction command, event, and placeholders;
+ `session/context/redact`; CLI support; contract regeneration; replay vectors
+ for redaction, its rejections, and a redacted tool result lowering as its
+ placeholder with pairing intact on every adapter.
+
+Slice 1 alone resolves the incident class. Slice 3 is the general recovery
+path for any rejection the runtime cannot predict.
+
+## Non-goals
+
+- Changing compaction. Compaction inherits normalization because it shares
+ lowering; its request shape and triggers are unchanged.
+- Suggesting fixes in clients, or redacting automatically after a rejection.
+- Retrying after a rejection by matching provider error text.
+- Removing tool calls or call/result pairs, and redacting tool calls.
+- Client-supplied replacement content.
+- Storing copies for a specific provider in session state, or rewriting
+ existing events.
+- The Anthropic Files API as an alternative to base64 payloads.
+
+## Open questions
+
+- **Preserved thinking under omission.** Anthropic enforces a check on edited
+ history for newer accounts on some models, which may drop or reject replayed
+ thinking blocks after an edit. Omission (Decision 2) edits history in healthy
+ sessions, so run a live check of it on the default Anthropic model and
+ record the result here. Redaction applies only to sessions that already
+ fail, so the check does not gate it.
+- **Announcing the resize.** Whether a normalized image's announcement should
+ state the dimensions the model sees (for example, `· shown at 2000×1400 of
+ 4000×2800`) to support coordinate-based work. It is deterministic, so it
+ does not affect caching.
From 4dde5759ef13d4d51fb39c5eba815df58534b3ac Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:45:04 +0200
Subject: [PATCH 05/28] repair doc
---
...-safe-media-and-context-entry-redaction.md | 133 ++++++++++++++----
1 file changed, 109 insertions(+), 24 deletions(-)
diff --git a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
index 15a36181f..c467cdaa4 100644
--- a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
+++ b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
@@ -5,14 +5,15 @@
## Outcome
-A session's history can always be sent to its provider. Media that was
-admitted once never makes a later request invalid, however many images
-accumulate. When a provider still rejects a request for a reason the runtime
-cannot predict, the failure says so plainly. An operator then repairs the
-session with one public API call that appends an ordinary event, instead of
-rewriting the session log.
+A session can recover when its active context no longer fits or is no longer
+accepted by its provider, while preserving its durable history and as much
+useful information as possible. Known media limits are handled before sending
+the request. If protected newest media cannot fit, or a provider rejects a
+request for a reason the runtime cannot predict, the failure says so plainly.
+An operator can neutralize an offending entry with one public API call that
+appends an ordinary event, instead of rewriting the session log.
-Four changes deliver this, the same way for every provider API kind:
+Five changes deliver this, with provider-specific lowering and continuation:
1. Every image is sent as a normalized copy for the model, bounded in pixels
and bytes. Stored originals are unchanged.
@@ -24,6 +25,38 @@ Four changes deliver this, the same way for every provider API kind:
4. `session/context/redact` replaces the content of chosen entries with a
fixed placeholder, in place, so an operator can neutralize the entry that
causes a rejection.
+5. A repair that invalidates preserved thinking uses the provider's supported
+ continuation policy, so incompatible past reasoning does not itself prevent
+ the session from continuing.
+
+## Rationale: repair active context to preserve task continuity
+
+Active context is a repairable projection of the session's durable history.
+Compaction is one existing repair strategy: when context grows too large, it
+summarizes older material so the task can continue. Media normalization,
+request-time omission, and operator redaction address other reasons the
+provider can no longer use the context. They share the same objective:
+preserve the work already done and restore a usable conversation.
+
+The choice of remedy depends on how confidently the runtime can identify the
+problem. A known context or media limit permits an automatic repair. An
+unexplained rejection calls for a visible failure and a precise operator
+repair mechanism. The runtime does not choose entries to redact on a guess.
+
+A repair preserves task continuity, but cannot promise identical reasoning
+continuity. Compaction loses detail; omission hides older media; redaction
+neutralizes selected content. If a repair invalidates provider-bound thinking,
+the provider may also need to discard that reasoning. Losing it can require
+the model to reconstruct conclusions or repeat analysis, but does not require
+discarding the rest of the conversation or starting a new session. Important
+decisions, constraints, progress, and remaining work should be explicit in
+messages or workspace artifacts; recovery must also support existing sessions
+without such checkpoints.
+
+Durable context repairs use revision guards, safe execution boundaries, and
+ordinary audit events. Request-time transformations leave stored entries and
+original blobs intact. Compaction and redaction retain their own policies and
+implementations; this proposal does not introduce a generic repair framework.
## Incident
@@ -132,6 +165,10 @@ durable state.
- One shared function in `llm-runtime` serves all three adapters and the
compaction request path, which shares lowering. The engine is unchanged.
+Normalizing an existing history can change image bytes the provider has
+already seen. This migration uses the thinking-continuation policy in Decision
+5, just as omission and redaction do.
+
### 2. Check the whole request against provider limits
After lowering, each adapter compares the request with its row of the limits
@@ -150,9 +187,10 @@ table: image block count, per-image encoded bytes, and total body bytes.
mark well below the limit, so a growing session invalidates its prompt
cache rarely rather than every turn.
- **If the protected newest media alone does not fit**, the adapter fails the
- turn with a typed error before any provider call. Admission bounds a single
- result to eight items within the per-image budget, so this needs unusually
- wide parallel tool batches.
+ turn with a typed error before any provider call. Per-item admission does
+ not bound the aggregate request: eight images at the 5 MB encoded budget
+ already exceed a 32 MB body limit, even in a single result. Documents and
+ parallel tool batches can exceed it too.
Omission rewrites earlier user content as the provider sees it, in sessions
that are still healthy. Decision 1 keeps it rare, because pixel limits no
@@ -189,6 +227,8 @@ which every provider rejects; removing the call as well breaks reasoning and
thinking that providers bind to it. Swapping content under the same entry ID
keeps both the ordering invariant and call/result pairing. A keyed upsert
cannot do this, because it appends a new entry at the tail.
+Pairing alone does not preserve thinking bound to the earlier content;
+Decision 5 supplies the continuation policy after that content changes.
**The engine chooses the placeholder.** Clients name entries; they never supply
replacement content, so this is a repair operation, not a general
@@ -218,18 +258,57 @@ the child's active context, so a redacted image no longer travels with it.
Redaction is available through the API and the runtime CLI. There is no web
affordance and no automatic redaction.
+### 5. Continue after a repair invalidates preserved thinking
+
+Thinking compatibility is part of recovery for normalization of existing
+histories, omission, redaction, and compaction that retains thinking from
+earlier turns. It is an implementation requirement, not an optional check on
+sessions that are still healthy.
+
+The Anthropic adapter defaults to
+`thinking.block_binding.prefix_mismatch_behavior: "drop_block"` wherever the
+model and thinking mode support it, with the
+`thinking-binding-controls-2026-08-01` beta header. Anthropic then drops
+incompatible thinking and subsequent thinking blocks while retaining the
+other request content. This policy remains on subsequent requests and after
+restart; it is not a one-request retry setting. See the provider's
+[preserved-thinking contract](https://platform.claude.com/docs/en/build-with-claude/preserved-thinking).
+
+The adapter continues sending the stored thinking unchanged and records
+reported `input_transformations` for diagnostics. Original reasoning stays
+in the event history. New responses may still generate thinking. Other
+request errors continue through Decision 3; this policy does not make every
+provider rejection recoverable.
+
+Models or modes that cannot use this policy need a separately tested fallback
+that durably excludes affected historical thinking from future requests,
+without gaps or later reintroduction. The exclusion must survive restart and
+preserve a valid tool sequence. Recovery support for such a mode is not
+complete until that path is verified. Other adapters follow their native
+contracts; Anthropic's policy does not authorize stripping OpenAI reasoning
+items or other provider-opaque content.
+
+Provider wire settings and interpretation remain in the adapter. The engine
+records provider-neutral repair facts and performs no signature inspection.
+Stable normalization and stepped omission still matter: fewer history edits
+preserve more reasoning and more of the prompt cache.
+
## How a failing session recovers
| Failure | Handled by | Automatic |
| --- | --- | --- |
-| Image over a provider's pixel or per-image byte limit | Normalization (Decision 1): the request is never built | Yes |
-| Request over a provider's image count or total size | Request limits check (Decision 2): oldest media omitted | Yes |
-| Any other rejection: a limit not yet in the table, an entry a provider no longer accepts, a provider change | `RequestRejected` (Decision 3), then redaction by an operator (Decision 4) | No |
+| Context exceeds its token budget | Existing compaction policy: older context summarized | According to session policy |
+| Image over a provider's pixel or per-image byte limit | Normalization (Decision 1): a bounded copy is sent | Yes |
+| Request over a provider's image count or total size | Request limits check (Decision 2): oldest eligible media omitted; fails locally if it still cannot fit | When eligible media can make it fit |
+| Unexplained rejection caused by a redactable entry | `RequestRejected` (Decision 3), then redaction by an operator (Decision 4) | No |
+| A repair invalidates preserved thinking | Provider-specific continuation (Decision 5): incompatible past thinking excluded from model input | For verified model and mode combinations |
The runtime repairs automatically only what it can predict. For anything
else it cannot know which entry is at fault, and removing context on a guess
is worse than a visible failure. The provider's message, shown word for word,
is the operator's starting point.
+Redaction is a repair mechanism for selected entries, not a guarantee that
+every rejection is caused by content it can repair.
## Slices
@@ -248,14 +327,26 @@ is the operator's starting point.
`session/context/redact`; CLI support; contract regeneration; replay vectors
for redaction, its rejections, and a redacted tool result lowering as its
placeholder with pairing intact on every adapter.
-
-Slice 1 alone resolves the incident class. Slice 3 is the general recovery
-path for any rejection the runtime cannot predict.
+4. **Thinking continuation.** Anthropic adapter defaults, beta header, and
+ transformation diagnostics; verified fallback for unsupported modes before
+ claiming recovery support. Tests: unchanged histories retain valid thinking;
+ existing-image normalization, omission, redaction, and compaction that
+ retains thinking continue with prefix enforcement enabled; later turns and
+ a restart continue; original history is unchanged. A credentialed live
+ suite exercises an enforcing model and mode explicitly, rather than relying
+ on the account age or the default model.
+
+Slice 1 addresses the incident's per-image limit. Slice 3 supplies operator
+repair for unexplained rejections caused by redactable entries. Slice 4 must
+land with any slice that changes previously sent content on a model enforcing
+thinking binding; repair is complete only when the session can continue after
+the change.
## Non-goals
-- Changing compaction. Compaction inherits normalization because it shares
- lowering; its request shape and triggers are unchanged.
+- Changing compaction triggers or summarization strategy. Compaction inherits
+ the request safeguards and applicable thinking-continuation policy.
+- Introducing a generic context-repair framework.
- Suggesting fixes in clients, or redacting automatically after a rejection.
- Retrying after a rejection by matching provider error text.
- Removing tool calls or call/result pairs, and redacting tool calls.
@@ -266,12 +357,6 @@ path for any rejection the runtime cannot predict.
## Open questions
-- **Preserved thinking under omission.** Anthropic enforces a check on edited
- history for newer accounts on some models, which may drop or reject replayed
- thinking blocks after an edit. Omission (Decision 2) edits history in healthy
- sessions, so run a live check of it on the default Anthropic model and
- record the result here. Redaction applies only to sessions that already
- fail, so the check does not gate it.
- **Announcing the resize.** Whether a normalized image's announcement should
state the dimensions the model sees (for example, `· shown at 2000×1400 of
4000×2800`) to support coordinate-based work. It is deterministic, so it
From 6ebf3cdc9638ee276eb507d26185a31443d88007 Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 16:50:45 +0200
Subject: [PATCH 06/28] repair doc
---
...-safe-media-and-context-entry-redaction.md | 112 +++++++++++-------
1 file changed, 68 insertions(+), 44 deletions(-)
diff --git a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
index c467cdaa4..dcd7f6522 100644
--- a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
+++ b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
@@ -8,8 +8,8 @@
A session can recover when its active context no longer fits or is no longer
accepted by its provider, while preserving its durable history and as much
useful information as possible. Known media limits are handled before sending
-the request. If protected newest media cannot fit, or a provider rejects a
-request for a reason the runtime cannot predict, the failure says so plainly.
+the request. If a provider rejects a request for a reason the runtime cannot
+predict, the failure says so plainly.
An operator can neutralize an offending entry with one public API call that
appends an ordinary event, instead of rewriting the session log.
@@ -48,10 +48,7 @@ continuity. Compaction loses detail; omission hides older media; redaction
neutralizes selected content. If a repair invalidates provider-bound thinking,
the provider may also need to discard that reasoning. Losing it can require
the model to reconstruct conclusions or repeat analysis, but does not require
-discarding the rest of the conversation or starting a new session. Important
-decisions, constraints, progress, and remaining work should be explicit in
-messages or workspace artifacts; recovery must also support existing sessions
-without such checkpoints.
+discarding the rest of the conversation or starting a new session.
Durable context repairs use revision guards, safe execution boundaries, and
ordinary audit events. Request-time transformations leave stored entries and
@@ -111,7 +108,7 @@ one entry it no longer accepts, and the runtime has no supported way out.
### Provider limits
The request limits check (Decision 2) keeps one row per provider API kind.
-Slice 2 is complete only when every API kind has its row.
+Slice 3 is complete only when every API kind has its row.
Anthropic Messages, first-party API, as of 2026-10-01:
@@ -179,18 +176,26 @@ table: image block count, per-image encoded bytes, and total body bytes.
replaced, oldest first, with the placeholder text the text-only path already
uses:
`[image · media:3f9a2c1d4e7b · omitted from this request to stay within provider limits]`.
- Media from the current run's input and from the latest tool batch is never
- omitted, because the model must see what it just asked for.
+ Media from the current run's input and from the latest tool batch is
+ protected: it is omitted only after all older media, because the model must
+ see what it just asked for.
+- **Protected media degrades newest-first.** Per-item admission does not bound
+ the aggregate request: eight images at the 5 MB encoded budget already
+ exceed a 32 MB body limit, even in a single result, and documents and
+ parallel tool batches can exceed it too. When the protected media alone
+ does not fit, its newest items are kept and the rest receive the same
+ placeholder. Failing the turn instead would not help: the latest tool batch
+ stays the latest after the run fails, so every later request would fail the
+ same way.
- **The cut point moves in steps.** Which media is omitted is a pure function
of the context, so retries produce identical requests. The boundary moves
only when the request crosses the limit, and then it drops to a low-water
mark well below the limit, so a growing session invalidates its prompt
cache rarely rather than every turn.
-- **If the protected newest media alone does not fit**, the adapter fails the
- turn with a typed error before any provider call. Per-item admission does
- not bound the aggregate request: eight images at the 5 MB encoded budget
- already exceed a 32 MB body limit, even in a single result. Documents and
- parallel tool batches can exceed it too.
+- **The check never fails a request.** It manages media only, and a single
+ image always fits within the per-image budget. A request whose non-media
+ content alone exceeds the body limit is a context-size problem for
+ compaction, or a rejection under Decision 3.
Omission rewrites earlier user content as the provider sees it, in sessions
that are still healthy. Decision 1 keeps it rare, because pixel limits no
@@ -274,24 +279,42 @@ other request content. This policy remains on subsequent requests and after
restart; it is not a one-request retry setting. See the provider's
[preserved-thinking contract](https://platform.claude.com/docs/en/build-with-claude/preserved-thinking).
-The adapter continues sending the stored thinking unchanged and records
-reported `input_transformations` for diagnostics. Original reasoning stays
-in the event history. New responses may still generate thinking. Other
-request errors continue through Decision 3; this policy does not make every
-provider rejection recoverable.
+Setting the field also changes behavior on accounts the provider does not
+enforce by default (created before 2026-08-31): any value opts the request into
+enforcement, so mismatched thinking that such accounts currently pass to the
+model is dropped instead. This is the intended contract, and it makes every
+deployment behave the same.
+
+The adapter continues sending the stored thinking unchanged. Original
+reasoning stays in the event history. New responses may still generate
+thinking. Other request errors continue through Decision 3; this policy does
+not make every provider rejection recoverable.
+
+`drop_block` also absorbs history edits the runtime makes by mistake, which
+would otherwise surface as rejections. Two measures keep such bugs visible:
+
+- Production logs every reported `input_transformations` entry with its path
+ and reason, and exports a count of dropped blocks per session, so unexpected
+ drops are observable.
+- Live and CI suites that do not exercise a repair send `"error"`, so an
+ unintended history edit fails a test instead of being absorbed.
Models or modes that cannot use this policy need a separately tested fallback
that durably excludes affected historical thinking from future requests,
without gaps or later reintroduction. The exclusion must survive restart and
preserve a valid tool sequence. Recovery support for such a mode is not
-complete until that path is verified. Other adapters follow their native
+complete until that path is verified. Anthropic accepts `block_binding` only
+with `adaptive` and `enabled` thinking, so this applies today to `disabled`,
+which the adapter sends for reasoning effort `none`, and would apply to
+`between_tools` if the adapter adopts it. Other adapters follow their native
contracts; Anthropic's policy does not authorize stripping OpenAI reasoning
items or other provider-opaque content.
-Provider wire settings and interpretation remain in the adapter. The engine
-records provider-neutral repair facts and performs no signature inspection.
-Stable normalization and stepped omission still matter: fewer history edits
-preserve more reasoning and more of the prompt cache.
+Provider wire settings, their interpretation, and dropped-block diagnostics
+remain in the adapter. This decision does not change the engine or the
+public contract, and nothing inspects signatures. Stable normalization and
+stepped omission still matter: fewer history edits preserve more reasoning
+and more of the prompt cache.
## How a failing session recovers
@@ -299,7 +322,7 @@ preserve more reasoning and more of the prompt cache.
| --- | --- | --- |
| Context exceeds its token budget | Existing compaction policy: older context summarized | According to session policy |
| Image over a provider's pixel or per-image byte limit | Normalization (Decision 1): a bounded copy is sent | Yes |
-| Request over a provider's image count or total size | Request limits check (Decision 2): oldest eligible media omitted; fails locally if it still cannot fit | When eligible media can make it fit |
+| Request over a provider's image count or total size | Request limits check (Decision 2): oldest media omitted, protected media last and newest-first | Yes |
| Unexplained rejection caused by a redactable entry | `RequestRejected` (Decision 3), then redaction by an operator (Decision 4) | No |
| A repair invalidates preserved thinking | Provider-specific continuation (Decision 5): incompatible past thinking excluded from model input | For verified model and mode combinations |
@@ -312,35 +335,36 @@ every rejection is caused by content it can repair.
## Slices
-1. **Normalization.** Shared lowering function in `llm-runtime`, header probe,
+1. **Thinking continuation.** Anthropic adapter defaults, beta header,
+ dropped-block logging and counts, `"error"` in suites that do not exercise a
+ repair; verified fallback for `disabled` thinking before claiming recovery
+ support there. Tests: unchanged histories retain valid thinking;
+ existing-image normalization, omission, redaction, and compaction that
+ retains thinking continue with prefix enforcement enabled; later turns and
+ a restart continue; original history is unchanged. A credentialed live
+ suite exercises an enforcing model and mode explicitly, rather than relying
+ on the account age or the default model.
+2. **Normalization.** Shared lowering function in `llm-runtime`, header probe,
fixed cap, byte budget, deterministic encoding, worker-local cache, all three
adapters. Tests: oversized PNG and JPEG are downscaled within the cap;
compliant images pass through byte-identical; output is identical across
calls; a 32-image history with a 2166 × 2464 image lowers within the
many-image rule.
-2. **Request limits check.** Limits rows for every API kind, stepped omission,
- typed failure before the provider call. Tests: omission order, protected
- newest media, identical requests across retries, the cut point holds steady
- while under the limit.
-3. **Rejection and redaction.** `RequestRejected` through the I/O boundary,
+3. **Request limits check.** Limits rows for every API kind, stepped omission,
+ newest-first degradation of protected media. Tests: omission order;
+ protected media omitted last; an eight-image result over the body limit
+ keeps its newest images, and the next request is identical; identical
+ requests across retries; the cut point holds steady while under the limit.
+4. **Rejection and redaction.** `RequestRejected` through the I/O boundary,
turn, and run failure; the redaction command, event, and placeholders;
`session/context/redact`; CLI support; contract regeneration; replay vectors
for redaction, its rejections, and a redacted tool result lowering as its
placeholder with pairing intact on every adapter.
-4. **Thinking continuation.** Anthropic adapter defaults, beta header, and
- transformation diagnostics; verified fallback for unsupported modes before
- claiming recovery support. Tests: unchanged histories retain valid thinking;
- existing-image normalization, omission, redaction, and compaction that
- retains thinking continue with prefix enforcement enabled; later turns and
- a restart continue; original history is unchanged. A credentialed live
- suite exercises an enforcing model and mode explicitly, rather than relying
- on the account age or the default model.
-Slice 1 addresses the incident's per-image limit. Slice 3 supplies operator
-repair for unexplained rejections caused by redactable entries. Slice 4 must
-land with any slice that changes previously sent content on a model enforcing
-thinking binding; repair is complete only when the session can continue after
-the change.
+Slice 1 comes first because every later slice can change content the provider
+has already seen; repair is complete only when the session can continue after
+the change. Slice 2 addresses the incident's per-image limit. Slice 4 supplies
+operator repair for unexplained rejections caused by redactable entries.
## Non-goals
From 5b405f047208ec532e6fc56a010c140c2840be8c Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 17:00:00 +0200
Subject: [PATCH 07/28] repair doc
---
...-safe-media-and-context-entry-redaction.md | 198 ++++++++++--------
1 file changed, 115 insertions(+), 83 deletions(-)
diff --git a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
index dcd7f6522..5a87f589e 100644
--- a/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
+++ b/docs/roadmap/p186-provider-safe-media-and-context-entry-redaction.md
@@ -17,14 +17,14 @@ Five changes deliver this, with provider-specific lowering and continuation:
1. Every image is sent as a normalized copy for the model, bounded in pixels
and bytes. Stored originals are unchanged.
-2. Each adapter checks the whole lowered request against its provider's
- request limits before sending it, and omits the oldest media when it would
- not fit.
+2. Each request keeps its media within one fixed, provider-independent media
+ budget, omitting the oldest media when it would not fit.
3. A provider rejecting a request is reported as a distinct run failure,
`RequestRejected`, carrying the provider's message word for word.
4. `session/context/redact` replaces the content of chosen entries with a
fixed placeholder, in place, so an operator can neutralize the entry that
- causes a rejection.
+ causes a rejection. `session/context/read` lists active context so the
+ operator can find that entry.
5. A repair that invalidates preserved thinking uses the provider's supported
continuation policy, so incompatible past reasoning does not itself prevent
the session from continuing.
@@ -94,7 +94,9 @@ one entry it no longer accepts, and the runtime has no supported way out.
increasing `entry_id`, and a keyed upsert removes the old entry and appends
its replacement at the tail. External context commands (`UpsertContext`,
`ReplaceContextPrefix`, `RemoveContext`) address entries only by key.
- Run-appended entries have no key and are unreachable.
+ Run-appended entries have no key and are unreachable. No public method
+ reads active context: an operator can only reconstruct it by folding
+ `session/events/read`, and the CLI is a plain API client.
- Tool calls and tool results are separate entries (`ToolCall`, `ToolResult`),
one per call. Tool-produced media are further separate user-role entries
that follow the result.
@@ -107,8 +109,9 @@ one entry it no longer accepts, and the runtime has no supported way out.
### Provider limits
-The request limits check (Decision 2) keeps one row per provider API kind.
-Slice 3 is complete only when every API kind has its row.
+These limits set the values of the media budget (Decision 2), which sits
+below the limits of every supported API kind. They are not consulted at
+request time.
Anthropic Messages, first-party API, as of 2026-10-01:
@@ -121,8 +124,9 @@ Anthropic Messages, first-party API, as of 2026-10-01:
| Request body | 32 MB |
| Native resolution, high-resolution tier (Claude 4.7 and later) | 2576 px long edge, 4784 visual tokens; larger images are downscaled server-side |
-OpenAI Responses and Chat Completions rows are taken from current provider
-documentation during implementation.
+OpenAI Responses and Chat Completions limits are checked against current
+provider documentation during implementation, to confirm that the budget
+sits below them too.
## Decisions
@@ -156,6 +160,10 @@ durable state.
optimization. A worker-local LRU keyed by (source blob ref, spec version)
avoids repeated decodes; a cache miss on another worker yields the same
bytes. A persistent copy store is added only if measurement calls for it.
+- **A resized image states what the model sees.** Its announcement gains the
+ dimensions the model sees beside the original's, for example
+ `· shown at 2000×1400 of 4000×2800`, so coordinate-based work can scale
+ back to the source. The text is deterministic and does not affect caching.
- **GIF and animated images** lower their first frame, which matches what
providers read.
- **PDFs are not normalized.** They count toward request totals in Decision 2.
@@ -166,40 +174,44 @@ Normalizing an existing history can change image bytes the provider has
already seen. This migration uses the thinking-continuation policy in Decision
5, just as omission and redaction do.
-### 2. Check the whole request against provider limits
+### 2. Keep each request within one fixed media budget
-After lowering, each adapter compares the request with its row of the limits
-table: image block count, per-image encoded bytes, and total body bytes.
+After normalization, lowering counts the media in the request against one
+media budget: a maximum number of media items and a maximum of encoded media
+bytes. The budget is a constant, the same for every provider and model, and
+conservative enough to sit below every supported API kind's limits.
+- **One budget, not a limits table per provider.** A table per API kind would
+ have to track provider limits by hand, and a stale row would either omit
+ media needlessly or miss a real limit. It would also move the cut point
+ whenever a session switches model or API kind, invalidating the prompt cache
+ and preserved thinking, which is the instability Decision 1's fixed pixel
+ cap avoids. After normalization, every image already meets the per-image
+ limits, so the budget only has to bound the aggregate.
- **When it fits, nothing changes.**
- **When it does not fit, the oldest media is omitted.** Media entries are
replaced, oldest first, with the placeholder text the text-only path already
uses:
`[image · media:3f9a2c1d4e7b · omitted from this request to stay within provider limits]`.
- Media from the current run's input and from the latest tool batch is
- protected: it is omitted only after all older media, because the model must
- see what it just asked for.
-- **Protected media degrades newest-first.** Per-item admission does not bound
- the aggregate request: eight images at the 5 MB encoded budget already
- exceed a 32 MB body limit, even in a single result, and documents and
- parallel tool batches can exceed it too. When the protected media alone
- does not fit, its newest items are kept and the rest receive the same
- placeholder. Failing the turn instead would not help: the latest tool batch
- stays the latest after the run fails, so every later request would fail the
- same way.
-- **The cut point moves in steps.** Which media is omitted is a pure function
- of the context, so retries produce identical requests. The boundary moves
- only when the request crosses the limit, and then it drops to a low-water
- mark well below the limit, so a growing session invalidates its prompt
- cache rarely rather than every turn.
-- **The check never fails a request.** It manages media only, and a single
+ Order is by recency alone. The latest tool batch is the newest media, so it
+ is omitted last without any special protection. If it alone exceeds the
+ budget, its newest items are kept: eight images at the 5 MB encoded
+ per-image budget already exceed a 32 MB body limit. Failing the turn instead
+ would not help, because the latest batch stays the latest after the run
+ fails and every later request would fail the same way.
+- **The cut point moves in fixed chunks.** The number of omitted items is
+ rounded up to a fixed chunk size. The cut point is therefore a pure function
+ of the context, with no stored state: retries produce identical requests,
+ and the boundary moves only once per chunk of new media, so a growing
+ session invalidates its prompt cache rarely rather than every turn.
+- **The budget never fails a request.** It manages media only, and a single
image always fits within the per-image budget. A request whose non-media
- content alone exceeds the body limit is a context-size problem for
- compaction, or a rejection under Decision 3.
+ content alone exceeds the provider's body limit is a context-size problem
+ for compaction, or a rejection under Decision 3.
Omission rewrites earlier user content as the provider sees it, in sessions
that are still healthy. Decision 1 keeps it rare, because pixel limits no
-longer trigger it; only request totals do.
+longer trigger it; only aggregate media does.
### 3. Report provider rejections as `RequestRejected`
@@ -213,6 +225,15 @@ provider error stays `ModelFailure`.
engine stays deterministic: it records the classification it is given and
never inspects provider errors.
- The failure record keeps the provider's message word for word.
+- A provider-reported `ContextLength` is a rejection like any other: it does
+ not trigger compaction. Compaction keeps its own token-budget policy, and
+ the operator can run `session/context/compact` after seeing the failure.
+- On a rejection, the adapter logs, next to the provider's message, the
+ position each entry ID was lowered to in the request (for example message
+ and content index). Provider messages cite request positions, not entry
+ IDs, and only the adapter knows how entries were merged into provider
+ messages. The mapping is a log line, not durable state: its shape is
+ provider-specific.
- Clients show that the provider rejected the request, with that message.
They do not suggest a fix or guess which entry caused it: the runtime cannot
know, and provider messages differ in how precisely they point at a cause.
@@ -263,6 +284,14 @@ the child's active context, so a redacted image no longer travels with it.
Redaction is available through the API and the runtime CLI. There is no web
affordance and no automatic redaction.
+**Finding the entry.** `session/context/read { sessionId }` returns the
+active context revision and its entries in context order, as the existing
+`ContextEntryView` (entry ID, key, kind, content reference, preview, token
+estimate), with media dimensions and byte size added. It is read-only, has
+viewer access, and gives the operator the IDs that redaction needs. The CLI lists it as a table. Together with the adapter's
+position log from Decision 3, an operator can go from a provider message that
+cites a request position to the entry ID to redact.
+
### 5. Continue after a repair invalidates preserved thinking
Thinking compatibility is part of recovery for normalization of existing
@@ -294,26 +323,26 @@ not make every provider rejection recoverable.
would otherwise surface as rejections. Two measures keep such bugs visible:
- Production logs every reported `input_transformations` entry with its path
- and reason, and exports a count of dropped blocks per session, so unexpected
- drops are observable.
+ and reason, so unexpected drops are observable. A per-session metric is
+ added only if the logs prove insufficient.
- Live and CI suites that do not exercise a repair send `"error"`, so an
unintended history edit fails a test instead of being absorbed.
-Models or modes that cannot use this policy need a separately tested fallback
-that durably excludes affected historical thinking from future requests,
-without gaps or later reintroduction. The exclusion must survive restart and
-preserve a valid tool sequence. Recovery support for such a mode is not
-complete until that path is verified. Anthropic accepts `block_binding` only
-with `adaptive` and `enabled` thinking, so this applies today to `disabled`,
-which the adapter sends for reasoning effort `none`, and would apply to
-`between_tools` if the adapter adopts it. Other adapters follow their native
-contracts; Anthropic's policy does not authorize stripping OpenAI reasoning
-items or other provider-opaque content.
+Anthropic accepts `block_binding` only with `adaptive` and `enabled` thinking.
+The adapter sends `disabled` for reasoning effort `none`, so the behavior of
+mismatched historical thinking under `disabled` is verified against the live
+API before recovery is claimed for that mode. If the provider ignores or
+drops historical thinking there, nothing more is needed. Only if it rejects
+the request does that mode need a fallback that durably excludes the affected
+thinking from future requests; that fallback is designed then, not in
+advance. Other adapters follow their native contracts; Anthropic's policy
+does not authorize stripping OpenAI reasoning items or other provider-opaque
+content.
Provider wire settings, their interpretation, and dropped-block diagnostics
remain in the adapter. This decision does not change the engine or the
public contract, and nothing inspects signatures. Stable normalization and
-stepped omission still matter: fewer history edits preserve more reasoning
+chunked omission still matter: fewer history edits preserve more reasoning
and more of the prompt cache.
## How a failing session recovers
@@ -322,7 +351,7 @@ and more of the prompt cache.
| --- | --- | --- |
| Context exceeds its token budget | Existing compaction policy: older context summarized | According to session policy |
| Image over a provider's pixel or per-image byte limit | Normalization (Decision 1): a bounded copy is sent | Yes |
-| Request over a provider's image count or total size | Request limits check (Decision 2): oldest media omitted, protected media last and newest-first | Yes |
+| Request over the media count or byte budget | Media budget (Decision 2): oldest media omitted in fixed chunks | Yes |
| Unexplained rejection caused by a redactable entry | `RequestRejected` (Decision 3), then redaction by an operator (Decision 4) | No |
| A repair invalidates preserved thinking | Provider-specific continuation (Decision 5): incompatible past thinking excluded from model input | For verified model and mode combinations |
@@ -335,42 +364,52 @@ every rejection is caused by content it can repair.
## Slices
-1. **Thinking continuation.** Anthropic adapter defaults, beta header,
- dropped-block logging and counts, `"error"` in suites that do not exercise a
- repair; verified fallback for `disabled` thinking before claiming recovery
- support there. Tests: unchanged histories retain valid thinking;
- existing-image normalization, omission, redaction, and compaction that
- retains thinking continue with prefix enforcement enabled; later turns and
- a restart continue; original history is unchanged. A credentialed live
- suite exercises an enforcing model and mode explicitly, rather than relying
- on the account age or the default model.
-2. **Normalization.** Shared lowering function in `llm-runtime`, header probe,
- fixed cap, byte budget, deterministic encoding, worker-local cache, all three
- adapters. Tests: oversized PNG and JPEG are downscaled within the cap;
- compliant images pass through byte-identical; output is identical across
- calls; a 32-image history with a 2166 × 2464 image lowers within the
- many-image rule.
-3. **Request limits check.** Limits rows for every API kind, stepped omission,
- newest-first degradation of protected media. Tests: omission order;
- protected media omitted last; an eight-image result over the body limit
- keeps its newest images, and the next request is identical; identical
- requests across retries; the cut point holds steady while under the limit.
-4. **Rejection and redaction.** `RequestRejected` through the I/O boundary,
- turn, and run failure; the redaction command, event, and placeholders;
- `session/context/redact`; CLI support; contract regeneration; replay vectors
- for redaction, its rejections, and a redacted tool result lowering as its
- placeholder with pairing intact on every adapter.
-
-Slice 1 comes first because every later slice can change content the provider
-has already seen; repair is complete only when the session can continue after
-the change. Slice 2 addresses the incident's per-image limit. Slice 4 supplies
+1. **Normalization and thinking continuation.** Shared lowering function in
+ `llm-runtime`, header probe, fixed cap, byte budget, deterministic
+ encoding, resize announcement, worker-local cache, all three adapters.
+ Anthropic `drop_block` default, beta header, `input_transformations`
+ logging, `"error"` in suites that do not exercise a repair, and the live
+ check of `disabled` thinking. Tests: oversized PNG and JPEG are downscaled
+ within the cap; compliant images pass through byte-identical; output is
+ identical across calls; a 32-image history with a 2166 × 2464 image lowers
+ within the many-image rule; unchanged histories retain valid thinking; an
+ existing history whose image is newly normalized continues with prefix
+ enforcement enabled, across later turns and a restart, with original
+ history unchanged; compaction that retains thinking continues likewise. A
+ credentialed live suite exercises an enforcing model and mode explicitly,
+ rather than relying on the account age or the default model.
+2. **Rejection.** `RequestRejected` through the I/O boundary, turn, and run
+ failure; the adapter's position log; contract regeneration. Tests: an
+ `InvalidRequest` and a `ContextLength` error fail the run as
+ `RequestRejected` with the provider message intact; other terminal errors
+ stay `ModelFailure`.
+3. **Redaction.** The redaction command, event, and placeholders;
+ `session/context/redact` and `session/context/read`; CLI support; contract
+ regeneration; replay vectors for redaction, its rejections, and a redacted
+ tool result lowering as its placeholder with pairing intact on every
+ adapter. Tests: a redacted history continues with prefix enforcement
+ enabled.
+4. **Media budget.** The budget constants, oldest-first omission rounded to a
+ fixed chunk, in all three adapters and the compaction request path. Tests:
+ omission order; an eight-image result over the budget keeps its newest
+ images, and the next request is identical; identical requests across
+ retries; the cut point holds steady until a chunk of new media arrives; a
+ history with omitted media continues with prefix enforcement enabled.
+
+Slice 1 closes the incident: normalization removes the per-image failure, and
+`drop_block` lets existing sessions continue once normalization changes image
+bytes the provider has already seen. Every later slice can also change such
+content, so each verifies continuation for its own repair. Slice 3 supplies
operator repair for unexplained rejections caused by redactable entries.
+Slice 4 comes last because, after normalization, only aggregate media can
+trigger it.
## Non-goals
- Changing compaction triggers or summarization strategy. Compaction inherits
- the request safeguards and applicable thinking-continuation policy.
+ the media safeguards and applicable thinking-continuation policy.
- Introducing a generic context-repair framework.
+- A limits table per provider API kind consulted at request time.
- Suggesting fixes in clients, or redacting automatically after a rejection.
- Retrying after a rejection by matching provider error text.
- Removing tool calls or call/result pairs, and redacting tool calls.
@@ -378,10 +417,3 @@ operator repair for unexplained rejections caused by redactable entries.
- Storing copies for a specific provider in session state, or rewriting
existing events.
- The Anthropic Files API as an alternative to base64 payloads.
-
-## Open questions
-
-- **Announcing the resize.** Whether a normalized image's announcement should
- state the dimensions the model sees (for example, `· shown at 2000×1400 of
- 4000×2800`) to support coordinate-based work. It is deterministic, so it
- does not affect caching.
From b76102be159bc5c4de35009b8489248db77d5a0d Mon Sep 17 00:00:00 2001
From: lb <542828+lukebuehler@users.noreply.github.com>
Date: Thu, 1 Oct 2026 17:00:07 +0200
Subject: [PATCH 08/28] voice composer
---
.../src/components/session/composer-voice.tsx | 4 +-
.../session/composer.dictation.test.tsx | 140 ++++++++++++++++--
.../web/src/components/session/composer.tsx | 56 ++++---
3 files changed, 165 insertions(+), 35 deletions(-)
diff --git a/platform/web/src/components/session/composer-voice.tsx b/platform/web/src/components/session/composer-voice.tsx
index 88591ffdf..36ac3ee41 100644
--- a/platform/web/src/components/session/composer-voice.tsx
+++ b/platform/web/src/components/session/composer-voice.tsx
@@ -45,7 +45,7 @@ export function VoiceControl({ voice, unavailableReason, settingsHref, demo, onS
- void voice.stop()}>
+ void voice.stop()}>
@@ -93,7 +93,7 @@ export function VoiceControl({ voice, unavailableReason, settingsHref, demo, onS
}
return (
diff --git a/platform/web/src/components/session/composer.dictation.test.tsx b/platform/web/src/components/session/composer.dictation.test.tsx
index daaf351b2..813a99361 100644
--- a/platform/web/src/components/session/composer.dictation.test.tsx
+++ b/platform/web/src/components/session/composer.dictation.test.tsx
@@ -1,16 +1,19 @@
// @vitest-environment jsdom
-import { act } from "react";
+import { act, type ComponentProps } from "react";
import { createRoot, type Root } from "react-dom/client";
import { afterEach, beforeEach, expect, it, vi } from "vitest";
import { SessionComposer } from "./composer";
-const mocks = vi.hoisted(() => ({ capture: vi.fn(), transcribe: vi.fn(), cancel: vi.fn() }));
+const mocks = vi.hoisted(() => ({ capture: vi.fn(), transcribe: vi.fn(), cancel: vi.fn(), api: vi.fn() }));
+vi.mock("@/api", async original => ({ ...await original(), api: mocks.api }));
vi.mock("@/lib/audio-capture", () => ({ startAudioCapture: mocks.capture, isDemoDictation: false }));
vi.mock("@/lib/dictation", () => ({ transcribeRecording: mocks.transcribe, cancelRecording: mocks.cancel }));
vi.mock("@/components/ui/popover", () => import("@/components/ui/popover.test-double"));
let root: Root;
let container: HTMLDivElement;
let finish: (text: string) => void;
+let failTranscription: (error: Error) => void;
+let finishUpload: (result: unknown) => void;
let stop: ReturnType;
let onSend: ReturnType;
beforeEach(() => {
@@ -18,7 +21,8 @@ beforeEach(() => {
vi.clearAllMocks();
stop = vi.fn();
mocks.capture.mockResolvedValue({ stop, name: "dictation.webm", result: Promise.resolve(new Blob(["audio"], { type: "audio/webm" })) });
- mocks.transcribe.mockImplementation(() => new Promise((resolve) => { finish = resolve; }));
+ mocks.transcribe.mockImplementation(() => new Promise((resolve, reject) => { finish = resolve; failTranscription = reject; }));
+ mocks.api.mockImplementation(() => new Promise(resolve => { finishUpload = resolve; }));
onSend = vi.fn();
container = document.createElement("div");
document.body.append(container);
@@ -30,9 +34,9 @@ afterEach(async () => {
localStorage.clear();
vi.unstubAllGlobals();
});
-async function show(disabledReason?: string, disabled = false) {
+async function show(disabledReason?: string, disabled = false, props: Partial> = {}) {
await act(async () => root.render());
+ dictation={{ universeId: "universe", disabledReason }} disabled={disabled} error={null} onSend={onSend} onStop={vi.fn()} {...props} />));
}
async function click(label: string) {
const button = [...container.querySelectorAll("button")].find((node) => node.getAttribute("aria-label") === label || node.textContent === label)!;
@@ -46,6 +50,9 @@ async function type(text: string) {
});
}
async function record() { await click("Dictate message"); await click("Stop recording"); }
+async function press(key: string, init: KeyboardEventInit = {}) {
+ await act(async () => container.querySelector("textarea")!.dispatchEvent(new KeyboardEvent("keydown", { key, bubbles: true, ...init })));
+}
it("appends to the latest edited draft and waits for an explicit send", async () => {
await show();
await type("Before");
@@ -74,17 +81,19 @@ it("inserts at the caret the field last had, when the text is unchanged", async
expect(input.value).toBe("Hello big world");
expect(input.selectionStart).toBe(9);
});
-it("stops and transcribes on Enter while recording instead of sending", async () => {
+it.each(["Send message", "Enter"])("stops recording on %s and sends once the transcript is ready", async action => {
await show();
- await type("Draft");
await click("Dictate message");
expect(container.querySelector('[aria-label^="Recording"]')).not.toBeNull();
const input = container.querySelector("textarea")!;
- await act(async () => input.dispatchEvent(new KeyboardEvent("keydown", { key: "Enter", bubbles: true })));
+ if (action === "Enter") await press("Enter");
+ else await click(action);
expect(stop).toHaveBeenCalled();
expect(onSend).not.toHaveBeenCalled();
await act(async () => finish("Spoken."));
- expect(input.value).toBe("Draft Spoken.");
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Spoken.", attachments: [] }, null);
+ expect(input.value).toBe("");
+ expect(localStorage.getItem("voice-test")).toBeNull();
});
it("discards the recording on Escape", async () => {
await show();
@@ -95,24 +104,33 @@ it("discards the recording on Escape", async () => {
expect(container.querySelector('[aria-label="Dictate message"]')).not.toBeNull();
expect(mocks.transcribe).not.toHaveBeenCalled();
});
-it.each(["Cancel dictation", "Send message"])("ignores late completion after %s", async (action) => {
+it.each(["Cancel dictation", "Escape", "pagehide"])("cancels a pending send and ignores late completion after %s", async action => {
await show();
await type("Keep this");
await record();
- await click(action);
+ await click("Send message");
+ if (action === "Escape") await press("Escape");
+ else if (action === "pagehide") await act(async () => window.dispatchEvent(new Event("pagehide")));
+ else await click(action);
expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true);
await act(async () => finish("Late text"));
- expect(container.querySelector("textarea")!.value).toBe(action === "Send message" ? "" : "Keep this");
- expect(onSend).toHaveBeenCalledTimes(action === "Send message" ? 1 : 0);
+ expect(container.querySelector("textarea")!.value).toBe("Keep this");
+ expect(onSend).not.toHaveBeenCalled();
+ await record();
+ await act(async () => finish("New recording"));
+ expect(container.querySelector("textarea")!.value).toBe("Keep this New recording");
+ expect(onSend).not.toHaveBeenCalled();
});
it("cancels on navigation without modifying the saved draft", async () => {
await show();
await type("Saved draft");
await record();
+ await click("Send message");
await act(async () => root.render(null));
await act(async () => finish("Late text"));
expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(true);
expect(localStorage.getItem("voice-test")).toBe("Saved draft");
+ expect(onSend).not.toHaveBeenCalled();
});
it("explains a missing speech default instead of recording", async () => {
await show("Set a speech-to-text default in Models to enable dictation.");
@@ -149,3 +167,99 @@ it("explains permission denial without losing text", async () => {
expect(container.querySelector("textarea")!.value).toBe("Draft");
expect(mocks.transcribe).not.toHaveBeenCalled();
});
+
+it.each(["", "Existing draft"])("waits for transcription with draft %j and sends the latest text exactly once", async draft => {
+ await show();
+ await type(draft);
+ await record();
+ expect(container.querySelector('[aria-label="Send message"]')!.disabled).toBe(false);
+ await click("Send message");
+ await click("Send message");
+ expect(onSend).not.toHaveBeenCalled();
+ expect(mocks.transcribe.mock.calls[0]![2].aborted).toBe(false);
+ expect(container.textContent).toContain("Your message will send when ready");
+ await type("Latest edit");
+ await act(async () => finish("Spoken text"));
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Latest edit Spoken text", attachments: [] }, null);
+ expect(container.querySelector("textarea")!.value).toBe("");
+});
+
+it.each([false, true])("preserves queue and keyboard steering after transcription (steer: %s)", async steer => {
+ await show(undefined, false, { runActive: true, canSteer: true });
+ await record();
+ await press("Enter", { ctrlKey: steer });
+ expect(onSend).not.toHaveBeenCalled();
+ await act(async () => finish("Spoken text"));
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Spoken text", attachments: [] }, steer ? "steer" : "queue");
+});
+
+it("keeps the draft after a transcription error and requires a new send after retry", async () => {
+ await show();
+ await type("Keep this");
+ await record();
+ await click("Send message");
+ await act(async () => failTranscription(new Error("Service unavailable")));
+ expect(container.textContent).toContain("Service unavailable");
+ expect(onSend).not.toHaveBeenCalled();
+ expect(container.querySelector("textarea")!.value).toBe("Keep this");
+ await click("Retry transcription");
+ await act(async () => finish("Recovered text"));
+ expect(container.querySelector("textarea")!.value).toBe("Keep this Recovered text");
+ expect(onSend).not.toHaveBeenCalled();
+ await click("Send message");
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Keep this Recovered text", attachments: [] }, null);
+});
+
+it("clears a pending send when the composer becomes disabled", async () => {
+ await show();
+ await type("Keep this");
+ await record();
+ await click("Send message");
+ await show(undefined, true);
+ await act(async () => finish("Late transcript"));
+ expect(onSend).not.toHaveBeenCalled();
+ expect(container.querySelector("textarea")!.value).toBe("Keep this");
+ await show();
+ await record();
+ await act(async () => finish("New transcript"));
+ expect(onSend).not.toHaveBeenCalled();
+});
+
+it.each(["transcript", "upload"])("waits for both the transcript and attachments when %s finishes first", async first => {
+ await show(undefined, false, { attachments: { universeId: "universe", apiKind: "anthropic:messages" } });
+ const input = container.querySelector('input[type="file"]')!;
+ Object.defineProperty(input, "files", { value: [new File(["pdf"], "notes.pdf", { type: "application/pdf" })] });
+ await act(async () => input.dispatchEvent(new Event("change", { bubbles: true })));
+ await act(async () => { await new Promise(resolve => setTimeout(resolve, 20)); });
+ expect(mocks.api).toHaveBeenCalled();
+ await record();
+ await click("Send message");
+ const completeTranscript = async () => { await act(async () => finish("Spoken text")); };
+ const completeUpload = async () => { await act(async () => finishUpload({ blobs: [{ blobRef: `sha256:${"a".repeat(64)}`, bytes: 3 }] })); };
+ if (first === "transcript") await completeTranscript();
+ else await completeUpload();
+ expect(onSend).not.toHaveBeenCalled();
+ if (first === "transcript") await completeUpload();
+ else await completeTranscript();
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Spoken text", attachments: [expect.objectContaining({ name: "notes.pdf" })] }, null);
+});
+
+it("keeps the completed transcript for review when an attachment upload fails", async () => {
+ mocks.api.mockRejectedValueOnce(new Error("Upload unavailable"));
+ await show(undefined, false, { attachments: { universeId: "universe", apiKind: "anthropic:messages" } });
+ const input = container.querySelector('input[type="file"]')!;
+ Object.defineProperty(input, "files", { value: [new File(["pdf"], "notes.pdf", { type: "application/pdf" })] });
+ await act(async () => input.dispatchEvent(new Event("change", { bubbles: true })));
+ await act(async () => { await new Promise(resolve => setTimeout(resolve, 20)); });
+ expect(container.textContent).toContain("Upload failed");
+ await record();
+ await click("Send message");
+ await act(async () => finish("Spoken text"));
+ expect(onSend).not.toHaveBeenCalled();
+ expect(container.querySelector("textarea")!.value).toBe("Spoken text");
+ expect(container.textContent).toContain("Remove or retry the attachments");
+ await click("Remove notes.pdf");
+ expect(onSend).not.toHaveBeenCalled();
+ await click("Send message");
+ expect(onSend).toHaveBeenCalledExactlyOnceWith({ text: "Spoken text", attachments: [] }, null);
+});
diff --git a/platform/web/src/components/session/composer.tsx b/platform/web/src/components/session/composer.tsx
index 45866b518..6bef8e6b7 100644
--- a/platform/web/src/components/session/composer.tsx
+++ b/platform/web/src/components/session/composer.tsx
@@ -112,7 +112,7 @@ export function SessionComposer({
const [flash, setFlash] = useState(false);
const [dragging, setDragging] = useState(false);
const dragDepth = useRef(0);
- const [pendingSubmit, setPendingSubmit] = useState<"send" | "steer" | null>(null);
+ const [pendingSubmit, setPendingSubmit] = useState<{ mode: "send" | "steer"; waitingForTranscript: boolean } | null>(null);
const updateText = (value: string) => {
textRef.current = value;
@@ -151,6 +151,7 @@ export function SessionComposer({
caret = next.length;
}
updateText(next);
+ setPendingSubmit(pending => pending?.waitingForTranscript ? { ...pending, waitingForTranscript: false } : pending);
selection.current = null;
caretAfterInsert.current = caret;
setAnnouncement("Dictation added. Review and edit before sending.");
@@ -172,26 +173,34 @@ export function SessionComposer({
}, [flash]);
const hasContent = text.trim().length > 0 || files.items.length > 0;
+ const voiceCanSubmit = voice.phase === "recording" || voice.phase === "transcribing";
+
+ const cancelVoice = () => {
+ setPendingSubmit(null);
+ voice.cancel();
+ };
const submit = (steer: boolean) => {
if (disabled) return;
- // Enter while recording finishes the recording; the transcript still
- // needs a review before anything is sent.
- if (voice.phase === "recording") {
- void voice.stop();
+ // An explicit send includes the transcript, even before it is ready.
+ if (voiceCanSubmit) {
+ setPendingSubmit({ mode: steer ? "steer" : "send", waitingForTranscript: true });
+ if (voice.phase === "recording") void voice.stop();
return;
}
if (!hasContent) return;
if (files.failed) {
+ setPendingSubmit(null);
setNotice("Remove or retry the attachments that failed to upload.");
return;
}
if (files.uploading) {
- setPendingSubmit(steer ? "steer" : "send");
+ setPendingSubmit({ mode: steer ? "steer" : "send", waitingForTranscript: false });
return;
}
const mode: ComposerMode | null = !runActive ? null : steer ? "steer" : "queue";
if (mode === "steer" && !canSteer) {
+ setPendingSubmit(null);
setNotice(`There is no run to steer right now. Press Enter to queue the message instead.`);
return;
}
@@ -207,21 +216,27 @@ export function SessionComposer({
updateText("");
};
- // A send asked for while uploads were running goes out once they finish.
+ // Wait for the requested transcript and uploads; cancellation or failure
+ // must never send a partial draft.
useEffect(() => {
- if (!pendingSubmit || files.uploading) return;
+ if (!pendingSubmit) return;
+ if (disabled || voice.phase === "error" || (pendingSubmit.waitingForTranscript && voice.phase === "idle")) {
+ setPendingSubmit(null);
+ return;
+ }
+ if (pendingSubmit.waitingForTranscript || files.uploading) return;
if (!hasContent) {
setPendingSubmit(null);
return;
}
- submit(pendingSubmit === "steer");
+ submit(pendingSubmit.mode === "steer");
// eslint-disable-next-line react-hooks/exhaustive-deps
- }, [pendingSubmit, files.uploading]);
+ }, [pendingSubmit, files.uploading, files.failed, voice.phase, disabled]);
const onKeyDown = (event: KeyboardEvent) => {
if (event.key === "Escape" && VOICE_BUSY.has(voice.phase)) {
event.preventDefault();
- voice.cancel();
+ cancelVoice();
return;
}
if (event.key !== "Enter" || event.shiftKey || event.nativeEvent.isComposing) {
@@ -232,6 +247,7 @@ export function SessionComposer({
};
const startVoice = () => {
+ setPendingSubmit(null);
setAnnouncement(undefined);
void voice.start();
};
@@ -274,8 +290,8 @@ export function SessionComposer({
: "Enter queues a follow-up…"
: "Message the agent…";
const sendLabel = runActive ? "Queue message" : "Send message";
- const sendTitle = voice.phase === "recording"
- ? "Enter stops the recording; review the transcript before sending"
+ const sendTitle = voiceCanSubmit
+ ? "Send when transcription finishes"
: pendingSubmit
? "Sends when the uploads finish"
: runActive
@@ -371,7 +387,7 @@ export function SessionComposer({
)}
{dictation && !disabled && (
-
)}
{runActive && canStop && (
@@ -384,7 +400,7 @@ export function SessionComposer({
{!disabled && (
)}
- {voice.phase === "recording" ? "Recording. Enter stops and transcribes; Escape discards."
+ {voice.phase === "recording" ? "Recording. Enter sends after transcription; Stop inserts the text; Escape discards."
: voice.phase === "requesting" ? "Waiting for microphone permission…"
- : voice.phase === "transcribing" ? "Transcribing… You can keep editing."
+ : voice.phase === "transcribing" ? pendingSubmit?.waitingForTranscript ? "Transcribing… Your message will send when ready." : "Transcribing… You can keep editing."
: announcement ?? ""}