From c540421f180a98e602a9a98f0b25700eae5fb571 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 17:03:17 +0000 Subject: [PATCH 01/43] wiki: log merges of #115 #127 #137 #139 #141 #142 #143 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/log.md | 1 + 1 file changed, 1 insertion(+) diff --git a/wiki/log.md b/wiki/log.md index 07bd431..25b9d67 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -24,3 +24,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-09-25 decide | #113 Addendum 4: content creation is a fourth goal; #114 item 11 parks media toolchains | decisions/0007-content-as-fourth-goal.md, entities/roadmap-issues.md, SCHEMA.md, index.md - 2026-09-25 decide | #113 Addendum 5: idea-agnostic harness wins on user control; 0007 superseded by 0008; verticals are dogfood | decisions/0008-idea-agnostic-extensible-harness.md, decisions/0007-content-as-fourth-goal.md, entities/roadmap-issues.md, SCHEMA.md, index.md - 2026-09-25 update | #134 P1: event envelope + SessionManager landed in 0.10.0; serve per-session queues; Ossuary parse accepts envelope | concepts/event-envelope.md, entities/lich-serve.md, index.md +- 2026-10-02 update | Merged #143 #141 #139 #137 #127 #115 (test coverage) and #142 (gateway history cap keeps tool-heavy windows, stub user turn when none); no wiki page covers the cap yet | log.md From 5a6efcdd6a640433d87f9bd20fc36a4388d6663f Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 20:38:09 +0000 Subject: [PATCH 02/43] wiki: decision models and Ollama model research (#148) Ingest decision-model and Ollama research, add the decision-models concept page, note ollama.com cloud auth and stale defaults on the providers page, and link the fast lane from action-terminal mode. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/concepts/action-terminal-mode.md | 4 +- wiki/concepts/decision-models.md | 42 +++++++++++ wiki/entities/lich-providers.md | 10 ++- wiki/index.md | 5 +- wiki/log.md | 3 + ...6-10-02-decision-models-ollama-research.md | 71 +++++++++++++++++++ 6 files changed, 130 insertions(+), 5 deletions(-) create mode 100644 wiki/concepts/decision-models.md create mode 100644 wiki/raw/audits/2026-10-02-decision-models-ollama-research.md diff --git a/wiki/concepts/action-terminal-mode.md b/wiki/concepts/action-terminal-mode.md index b6c9533..e7e5ec0 100644 --- a/wiki/concepts/action-terminal-mode.md +++ b/wiki/concepts/action-terminal-mode.md @@ -1,7 +1,7 @@ --- title: Action-terminal mode (one LLM call per decision) created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-02 type: concept tags: [games, npc, core, performance] sources: [raw/audits/2026-09-23-game-surface-audit.md, "#113"] @@ -28,4 +28,6 @@ status_note: proposed, not implemented | Anthropic | `tool_choice: {"type":"tool","name":…}` or `{"type":"any"}` | | Ollama | `format: ` (the tool-choice support depends on the model) | +**Alternative fast lane (#148):** for a pure "pick one action" step, a [[decision-models]] `choice` call (tens of milliseconds, with a confidence value) can replace the LLM call entirely, falling back to the LLM or the scripted action when confidence is low. + Related: [[client-executed-tools]], [[lich-providers]], [[game-bridge-example]]. diff --git a/wiki/concepts/decision-models.md b/wiki/concepts/decision-models.md new file mode 100644 index 0000000..478a913 --- /dev/null +++ b/wiki/concepts/decision-models.md @@ -0,0 +1,42 @@ +--- +title: Decision models (typed, calibrated decisions) +created: 2026-10-02 +updated: 2026-10-02 +type: concept +tags: [providers, performance, research, games, security] +sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, "#148"] +confidence: medium +--- + +# Decision models + +**What they are.** A decision model does not chat. It takes state plus typed questions and returns a typed answer with a probability for every option, in one forward pass. The API shape comes from TypeSafe's Jev (`POST /v1/systemone`). Cloudflare's Clef / Clef-flash (Apache 2.0) and Ollama 0.35's local endpoint speak the same API. ^[raw/audits/2026-10-02-decision-models-ollama-research.md] + +| Question type | Returns | Typical use | +|---|---|---| +| `noul` | probability of yes | gates, filters, guards | +| `choice` | one labelled option, per-option probabilities, confidence | routing, action pick, intent | +| `score` | position on an ordered rubric, plus distribution | urgency, risk, quality | + +**Why Lich cares.** The [[tao-loop]] spends a full LLM call on every narrow decision. Decision models answer those in roughly 40-500 ms (Clef-flash fastest), locally or hosted, with a confidence value that tells the caller when to fall back to the LLM. That is the same latency problem [[action-terminal-mode]] attacks for game NPCs. + +## Where it could fit + +1. **Game action selection**, the first prototype in #148: a `choice` over the legal actions in [[game-bridge-example]], with a scripted or LLM fallback below a confidence threshold. +2. **Gateway triage** in [[lich-gateway]]: "should the agent answer this message?" before a run starts. +3. **Persona routing** in [[persona-orchestrator-example]], from message content. +4. **Tool-list narrowing** before an LLM turn, which cuts prompt tokens. +5. **A second-opinion guard** in a `before_tool_call` hook ([[lich-plugins-and-hooks]]). + +## Rules of use + +- **Optional plugin, never core.** This follows [[0008-idea-agnostic-extensible-harness]]: the loop stays model-agnostic and the decision client is an extension. +- **Always a fallback.** Low confidence, a timeout, or a compound or vague question goes to the LLM or a scripted action. +- **Add vetoes, never remove them.** Published work shows prompt injection shifts the probabilities and can move the answer. A decision-model verdict may block an action, but deterministic checks (URL guard, gatekeeper, MCP refusals) always still run. See [[embedded-safety-profile]]. + +## Getting the models + +- **Ollama (local or ollama.com):** 0.35 adds `/v1/systemone` with `nimble` (9B) and `tev1` (4B, 0.8B). Lich's [[lich-providers]] Ollama client only speaks `/api/chat` (`src/providers/ollama.ts:155-158@e9bdd82`), so a separate small client is needed. +- **Cloudflare Workers AI:** `clef` (27B) and `clef-flash` (9B). Clef is not in the Ollama library; Ollaya runs it locally. + +Numbers above come from search summaries, not the primary pages (blocked during research); re-verify before relying on them. diff --git a/wiki/entities/lich-providers.md b/wiki/entities/lich-providers.md index cd7f066..f498996 100644 --- a/wiki/entities/lich-providers.md +++ b/wiki/entities/lich-providers.md @@ -1,10 +1,10 @@ --- title: Lich providers and failover created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-02 type: entity tags: [providers, runtime, performance] -sources: [raw/audits/2026-09-23-core-engine-audit.md] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-decision-models-ollama-research.md, "#148"] confidence: high --- @@ -20,6 +20,12 @@ confidence: high - Consecutive tool results are merged into one Anthropic user turn (`anthropic.ts:203-233@77bc148`). - v0.9.0 adds `provider_content` so Anthropic thinking blocks round-trip. +## Ollama: local and ollama.com cloud + +- The Ollama client defaults to `http://localhost:11434` and needs no key. When `api_key` or `api_key_env` resolves, it sends `Authorization: Bearer` (`src/providers/ollama.ts:137-165@e9bdd82`). That is enough for Ollama's hosted models: `LICH_BASE_URL=https://ollama.com` plus `LICH_API_KEY_ENV=OLLAMA_API_KEY`. Not yet smoke-tested; tracked in #148. +- The README still calls the key "unused by ollama", and the documented default model is `llama3.2` (3B), which is weak at tool calling. #148 proposes `qwen3:8b`. +- Ollama 0.35's decision endpoint (`/v1/systemone`) is a different API from `/api/chat`; see [[decision-models]]. + ## Gaps (open on v0.9.0) - **No streaming.** Ollama sends `stream:false`; see [[streaming-deltas]]. diff --git a/wiki/index.md b/wiki/index.md index 8cd6332..c729581 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -1,12 +1,12 @@ --- title: Wiki index type: index -updated: 2026-09-25 +updated: 2026-10-02 --- # Lich Wiki: Index -Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) for recent activity. There are 39 pages. +Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) for recent activity. There are 40 pages. **New to the codebase?** Read [[tao-loop]], then [[lich-agent-loop]], [[lich-vs-hermes]], [[0008-idea-agnostic-extensible-harness]] and [[roadmap-issues]]. **Working on the game features?** Read [[game-transports]], then [[action-terminal-mode]], [[client-executed-tools]] and [[embedded-safety-profile]]. @@ -44,6 +44,7 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) - [[npc-memory-namespaces]]: private `npc:` memory and shared `world` memory, with identity carried in `ToolContext`. - [[streaming-deltas]]: `text_delta` events for dialogue, TTS and the chat pane. Lich has no streaming today. - [[embedded-safety-profile]]: the game-safe preset: no builtins, hooks that fail closed, untrusted player text, and budgets. +- [[decision-models]]: typed, calibrated decisions (Jev / Clef / Ollama `/v1/systemone`) as an optional fast lane beside the TAO loop; prototype in #148. - [[llm-wiki-pattern]]: Karpathy's compiled-knowledge wiki, which this wiki uses. It also doubles as a design for Lich's memory. ## Comparisons diff --git a/wiki/log.md b/wiki/log.md index 25b9d67..ccd3655 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -25,3 +25,6 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-09-25 decide | #113 Addendum 5: idea-agnostic harness wins on user control; 0007 superseded by 0008; verticals are dogfood | decisions/0008-idea-agnostic-extensible-harness.md, decisions/0007-content-as-fourth-goal.md, entities/roadmap-issues.md, SCHEMA.md, index.md - 2026-09-25 update | #134 P1: event envelope + SessionManager landed in 0.10.0; serve per-session queues; Ossuary parse accepts envelope | concepts/event-envelope.md, entities/lich-serve.md, index.md - 2026-10-02 update | Merged #143 #141 #139 #137 #127 #115 (test coverage) and #142 (gateway history cap keeps tool-heavy windows, stub user turn when none); no wiki page covers the cap yet | log.md +- 2026-10-02 ingest | Decision models (Jev, Clef) and Ollama model research from web-search summaries; primary pages blocked | raw/audits/2026-10-02-decision-models-ollama-research.md +- 2026-10-02 create | Decision models concept: typed calibrated decisions as an optional plugin fast lane with LLM fallback; filed #148 | concepts/decision-models.md, index.md +- 2026-10-02 update | Providers: Ollama Bearer auth already supports ollama.com cloud; README key note and llama3.2 default are stale (#148); action-terminal-mode links the decision-model fast lane | entities/lich-providers.md, concepts/action-terminal-mode.md diff --git a/wiki/raw/audits/2026-10-02-decision-models-ollama-research.md b/wiki/raw/audits/2026-10-02-decision-models-ollama-research.md new file mode 100644 index 0000000..0a334c7 --- /dev/null +++ b/wiki/raw/audits/2026-10-02-decision-models-ollama-research.md @@ -0,0 +1,71 @@ +--- +source_url: session:2026-10-02 web research for #148 (search summaries; primary pages blocked) +ingested: 2026-10-02 +sha256: f056c4df5b70f80bcff7aa6ad239502bc0b53d71d9b94549ce28378d78e1be78 +--- +# Decision models and Ollama models: research notes (2026-10-02) + +Session research for #148. The primary pages (blog.cloudflare.com, ollama.com, +marktechpost.com) were blocked by the session's network proxy, so these notes +come from web-search result summaries. Treat sizes, latencies and reliability +percentages as approximate and re-check them against the primary sources. + +## Decision models ("System One" models) + +- TypeSafe AI released Jev on 2026-09-15. It takes program state plus typed + questions and returns typed answers with calibrated probabilities in one + parallel pass. API: `POST /v1/systemone`. +- Question types: `noul` (yes/no probability), `choice` (one of up to 255 + labelled options, per-option probabilities plus a confidence value), `score` + (position on an ordered rubric plus the distribution). Up to 64 named + questions per request. +- Cloudflare released Clef (27B) and Clef-flash (9B) on Workers AI on + 2026-10-01, Jev-API compatible, weights Apache 2.0 on Hugging Face. + Clef-flash is built on Qwen3.5-9B with a "joint schema head" that scores + every option of every question in one forward pass. +- Reported median decision latency: Clef 209.3 ms, Clef-flash 38.8 ms, + Jev 524.1 ms. Clef leads the Jev Decision Index (132,422 requests across + 37 benchmarks, 30+ open-weight models). +- Cloudflare also launched an RL fine-tuning platform for decision models. +- Known limits: weak on large context, compound questions and vague criteria; + prompt injection shifts probabilities and can move the answer. Typed output + alone is not injection resistance; guards built on these models belong + alongside deterministic checks, not instead of them. +- Typical agent uses: tool or skill routing, safety gates, triage, ranking, + with "fall back to the LLM when confidence is low". + +## Ollama + +- Ollama 0.35 (late September 2026) added a local Jev-compatible + `/v1/systemone` endpoint. Launch models: Bespoke Labs `nimble` (9B, + fine-tuned from Qwen3.5-9B, ~9.5 GB Q8_0, ~91 ms per decision on an M5 Max) + and Together AI `tev1` (4B ~4.5 GB, 0.8B ~812 MB, experimental). TypeSafe's + Python SDK works unchanged against `http://localhost:11434`. +- Clef is not in the Ollama library; Ollaya (ollaya.dev) runs open decision + models locally behind a TypeSafe-compatible API. +- Ollama cloud: create an API key at ollama.com, call `https://ollama.com` with + `Authorization: Bearer `. About 94 library models carry the tools tag; + about 14 of them are cloud-only. +- Tool-calling picks reported for local agents: `qwen3:8b` (~5 GB at Q4_K_M, + ~85% tool-call reliability), `gemma4` (~90%), `llama3.1:8b` (~80%, fastest + first token), `qwen3:30b-a3b` for 16-24 GB machines, and + `llama3-groq-tool-use:8b` (top small fine-tune on BFCL). + +## Sources + +- https://blog.cloudflare.com/clef-decision-models/ +- https://developers.cloudflare.com/changelog/post/2026-10-01-clef-workers-ai/ +- https://developers.cloudflare.com/workers-ai/models/clef-flash/ +- https://flaviocopes.com/clef/ +- https://openrouter.ai/blog/insights/what-is-jev/ +- https://jevtypesafeai.com/docs +- https://dev.to/aitejiu/benchmarking-jev-what-a-decision-model-can-and-cant-do-in-an-agent-harness-20po +- https://arxiv.org/html/2609.28613v1 +- https://github.com/ollama/ollama/releases/tag/v0.35.0 +- https://ollama.com/library/nimble +- https://modelsystem.one/news/ollama-systemone-decision-models/ +- https://modelfit.io/blog/ollama-decision-models-nimble-tev1-mac/ +- https://github.com/ollaya-dev/ollaya +- https://docs.ollama.com/api/introduction +- https://localaimaster.com/blog/best-ollama-models-tool-calling +- https://localaimaster.com/blog/best-ollama-models-for-agents From 97992d736e494666d9da5b46100a55ac9f113092 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 21:07:32 +0000 Subject: [PATCH 03/43] wiki: Hermes model roles, memory embedders and Jev plugins (#148, #149) Ingest a read of Hermes upstream at bed0d535 and record how it mixes models (fallback chain plus per-task auxiliary models), where embeddings live (memory plugins), and how its 14 community Jev plugins work. Add comparison rows and point decision models and plugin hooks at the revised #148/#149 scope. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/comparisons/lich-vs-hermes.md | 8 +- wiki/concepts/decision-models.md | 9 +- wiki/entities/hermes-agent.md | 13 ++- wiki/entities/lich-plugins-and-hooks.md | 6 +- wiki/index.md | 4 +- wiki/log.md | 2 + ...26-10-02-hermes-models-memory-decisions.md | 86 +++++++++++++++++++ 7 files changed, 117 insertions(+), 11 deletions(-) create mode 100644 wiki/raw/audits/2026-10-02-hermes-models-memory-decisions.md diff --git a/wiki/comparisons/lich-vs-hermes.md b/wiki/comparisons/lich-vs-hermes.md index f697537..4a2a36c 100644 --- a/wiki/comparisons/lich-vs-hermes.md +++ b/wiki/comparisons/lich-vs-hermes.md @@ -1,10 +1,10 @@ --- title: Lich vs Hermes created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-02 type: comparison tags: [hermes, ecosystem, research] -sources: [raw/audits/2026-09-23-hermes-vs-lich.md, raw/audits/2026-09-23-core-engine-audit.md, "#113"] +sources: [raw/audits/2026-09-23-hermes-vs-lich.md, raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#113", "#149"] confidence: high --- @@ -38,6 +38,10 @@ This compares Lich with [[hermes-agent]], each feature checked against the Lich | Cron | full scheduler + agent tool | none | **skip for now**: sim world ticks later | | File checkpoints | shadow git | none | **port later**: cheap, useful for coding games | | UIs | CLI, Ink TUI, Electron, web, ACP | CLI, Ink TUI, library | in progress ([[ossuary]]) | +| Model roles (2026-10-02) | `fallback_model` chain + per-task `auxiliary.` provider/model | one failover chain for everything, compression included | **adapt**: `models.chat` / `models.compress` naming providers (#149) | +| Plugin settings + model access (2026-10-02) | `plugins.entries..settings`, host-owned `ctx.llm` | module paths only, no model access | **port**: `{ path, settings, models }` + granted roles (#149) | +| LLM-call hooks (2026-10-02) | `pre_llm_call`, `post_llm_call`, `llm_request` middleware, aux-call hooks | tool and run hooks only | **port**: `before_llm_call` first (#149) | +| Decision models (2026-10-02) | none in core; 14 community Jev plugins | none | **adapt**: plugin-owned client, shadow first ([[decision-models]], #148) | ## What Lich does better diff --git a/wiki/concepts/decision-models.md b/wiki/concepts/decision-models.md index 478a913..aef2e93 100644 --- a/wiki/concepts/decision-models.md +++ b/wiki/concepts/decision-models.md @@ -4,7 +4,7 @@ created: 2026-10-02 updated: 2026-10-02 type: concept tags: [providers, performance, research, games, security] -sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, "#148"] +sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#148", "#149"] confidence: medium --- @@ -28,15 +28,20 @@ confidence: medium 4. **Tool-list narrowing** before an LLM turn, which cuts prompt tokens. 5. **A second-opinion guard** in a `before_tool_call` hook ([[lich-plugins-and-hooks]]). +## How Hermes does it + +[[hermes-agent]] core has no decision-model support. Fourteen community plugins add it on generic host features: per-plugin settings, host-owned model access, and hooks before tool and LLM calls. They route skills (`pre_llm_call`), gate tools (`pre_tool_call`, shadow by default), review approvals (failures escalate), pick per-turn model and effort (request middleware), and skip idle cron runs. ^[raw/audits/2026-10-02-hermes-models-memory-decisions.md] + ## Rules of use - **Optional plugin, never core.** This follows [[0008-idea-agnostic-extensible-harness]]: the loop stays model-agnostic and the decision client is an extension. +- **Shadow mode first,** with one log line per decision (question, answer, confidence, latency, fallback reason), as the Hermes plugins do. - **Always a fallback.** Low confidence, a timeout, or a compound or vague question goes to the LLM or a scripted action. - **Add vetoes, never remove them.** Published work shows prompt injection shifts the probabilities and can move the answer. A decision-model verdict may block an action, but deterministic checks (URL guard, gatekeeper, MCP refusals) always still run. See [[embedded-safety-profile]]. ## Getting the models -- **Ollama (local or ollama.com):** 0.35 adds `/v1/systemone` with `nimble` (9B) and `tev1` (4B, 0.8B). Lich's [[lich-providers]] Ollama client only speaks `/api/chat` (`src/providers/ollama.ts:155-158@e9bdd82`), so a separate small client is needed. +- **Ollama (local or ollama.com):** 0.35 adds `/v1/systemone` with `nimble` (9B) and `tev1` (4B, 0.8B). Lich's [[lich-providers]] Ollama client only speaks `/api/chat` (`src/providers/ollama.ts:155-158@e9bdd82`), so the decision client lives in the plugin, not core, as in Hermes. The plugin gets its endpoint from per-plugin settings (#149). - **Cloudflare Workers AI:** `clef` (27B) and `clef-flash` (9B). Clef is not in the Ollama library; Ollaya runs it locally. Numbers above come from search summaries, not the primary pages (blocked during research); re-verify before relying on them. diff --git a/wiki/entities/hermes-agent.md b/wiki/entities/hermes-agent.md index c4ddd3f..5afce07 100644 --- a/wiki/entities/hermes-agent.md +++ b/wiki/entities/hermes-agent.md @@ -1,10 +1,10 @@ --- title: Hermes Agent (Nous Research) created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-02 type: entity tags: [hermes, research, ecosystem] -sources: [raw/audits/2026-09-23-hermes-vs-lich.md] +sources: [raw/audits/2026-09-23-hermes-vs-lich.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149"] confidence: high --- @@ -36,6 +36,15 @@ Details and file paths are in ^[raw/audits/2026-09-23-hermes-vs-lich.md]. - **The gateway owns sessions and cron ticks.** This is the model for [[0001-gateway-as-hub]]. - **Trajectory output** (ShareGPT JSONL), `batch_runner.py`, **computer use** and **vision**. +## Models, memory and decision plugins (read at `bed0d535`, 2026-10-02) + +A fresh clone of upstream was read for #148 and #149. Paths are in the Hermes repo. ^[raw/audits/2026-10-02-hermes-models-memory-decisions.md] + +- **Model roles.** The main model has a `fallback_model` chain (`hermes_cli/config.py:1003`). Each side task (compression, vision, web extract, titles, session search, background review) has its own `auxiliary.` provider and model, defaulting to "auto" (main model first) and resolved in `_resolve_task_provider_model` (`agent/auxiliary_client.py:6077`). That file is 8,256 lines, and overrides are still labelled experimental. +- **Embeddings live in memory plugins, not core.** One `MemoryProvider` at a time (`agent/memory_provider.py:84`); mem0 brings its own embedder, defaulting to Ollama `nomic-embed-text`. +- **Decision models are plugins only.** 14 community Jev plugins in `plugin-catalog/` route skills, gate tools, review approvals, pick per-turn model and effort, and skip idle cron runs. Shared rules: shadow mode first, one log line per decision, thresholds in code, capped payloads, egress disclosure, fail open for routing and fail closed for approvals. See [[decision-models]]. +- **What those plugins stand on:** per-plugin `settings` (`ctx.get_config`, `hermes_cli/plugins.py:270`), host-owned model access (`ctx.llm`, keys never exposed, overrides fail closed) and hooks such as `pre_llm_call` (`hermes_cli/plugins.py:109`). Lich plans the same three in #149; see [[lich-plugins-and-hooks]]. + ## History note The Atropos RL `environments/` directory was **removed** on 2026-05-15 in commit `5af672c753` (#26106). The old code is still useful as a reference when designing `lich env`: `git show 5af672c753^:environments/hermes_base_env.py`. diff --git a/wiki/entities/lich-plugins-and-hooks.md b/wiki/entities/lich-plugins-and-hooks.md index 747b6e1..f177966 100644 --- a/wiki/entities/lich-plugins-and-hooks.md +++ b/wiki/entities/lich-plugins-and-hooks.md @@ -1,10 +1,10 @@ --- title: Lich plugins, hooks and the gatekeeper created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-02 type: entity tags: [plugins, security, runtime] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149"] confidence: high --- @@ -17,7 +17,7 @@ A plugin is `{name, tools?, hooks?}`, loaded from paths listed in `config.plugin - **`before_tool_call`** can veto. A veto becomes a tool error whose text starts with `blocked_by_plugin:`. - **`after_tool_call`**, **`on_run_start`** and **`on_run_end`**. -There is **no hook that sees the prompt or the messages**. A plugin therefore cannot inject memory, skills or world state before an LLM call. Adding `before_llm_call` and `build_system_prompt` hooks is item 7 of #113 §2c. Hermes gets the same effect with prompt tiers; see [[prompt-cache-tiers]]. +There is **no hook that sees the prompt or the messages**. A plugin therefore cannot inject memory, skills or world state before an LLM call. Adding `before_llm_call` and `build_system_prompt` hooks is item 7 of #113 §2c. Hermes gets the same effect with prompt tiers; see [[prompt-cache-tiers]]. #149 plans a `before_llm_call` hook, per-plugin `settings`, and model calls limited to granted roles, matching what [[hermes-agent]]'s decision and memory plugins rely on. Hook state was a module-global WeakMap in v0.8.0. It is **per run via AsyncLocalStorage on v0.9.0**, which fixes the case where concurrent runs clobbered each other's state. diff --git a/wiki/index.md b/wiki/index.md index c729581..8cb97d2 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -17,7 +17,7 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) - [[lich-agent-loop]]: `run_conversation` + `Agent`. A dependency-injected TAO loop. Its P0 gaps are that events aren't scoped to a run, there is no streaming, it has global state, and each agent is heavyweight. - [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover. They have no streaming, `tool_choice` or cache_control, and ~150 lines of their helpers are duplicated. - [[lich-tools-and-guardrails]]: builtins, an executor that never throws, and the wards. Known holes: `tools_enabled` doesn't restrict plugin tools, and `terminal` isn't sandboxed. -- [[lich-plugins-and-hooks]]: tool-call hooks with veto, and the gatekeeper's single gated `git_commit`. There are no prompt-level hooks, and hooks fail open. +- [[lich-plugins-and-hooks]]: tool-call hooks with veto, and the gatekeeper's single gated `git_commit`. There are no prompt-level hooks or plugin settings yet (#149), and hooks fail open. - [[lich-sessions]]: JSONL phylacteries used as combat logs. There is no search, and gateway files are supersets of each other. - [[lich-mcp]]: an MCP client and catalog. Redot is a real entry and Godot has none. The code is spread over 21 micro-files. - [[lich-gateway]]: familiars routed into one shared Agent. The per-chat bus and read-only defaults make it a good hub. @@ -26,7 +26,7 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) ## Entities: the ecosystem and games -- [[hermes-agent]]: the Python agent that inspired Lich, with local paths and the mechanisms worth studying. Its RL environments were removed in `5af672c753`. +- [[hermes-agent]]: the Python agent that inspired Lich, with local paths and the mechanisms worth studying, including per-task model roles, memory-plugin embedders and its Jev plugin ecosystem. Its RL environments were removed in `5af672c753`. - [[godot-and-redot]]: the game→Lich direction (webhook plus file bus) and the Lich→editor direction (Redot MCP). `WebSocketPeer` is the path to a GDScript SDK. - [[game-bridge-example]]: the file-bus enemy commander. It's racy, needs 2 LLM calls per decision, and will retire once client tools exist. - [[persona-orchestrator-example]]: one Agent per NPC persona, which collapses to ~20 lines once Profiles and Sessions exist. diff --git a/wiki/log.md b/wiki/log.md index ccd3655..cd21229 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -28,3 +28,5 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-02 ingest | Decision models (Jev, Clef) and Ollama model research from web-search summaries; primary pages blocked | raw/audits/2026-10-02-decision-models-ollama-research.md - 2026-10-02 create | Decision models concept: typed calibrated decisions as an optional plugin fast lane with LLM fallback; filed #148 | concepts/decision-models.md, index.md - 2026-10-02 update | Providers: Ollama Bearer auth already supports ollama.com cloud; README key note and llama3.2 default are stale (#148); action-terminal-mode links the decision-model fast lane | entities/lich-providers.md, concepts/action-terminal-mode.md +- 2026-10-02 ingest | Hermes upstream at bed0d535: per-task auxiliary models and fallback chain, memory-plugin embedders, 14 community Jev plugins and the host features they use | raw/audits/2026-10-02-hermes-models-memory-decisions.md +- 2026-10-02 update | Hermes model roles, memory and decision plugins; comparison rows for model roles, plugin settings, LLM-call hooks and decision models; decision models stay plugin-owned with shadow mode first; #148/#149 revised to match | entities/hermes-agent.md, comparisons/lich-vs-hermes.md, concepts/decision-models.md, entities/lich-plugins-and-hooks.md, index.md diff --git a/wiki/raw/audits/2026-10-02-hermes-models-memory-decisions.md b/wiki/raw/audits/2026-10-02-hermes-models-memory-decisions.md new file mode 100644 index 0000000..69e3a5d --- /dev/null +++ b/wiki/raw/audits/2026-10-02-hermes-models-memory-decisions.md @@ -0,0 +1,86 @@ +--- +source_url: https://github.com/NousResearch/hermes-agent (shallow clone at bed0d535, read 2026-10-02) +ingested: 2026-10-02 +sha256: f8d5f79f98e25390922eaf499b1478fab4a49c5b31b462a352d7c82deda9ad98 +--- +# Hermes: model roles, memory/embeddings and decision models (2026-10-02) + +Read-only study of the Hermes Agent source (github.com/NousResearch/hermes-agent, +shallow clone at `bed0d535`, committed 2026-10-02) for #148 and #149. Paths are +relative to the Hermes repo root. + +## Mixing models + +- Main model: `model.default` in config, plus `fallback_model`, either one dict + `{provider, model}` or an ordered chain list (`hermes_cli/config.py:1003`, + validated in `_validate_fallback_model` at `:1147`). +- Side tasks ("auxiliary" models): each task has its own block + `auxiliary.` with `provider`, `model`, `base_url`, `api_key` or + `key_env`, `timeout`, `reasoning_effort` and optional `max_concurrency`. + Tasks documented in `cli-config.yaml.example` (Auxiliary Models section): + vision, web_extract, tts_audio_tags, title_generation, session_search, + compression, background_review (also curator, moa_reference). +- Provider `auto` means: main provider+model, then OpenRouter, Nous Portal, + custom endpoint, native Anthropic, direct API-key providers + (`agent/auxiliary_client.py` module docstring). `ollama-cloud` is a named + provider choice (needs `OLLAMA_API_KEY`). +- Resolution priority: explicit call args > `auxiliary..*` config > auto + (`_resolve_task_provider_model`, `agent/auxiliary_client.py:6077`). + Entry point `call_llm(task=..., ...)` at `:7883`, with a per-task semaphore. +- Compression is summarized "using a fast/cheap model"; pin it with + `auxiliary.compression.provider/model`. `resolve_compression_fast_lane` + (`:6211`) certifies a non-reasoning fast lane only for an explicit match. +- Delegated subagents have their own `delegation.model` and + `fallback_providers` chain. +- Size: `agent/auxiliary_client.py` is 8,256 lines, and the config still marks + auxiliary overrides "Advanced — Experimental". + +## Memory and embeddings + +- Core has no embedding client. Memory is pluggable: ONE external provider at a + time via `memory.provider`, plugins under `plugins/memory//` + (byterover, holographic, mem0, openviking, retaindb). +- `MemoryProvider` ABC (`agent/memory_provider.py:84`): `is_available`, + `initialize`, `system_prompt_block`, `prefetch` / `queue_prefetch`, + `sync_turn`, `get_tool_schemas`, `handle_tool_call`, `shutdown`, plus + optional `on_turn_start`, `on_session_end`, `on_pre_compress`, + `on_delegation`, `on_memory_write`, config schema and backup paths. +- Embedders live inside providers: mem0's OSS registry lists OpenAI + `text-embedding-3-small` (1536 dims) and Ollama `nomic-embed-text` (768 dims, + default URL `http://localhost:11434`) (`plugins/memory/mem0/_oss_providers.py:15-17`). +- The built-in store is bounded `MEMORY.md` / `USER.md` frozen into the prompt. + +## Decision models (Jev) + +- Core has no decision-model support (no `systemone` / Jev code outside the + plugin catalog and one dashboard reference). +- `plugin-catalog/` lists 14 community Jev entries, including: `jev`, + `jev-typesafe` (tools `jev_evaluate`/`jev_check`/`jev_route`/`jev_score`), + `jev-judge` (`pre_tool_call` gate, shadow by default, enforce escalates, + JSONL per decision, fail-open), `jev-approvals` (smart-approval reviewer; + failures fail closed to ESCALATE), `jev-skill-router` and + `typesafe-skill-router` (`pre_llm_call`, names one skill, says nothing when + nothing fits), `jev-effort-router` (`llm_request` middleware rewrites the + Ollama Cloud request's model and reasoning effort per turn), `jev-cron-gate` + (skip pointless cron runs, forced wake after N skips), `jev-curator`, + `hermes-structured-aux-models` (routes approval / MCP sampling / compression + aux tasks through decision calls, failing open to the aux provider), and + MCP-server packages (`jev-model-router`, `jev-agent-router`, + `jev-mcp-router`, `jev-memory-selector`). +- Shared traits: thresholds in code, shadow mode first, one JSONL line per + decision, capped payloads, explicit egress disclosure in the catalog entry, + fail open for routing, fail closed (escalate) for approvals. + +## Plugin host features the decision/memory plugins rely on + +- Hook names (`VALID_HOOKS`, `hermes_cli/plugins.py:109`): pre/post_tool_call, + transform_terminal_output, transform_tool_result, transform_llm_output, + pre/post_llm_call, stream observers, pre_verify, pre/post_api_request, + api_request_error, pre/post_auxiliary_call, and more; plus `llm_request` + middleware. +- Per-plugin settings: `ctx.get_config(key)` reads + `plugins.entries..settings.` (`hermes_cli/plugins.py:270`). +- Host-owned model access: `ctx.llm` (`agent/plugin_llm.py`) gives + `complete` / `complete_structured` on the user's routing and auth; plugins + never see keys; overrides are fail-closed behind + `plugins.entries..llm.allow_*_override`. Backed by `call_llm`. From af2d08fb8c925dd5a3b8b2310928a343859d3d84 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 21:17:12 +0000 Subject: [PATCH 04/43] docs: default local Ollama to qwen3:8b and document Ollama cloud (#148) llama3.2 (3B) is weak at tool calling; qwen3:8b is the common local pick for agents. Switch the README, docs, persona example and the setup wizard default to qwen3:8b, keeping llama3.2 as the low-memory note. The Ollama provider already sends a Bearer header when api_key or api_key_env resolves, so Ollama cloud works today. Document it (LICH_BASE_URL=https://ollama.com, LICH_API_KEY_ENV=OLLAMA_API_KEY) and replace "unused by ollama" with "optional for ollama". Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 4 ++++ README.md | 29 +++++++++++++++++++++------- docs/getting-started.md | 4 ++-- docs/user-guide/cli.md | 12 ++++++++---- docs/user-guide/library.md | 8 ++++---- docs/user-guide/ossuary.md | 2 +- docs/user-guide/tui.md | 6 +++--- examples/persona_orchestrator/run.ts | 2 +- src/cli.ts | 2 +- src/setup_wizard.ts | 2 +- 10 files changed, 47 insertions(+), 24 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index cb007d2..c6fa5fe 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased +- Docs, examples and the setup wizard now default local Ollama to `qwen3:8b` + (`llama3.2` stays as the low-memory option), and document Ollama cloud: + `LICH_BASE_URL=https://ollama.com` with `LICH_API_KEY_ENV=OLLAMA_API_KEY`. + The Ollama api key is optional, not unused (#148). - Gateway history cap no longer drops tool-heavy history when the capped window has no user turn. It keeps the window from the owning assistant turn and prepends a stub user turn so providers that require user-first history (Anthropic) diff --git a/README.md b/README.md index c41e056..b16d2d4 100644 --- a/README.md +++ b/README.md @@ -55,8 +55,8 @@ LICH_MODEL=gpt-4o-mini LICH_PROVIDER_KIND=openai_compat lich "summarize this rep LICH_MODEL=claude-sonnet-4-20250514 LICH_PROVIDER_KIND=anthropic lich chat # local ollama (no api key needed) -ollama pull llama3.2 -LICH_PROVIDER_KIND=ollama LICH_MODEL=llama3.2 lich "hello" +ollama pull qwen3:8b +LICH_PROVIDER_KIND=ollama LICH_MODEL=qwen3:8b lich "hello" # terminal UI lich tui @@ -81,7 +81,7 @@ import { run_agent } from "@moikapy/lich"; const result = await run_agent( { providers: [ - { kind: "ollama", name: "local", model: "llama3.2:latest" }, + { kind: "ollama", name: "local", model: "qwen3:8b" }, ], }, "Use the list_dir tool to list files, then summarize.", @@ -133,10 +133,10 @@ the next self-commit. One gated `git_commit` per run requires | Variable | Purpose | | --- | --- | -| `LICH_MODEL` | model id your provider accepts (e.g. `gpt-4o-mini`, `claude-sonnet-4-20250514`, `llama3.2`) | +| `LICH_MODEL` | model id your provider accepts (e.g. `gpt-4o-mini`, `claude-sonnet-4-20250514`, `qwen3:8b`) | | `LICH_PROVIDER_KIND` | `openai_compat` \| `anthropic` \| `ollama` (default `openai_compat`) | | `LICH_BASE_URL` | provider base url (ollama default: `http://localhost:11434`) | -| `LICH_API_KEY_ENV` | env var holding the api key (unused by ollama) | +| `LICH_API_KEY_ENV` | env var holding the api key (optional for ollama; set it for Ollama cloud) | | `LICH_ALLOW_SELF_COMMIT` | set to `1` to allow one gated `git_commit` per run; unset is fail-closed | | `LICH_ALLOW_PRIVATE_URLS` | set to exactly `1` to let `fetch_url` / `http_request` reach private or loopback URLs; unset or any other value is fail-closed (blocked) | | `LICH_TEST_COMMAND` | command `run_tests` runs (default: `node node_modules/vitest/vitest.mjs run`) | @@ -159,10 +159,25 @@ Short map of who can do what: ## Ollama -Ollama needs no api key and defaults to `http://localhost:11434`: +Local Ollama needs no api key and defaults to `http://localhost:11434`: ```sh -LICH_PROVIDER_KIND=ollama LICH_MODEL=llama3.2 lich "Reply with ok" +ollama pull qwen3:8b +LICH_PROVIDER_KIND=ollama LICH_MODEL=qwen3:8b lich "Reply with ok" +``` + +`qwen3:8b` (about 5 GB) calls tools reliably. On a low-memory machine, +`llama3.2` (3B) still works but calls tools less reliably. + +Ollama's hosted models use the same provider kind. Create an API key at +[ollama.com](https://ollama.com), then point the base url there; when +`api_key` or `api_key_env` resolves, requests carry +`Authorization: Bearer `: + +```sh +export OLLAMA_API_KEY=... +LICH_PROVIDER_KIND=ollama LICH_BASE_URL=https://ollama.com \ + LICH_API_KEY_ENV=OLLAMA_API_KEY LICH_MODEL= lich "Reply with ok" ``` Notes: diff --git a/docs/getting-started.md b/docs/getting-started.md index 792179f..e7c5617 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -21,7 +21,7 @@ Lich needs exactly one thing before it runs: a model. You can provide it three w ```sh # ollama — no api key needed -LICH_PROVIDER_KIND=ollama LICH_MODEL=llama3.2 lich "Reply with ok" +LICH_PROVIDER_KIND=ollama LICH_MODEL=qwen3:8b lich "Reply with ok" # openai-compatible (api.openai.com/v1 by default; use a model id your provider accepts) LICH_PROVIDER_KIND=openai_compat LICH_MODEL=gpt-4o-mini lich "Reply with ok" @@ -30,7 +30,7 @@ LICH_PROVIDER_KIND=openai_compat LICH_MODEL=gpt-4o-mini lich "Reply with ok" LICH_PROVIDER_KIND=anthropic LICH_MODEL=claude-sonnet-4-20250514 lich "Reply with ok" ``` -Defaults per kind when `LICH_BASE_URL`/`LICH_API_KEY_ENV` are unset: `openai_compat` uses `https://api.openai.com/v1` and reads `OPENAI_API_KEY`; `anthropic` uses `https://api.anthropic.com` and reads `ANTHROPIC_API_KEY`; `ollama` uses `http://localhost:11434` and needs no key. +Defaults per kind when `LICH_BASE_URL`/`LICH_API_KEY_ENV` are unset: `openai_compat` uses `https://api.openai.com/v1` and reads `OPENAI_API_KEY`; `anthropic` uses `https://api.anthropic.com` and reads `ANTHROPIC_API_KEY`; `ollama` uses `http://localhost:11434` and needs no key. For Ollama's hosted models, set `LICH_BASE_URL=https://ollama.com` and `LICH_API_KEY_ENV=OLLAMA_API_KEY` (create the key at ollama.com). ### Path B: the `config` template (recommended) diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index a021110..a9757bc 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -79,7 +79,7 @@ Per-kind defaults: | --- | --- | --- | --- | | `openai_compat` | `https://api.openai.com/v1` | `OPENAI_API_KEY` | Works with any OpenAI-shaped `/chat/completions` API. | | `anthropic` | `https://api.anthropic.com` | `ANTHROPIC_API_KEY` | | -| `ollama` | `http://localhost:11434` | none | No key needed; `api_key`/`api_key_env` are sent as a Bearer header for cloud proxies when set. | +| `ollama` | `http://localhost:11434` | none | No key needed locally; `api_key`/`api_key_env` are sent as a Bearer header when set, which Ollama cloud (`https://ollama.com`, `OLLAMA_API_KEY`) requires. | ## Config file reference @@ -98,7 +98,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid { "kind": "ollama", "name": "local", - "model": "llama3.2:latest", + "model": "qwen3:8b", "base_url": "http://localhost:11434", "keep_alive": "10m" } @@ -150,7 +150,11 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid Minimal per-provider examples: ```json -{ "providers": [{ "kind": "ollama", "name": "local", "model": "llama3.2" }] } +{ "providers": [{ "kind": "ollama", "name": "local", "model": "qwen3:8b" }] } +``` + +```json +{ "providers": [{ "kind": "ollama", "name": "cloud", "model": "", "base_url": "https://ollama.com", "api_key_env": "OLLAMA_API_KEY" }] } ``` ```json @@ -229,6 +233,6 @@ Batch one-shots from a script, checking each exit code: set -u for task in "summarize README.md" "list the largest files with disk_usage" "grep for TODO comments"; do echo "== $task" - LICH_PROVIDER_KIND=ollama LICH_MODEL=llama3.2 lich --max-turns 10 "$task" || echo "FAILED ($?)" + LICH_PROVIDER_KIND=ollama LICH_MODEL=qwen3:8b lich --max-turns 10 "$task" || echo "FAILED ($?)" done ``` \ No newline at end of file diff --git a/docs/user-guide/library.md b/docs/user-guide/library.md index 5dea34e..25e413a 100644 --- a/docs/user-guide/library.md +++ b/docs/user-guide/library.md @@ -26,7 +26,7 @@ The package ships ESM (`dist/index.js`, types at `dist/index.d.ts`, binary at `d import { create_agent } from "@moikapy/lich"; const agent = create_agent({ - providers: [{ kind: "ollama", name: "local", model: "llama3.2:latest" }], + providers: [{ kind: "ollama", name: "local", model: "qwen3:8b" }], }); const result = await agent.run({ input: "Use list_dir to list the files, then summarize." }); @@ -40,7 +40,7 @@ console.log(`tokens: ${result.usage_total.total_tokens}`); import { run_agent } from "@moikapy/lich"; const result = await run_agent( - { providers: [{ kind: "ollama", name: "local", model: "llama3.2:latest" }] }, + { providers: [{ kind: "ollama", name: "local", model: "qwen3:8b" }] }, "Reply with ok", ); ``` @@ -132,7 +132,7 @@ const config = { providers: [ { kind: "openai_compat", name: "openrouter", model: "meta-llama/llama-3.1-8b-instruct", base_url: "https://openrouter.ai/api/v1", api_key_env: "OPENROUTER_API_KEY" }, - { kind: "ollama", name: "local", model: "llama3.2" }, // failover target + { kind: "ollama", name: "local", model: "qwen3:8b" }, // failover target ], max_turns: 25, tools_enabled: ["read_file", "list_dir", "terminal", "web_search", "fetch_url"], @@ -152,7 +152,7 @@ Listed providers form a failover chain tried in order: `rate_limit`/`network` er ```ts const agent = create_agent({ - providers: [{ kind: "ollama", name: "local", model: "llama3.2" }], + providers: [{ kind: "ollama", name: "local", model: "qwen3:8b" }], tools_enabled: ["read_file", "grep_files", "list_dir"], }); ``` diff --git a/docs/user-guide/ossuary.md b/docs/user-guide/ossuary.md index 9fbeb41..1dfa0a9 100644 --- a/docs/user-guide/ossuary.md +++ b/docs/user-guide/ossuary.md @@ -14,7 +14,7 @@ bun install bun install --cwd apps/ossuary # write a local config once (or use the bare-lich wizard in a TTY) -bun src/cli.ts init --provider-kind ollama --model llama3.2 +bun src/cli.ts init --provider-kind ollama --model qwen3:8b # open ossuary (builds the Electron app on first run, then opens a window) bun src/cli.ts ossuary diff --git a/docs/user-guide/tui.md b/docs/user-guide/tui.md index c7f5a72..823d5cc 100644 --- a/docs/user-guide/tui.md +++ b/docs/user-guide/tui.md @@ -9,17 +9,17 @@ lich # front door: TUI, plus a first-run setup wizard when no config exi lich tui # same TUI, no wizard. From a clone: bun src/cli.ts tui ``` -The TUI needs a TTY and a resolvable provider (same resolution as every mode). On startup it prints a dim header from the active theme welcome string, e.g. `⚱ lich v0.8.0 — the agent that will not stay dead · llama3.2 (ollama)`. `{version}` is `LICH_VERSION` from `package.json`. That banner is the only tagline placement. Quit with `/exit`, `/quit`, `/q`, or Ctrl+C. +The TUI needs a TTY and a resolvable provider (same resolution as every mode). On startup it prints a dim header from the active theme welcome string, e.g. `⚱ lich v0.8.0 — the agent that will not stay dead · qwen3:8b (ollama)`. `{version}` is `LICH_VERSION` from `package.json`. That banner is the only tagline placement. Quit with `/exit`, `/quit`, `/q`, or Ctrl+C. ## Anatomy ``` -⚱ lich v0.8.0 — the agent that will not stay dead · llama3.2 (ollama) +⚱ lich v0.8.0 — the agent that will not stay dead · qwen3:8b (ollama) mortal › list the files here <- your input, echoed into the transcript ⏺ list_dir({}) <- live tool-call row (name + args preview) ⏷ list_dir: ok (d src/ d test/ ...) <- result row (ok/error + output preview) lich › Here is what I found ... <- the agent's reply (`response_label`) -model llama3.2 · turns 2 · tokens 1,204 · [dormant] · /path/.lich/sessions/...jsonl +model qwen3:8b · turns 2 · tokens 1,204 · [dormant] · /path/.lich/sessions/...jsonl › ▌ <- input row (cursor block) ``` diff --git a/examples/persona_orchestrator/run.ts b/examples/persona_orchestrator/run.ts index 0f64a1b..9628d3f 100644 --- a/examples/persona_orchestrator/run.ts +++ b/examples/persona_orchestrator/run.ts @@ -18,7 +18,7 @@ export async function main(): Promise { const orchestrator = create_orchestrator({ factory, shared: { - providers: [{ kind: "ollama", name: "local", model: "llama3.2" }], + providers: [{ kind: "ollama", name: "local", model: "qwen3:8b" }], work_dir: process.cwd(), }, personas: PERSONA_TABLE, diff --git a/src/cli.ts b/src/cli.ts index 7472ed0..a6befd5 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -86,7 +86,7 @@ function usage_text(): string { " --model model name (default from LICH_MODEL)", " --provider-kind openai_compat | anthropic | ollama (default LICH_PROVIDER_KIND)", " --base-url provider base url (default LICH_BASE_URL)", - " --api-key-env env var holding the api key (default LICH_API_KEY_ENV; unused by ollama)", + " --api-key-env env var holding the api key (default LICH_API_KEY_ENV; optional for ollama)", " --system-prompt system prompt override", " --session-dir session transcript directory", " --resume TUI only: load an existing session transcript", diff --git a/src/setup_wizard.ts b/src/setup_wizard.ts index 7b7b0b1..5864b82 100644 --- a/src/setup_wizard.ts +++ b/src/setup_wizard.ts @@ -13,7 +13,7 @@ const PLATFORMS = ["webhook", "telegram", "discord", "twitch"] as const; const PLUGIN_EXT = /\.(mjs|js|ts|mts|cts|jsx|tsx)$/; const PROVIDER_DEFAULTS: Record = { - ollama: { model: "llama3.2", base_url: "http://localhost:11434" }, + ollama: { model: "qwen3:8b", base_url: "http://localhost:11434" }, openai_compat: { model: "gpt-4.1-mini", base_url: "https://api.openai.com/v1", api_key_env: "OPENAI_API_KEY" }, anthropic: { model: "claude-sonnet-4", base_url: "https://api.anthropic.com", api_key_env: "ANTHROPIC_API_KEY" }, }; From 45f9ef1c66fa0171a00338d120fc9ae76dfc3791 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 21:33:25 +0000 Subject: [PATCH 05/43] wiki: fix Hermes Jev plugin count and update providers after #152 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/comparisons/lich-vs-hermes.md | 2 +- wiki/entities/hermes-agent.md | 2 +- wiki/entities/lich-providers.md | 4 ++-- wiki/log.md | 1 + 4 files changed, 5 insertions(+), 4 deletions(-) diff --git a/wiki/comparisons/lich-vs-hermes.md b/wiki/comparisons/lich-vs-hermes.md index 4a2a36c..d599658 100644 --- a/wiki/comparisons/lich-vs-hermes.md +++ b/wiki/comparisons/lich-vs-hermes.md @@ -41,7 +41,7 @@ This compares Lich with [[hermes-agent]], each feature checked against the Lich | Model roles (2026-10-02) | `fallback_model` chain + per-task `auxiliary.` provider/model | one failover chain for everything, compression included | **adapt**: `models.chat` / `models.compress` naming providers (#149) | | Plugin settings + model access (2026-10-02) | `plugins.entries..settings`, host-owned `ctx.llm` | module paths only, no model access | **port**: `{ path, settings, models }` + granted roles (#149) | | LLM-call hooks (2026-10-02) | `pre_llm_call`, `post_llm_call`, `llm_request` middleware, aux-call hooks | tool and run hooks only | **port**: `before_llm_call` first (#149) | -| Decision models (2026-10-02) | none in core; 14 community Jev plugins | none | **adapt**: plugin-owned client, shadow first ([[decision-models]], #148) | +| Decision models (2026-10-02) | none in core; 15 community Jev plugins (+4 with Jev backends) | none | **adapt**: plugin-owned client, shadow first ([[decision-models]], #148) | ## What Lich does better diff --git a/wiki/entities/hermes-agent.md b/wiki/entities/hermes-agent.md index 5afce07..b188208 100644 --- a/wiki/entities/hermes-agent.md +++ b/wiki/entities/hermes-agent.md @@ -42,7 +42,7 @@ A fresh clone of upstream was read for #148 and #149. Paths are in the Hermes re - **Model roles.** The main model has a `fallback_model` chain (`hermes_cli/config.py:1003`). Each side task (compression, vision, web extract, titles, session search, background review) has its own `auxiliary.` provider and model, defaulting to "auto" (main model first) and resolved in `_resolve_task_provider_model` (`agent/auxiliary_client.py:6077`). That file is 8,256 lines, and overrides are still labelled experimental. - **Embeddings live in memory plugins, not core.** One `MemoryProvider` at a time (`agent/memory_provider.py:84`); mem0 brings its own embedder, defaulting to Ollama `nomic-embed-text`. -- **Decision models are plugins only.** 14 community Jev plugins in `plugin-catalog/` route skills, gate tools, review approvals, pick per-turn model and effort, and skip idle cron runs. Shared rules: shadow mode first, one log line per decision, thresholds in code, capped payloads, egress disclosure, fail open for routing and fail closed for approvals. See [[decision-models]]. +- **Decision models are plugins only.** 15 dedicated community Jev plugins in `plugin-catalog/` (plus 4 broader plugins with Jev backends) route skills, gate tools, review approvals, pick per-turn model and effort, and skip idle cron runs. Shared rules: shadow mode first, one log line per decision, thresholds in code, capped payloads, egress disclosure, fail open for routing and fail closed for approvals. See [[decision-models]]. - **What those plugins stand on:** per-plugin `settings` (`ctx.get_config`, `hermes_cli/plugins.py:270`), host-owned model access (`ctx.llm`, keys never exposed, overrides fail closed) and hooks such as `pre_llm_call` (`hermes_cli/plugins.py:109`). Lich plans the same three in #149; see [[lich-plugins-and-hooks]]. ## History note diff --git a/wiki/entities/lich-providers.md b/wiki/entities/lich-providers.md index f498996..9a1fdce 100644 --- a/wiki/entities/lich-providers.md +++ b/wiki/entities/lich-providers.md @@ -4,7 +4,7 @@ created: 2026-09-23 updated: 2026-10-02 type: entity tags: [providers, runtime, performance] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-decision-models-ollama-research.md, "#148"] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-decision-models-ollama-research.md, "#148", "#152"] confidence: high --- @@ -23,7 +23,7 @@ confidence: high ## Ollama: local and ollama.com cloud - The Ollama client defaults to `http://localhost:11434` and needs no key. When `api_key` or `api_key_env` resolves, it sends `Authorization: Bearer` (`src/providers/ollama.ts:137-165@e9bdd82`). That is enough for Ollama's hosted models: `LICH_BASE_URL=https://ollama.com` plus `LICH_API_KEY_ENV=OLLAMA_API_KEY`. Not yet smoke-tested; tracked in #148. -- The README still calls the key "unused by ollama", and the documented default model is `llama3.2` (3B), which is weak at tool calling. #148 proposes `qwen3:8b`. +- The documented local default is `qwen3:8b` (the setup wizard writes it too; `src/setup_wizard.ts:16@d28dd0b`), replacing `llama3.2` (3B), which is weak at tool calling. The README documents Ollama cloud via `LICH_BASE_URL` plus `LICH_API_KEY_ENV` (`README.md:160-181@d28dd0b`). Merged in #152. - Ollama 0.35's decision endpoint (`/v1/systemone`) is a different API from `/api/chat`; see [[decision-models]]. ## Gaps (open on v0.9.0) diff --git a/wiki/log.md b/wiki/log.md index cd21229..be3d6a7 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -30,3 +30,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-02 update | Providers: Ollama Bearer auth already supports ollama.com cloud; README key note and llama3.2 default are stale (#148); action-terminal-mode links the decision-model fast lane | entities/lich-providers.md, concepts/action-terminal-mode.md - 2026-10-02 ingest | Hermes upstream at bed0d535: per-task auxiliary models and fallback chain, memory-plugin embedders, 14 community Jev plugins and the host features they use | raw/audits/2026-10-02-hermes-models-memory-decisions.md - 2026-10-02 update | Hermes model roles, memory and decision plugins; comparison rows for model roles, plugin settings, LLM-call hooks and decision models; decision models stay plugin-owned with shadow mode first; #148/#149 revised to match | entities/hermes-agent.md, comparisons/lich-vs-hermes.md, concepts/decision-models.md, entities/lich-plugins-and-hooks.md, index.md +- 2026-10-02 update | Fix Hermes Jev plugin count (15 dedicated plus 4 with Jev backends; raw source left as captured); providers page now reflects merged #152 (qwen3:8b default, Ollama cloud documented) | entities/hermes-agent.md, comparisons/lich-vs-hermes.md, entities/lich-providers.md From 84e91ab5c6be4559a958b1b32b8d759d87bf8ac1 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 21:37:54 +0000 Subject: [PATCH 06/43] feat(config): per-role provider chains via models.chat and models.compress Optional `models` block names provider chains per role. `chat` sets the main loop's failover order; `compress` routes context compression and falls back to the chat chain on failure. ProviderRouter.for_role shares built clients. Without `models`, behavior is unchanged. The TUI labels the first chat-role provider. Part of #149. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 3 + docs/user-guide/cli.md | 1 + src/agent/agent.ts | 7 +- src/agent/config.ts | 41 ++++++++++ src/agent/loop.ts | 27 ++++++- src/providers/router.ts | 17 +++- src/tui/app.tsx | 3 +- src/tui/state.ts | 18 ++++- test/model_roles.test.ts | 165 +++++++++++++++++++++++++++++++++++++++ test/tui.test.ts | 11 +++ 10 files changed, 286 insertions(+), 7 deletions(-) create mode 100644 test/model_roles.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index c6fa5fe..a2bac24 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,9 @@ ## Unreleased +- Optional `models` config block assigns provider chains per role: `chat` for + the main loop and `compress` for context compression (falls back to `chat` + on failure). Without it, behavior is unchanged (#149). - Docs, examples and the setup wizard now default local Ollama to `qwen3:8b` (`llama3.2` stays as the low-memory option), and document Ollama cloud: `LICH_BASE_URL=https://ollama.com` with `LICH_API_KEY_ENV=OLLAMA_API_KEY`. diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index a9757bc..6a8bb0e 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -130,6 +130,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid | `providers[].timeout_ms` | positive int | none | Per-request abort deadline. | | `providers[].think` | boolean | – | Ollama only: request thinking mode. | | `providers[].keep_alive` | string | – | Ollama only: model residency (e.g. `"10m"`). | +| `models` | object | omitted | Optional per-role provider chains by name. `chat` is the main loop's failover order; `compress` is the context-compression chain and falls back to the `chat` chain when it fails. An omitted role uses `providers` order. Names must exist in `providers` and appear once per role. Example: `"models": { "chat": ["claude", "local"], "compress": ["local"] }`. | | `agent_name` | string | `lich` | Wizard label. The TUI banner uses the active theme welcome string, not this field. | | `theme` | string | `lich` | Display theme name. See [Themes](https://github.com/Moikapy/lich/blob/main/README.md#themes). | | `gateway` | object | omitted | Optional. `platforms` (`webhook` \| `telegram` \| `discord` \| `twitch`) and `token_envs` (platform → env-var name). Secrets stay in the environment. | diff --git a/src/agent/agent.ts b/src/agent/agent.ts index 7e0cf6b..47dd825 100644 --- a/src/agent/agent.ts +++ b/src/agent/agent.ts @@ -122,6 +122,7 @@ export class Agent { readonly events: EnvelopedAgentEmitter; readonly config: AgentConfig; private readonly router: ProviderRouter; + private readonly compress_router: ProviderRouter | undefined; private readonly registry: ToolRegistry; private readonly executor: ToolExecutor | HookedToolRunner; private readonly hook_runner: HookedToolRunner | undefined; @@ -133,7 +134,10 @@ export class Agent { this.config = config; this.mcp_runtime = runtime?.mcp; this.events = new EnvelopedAgentEmitter(); - this.router = new ProviderRouter(config.providers); + const all_providers = new ProviderRouter(config.providers); + const roles = config.models; + this.router = roles?.chat !== undefined ? all_providers.for_role(roles.chat) : all_providers; + this.compress_router = roles?.compress !== undefined ? all_providers.for_role(roles.compress) : undefined; const base_registry = new ToolRegistry(); register_builtin_tools(base_registry); this.registry = filter_registry(base_registry, config.tools_enabled); @@ -285,6 +289,7 @@ export class Agent { private loop_deps(tool_context: ToolContext, emitter: AgentEmitter): LoopDeps { return { chat: (messages, tools, chat_options) => this.router.chat_with_failover(messages, tools, chat_options), + compress_chat: this.compress_router?.chat_with_failover.bind(this.compress_router), tools: this.executor, definitions: () => this.registry.definitions(), emitter, diff --git a/src/agent/config.ts b/src/agent/config.ts index d346b65..b8fa84f 100644 --- a/src/agent/config.ts +++ b/src/agent/config.ts @@ -96,6 +96,19 @@ const providers_schema = z } }); +const role_schema = z.array(z.string().min(1)).min(1).optional(); + +/** Per-role provider chains by name. Omitted roles use `providers` order. */ +const models_schema = z + .object({ + /** Main loop failover order. */ + chat: role_schema, + /** Context-compression chain; falls back to `chat` when it fails. */ + compress: role_schema, + }) + .strict() + .optional(); + const agent_config_schema = z .object({ /** Wizard label. The TUI banner uses the active theme welcome string. */ @@ -103,6 +116,7 @@ const agent_config_schema = z system_prompt: z.string().optional(), max_turns: z.number().int().min(1).default(25), providers: providers_schema, + models: models_schema, work_dir: z.string().optional(), tools_enabled: z.union([z.literal("all"), z.array(z.string())]).default("all"), temperature: z.number().min(0).max(2).optional(), @@ -119,6 +133,28 @@ const agent_config_schema = z /** Named MCP servers. Each entry is stdio or loopback http. Default off. */ mcp_servers: mcp_servers_schema, }) + .superRefine((config, ctx) => { + const known = new Set(config.providers.map((provider) => provider.name)); + for (const [role, names] of Object.entries(config.models ?? {})) { + const seen = new Set(); + for (const [index, name] of (names ?? []).entries()) { + if (known.has(name) === false) { + ctx.addIssue({ + code: z.ZodIssueCode.custom, + path: ["models", role, index], + message: `unknown provider "${name}"`, + }); + } else if (seen.has(name) === true) { + ctx.addIssue({ + code: z.ZodIssueCode.custom, + path: ["models", role, index], + message: `provider "${name}" listed twice`, + }); + } + seen.add(name); + } + } + }) .transform((config) => { const work_dir = config.work_dir ?? process.cwd(); return { @@ -137,6 +173,11 @@ function freeze_config(config: AgentConfig): AgentConfig { for (const provider of config.providers) { Object.freeze(provider); } + if (config.models !== undefined) { + Object.freeze(config.models.chat); + Object.freeze(config.models.compress); + Object.freeze(config.models); + } Object.freeze(config.plugins); for (const plugin of config.plugins) { Object.freeze(plugin); diff --git a/src/agent/loop.ts b/src/agent/loop.ts index f9badd0..d935980 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -10,6 +10,7 @@ import { compress_messages, should_compress, split_keep_recent, type ChatFn } fr import { estimate_messages_tokens } from "../context/tokens.js"; import type { AssistantMessage, + ChatOptions, ChatResult, Message, ToolCall, @@ -34,6 +35,8 @@ export interface ToolRunner { export interface LoopDeps { chat: ChatFn; + /** Context-compression chat; falls back to `chat` when unset or failing. */ + compress_chat?: ChatFn; tools: ToolRunner; definitions: () => ToolDefinition[]; emitter?: AgentEmitter; @@ -180,6 +183,28 @@ function schedule_compress_backoff(backoff: CompressBackoff, turn: number): void backoff.skip_until_turn = turn + COMPRESS_BACKOFF_TURNS + 1; } +/** Compression role first; on a non-abort failure, retry on the main chat chain. */ +async function compress_chat( + deps: LoopDeps, + messages: readonly Message[], + tools: readonly ToolDefinition[], + options?: ChatOptions, +): Promise { + if (deps.compress_chat === undefined) { + return await deps.chat(messages, tools, options); + } + try { + return await deps.compress_chat(messages, tools, options); + } catch (error) { + if (options?.signal?.aborted === true) { + throw error; + } + const message = error instanceof Error ? error.message : String(error); + logger.warn(`compress chain failed (${message}); falling back to chat chain`); + return await deps.chat(messages, tools, options); + } +} + async function compress_if_needed( deps: LoopDeps, history: Message[], @@ -207,7 +232,7 @@ async function compress_if_needed( emitter?.emit({ type: "compress_start", estimated_tokens: estimate_messages_tokens(history) }); let summarizer_usage: Usage | undefined; const counting_chat: ChatFn = async (messages, tools, options) => { - const result = await deps.chat(messages, tools, options); + const result = await compress_chat(deps, messages, tools, options); summarizer_usage = result.usage; return result; }; diff --git a/src/providers/router.ts b/src/providers/router.ts index c085432..437f316 100644 --- a/src/providers/router.ts +++ b/src/providers/router.ts @@ -18,13 +18,26 @@ const FAILOVER_MAX_ATTEMPTS = 3; export class ProviderRouter { private readonly configs: ProviderConfig[]; - private readonly cache: Map = new Map(); + private readonly cache: Map; - constructor(configs: ProviderConfig[]) { + constructor(configs: ProviderConfig[], cache: Map = new Map()) { if (configs.length === 0) { throw new Error("at least one provider is required"); } this.configs = [...configs]; + this.cache = cache; + } + + /** Router over a named subset, in the given order, sharing built clients. */ + for_role(names: readonly string[]): ProviderRouter { + const configs = names.map((name) => { + const config = this.find_config(name); + if (config === undefined) { + throw new Error(`unknown provider "${name}"`); + } + return config; + }); + return new ProviderRouter(configs, this.cache); } get(name: string): LLMProvider | undefined { diff --git a/src/tui/app.tsx b/src/tui/app.tsx index 0a285bc..f6a351a 100644 --- a/src/tui/app.tsx +++ b/src/tui/app.tsx @@ -22,6 +22,7 @@ import { help_block, HISTORY_CAP, INITIAL_UI_STATE, + chat_provider, model_label_block, parse_command, tui_banner_text, @@ -321,7 +322,7 @@ export function TuiApp({ agent, theme, session, initial_history, resumed_id }: T [handle_slash, start_message_run], ); - const provider = agent.config.providers[0]; + const provider = chat_provider(agent.config); const banner = tui_banner_text(theme, LICH_VERSION, provider?.model ?? "unknown", provider?.kind ?? "unknown"); return ( diff --git a/src/tui/state.ts b/src/tui/state.ts index 30eb7ec..2478d62 100644 --- a/src/tui/state.ts +++ b/src/tui/state.ts @@ -282,8 +282,22 @@ export function help_block(): HistoryBlock { return { role: "meta", lines: [...HELP_LINES] }; } -export function model_label_block(config: { providers: readonly { model: string; kind: string }[] }): HistoryBlock { - const provider = config.providers[0]; +interface ChatProviderView { + providers: readonly { name?: string; model: string; kind: string }[]; + models?: { chat?: readonly string[] }; +} + +/** First provider of the main chat chain: `models.chat[0]` when set, else `providers[0]`. */ +export function chat_provider(config: ChatProviderView): { model: string; kind: string } | undefined { + const first = config.models?.chat?.[0]; + if (first === undefined) { + return config.providers[0]; + } + return config.providers.find((provider) => provider.name === first); +} + +export function model_label_block(config: ChatProviderView): HistoryBlock { + const provider = chat_provider(config); return { role: "meta", lines: [`\u00b7 model: ${provider?.model ?? "unknown"} \u00b7 provider: ${provider?.kind ?? "unknown"}`], diff --git a/test/model_roles.test.ts b/test/model_roles.test.ts new file mode 100644 index 0000000..9c2acc0 --- /dev/null +++ b/test/model_roles.test.ts @@ -0,0 +1,165 @@ +/** + * #149 PR A: `models.chat` / `models.compress` provider roles. + */ +import { mkdir, mkdtemp, rm } from "node:fs/promises"; +import path from "node:path"; +import { afterAll, describe, expect, it } from "vitest"; +import { create_agent } from "../src/agent/agent.js"; +import { parse_agent_config } from "../src/agent/config.js"; +import { run_conversation, type LoopDeps } from "../src/agent/loop.js"; +import type { ChatFn } from "../src/context/compressor.js"; +import { ProviderRouter } from "../src/providers/router.js"; +import type { ChatResult, Message, ProviderConfig } from "../src/providers/types.js"; +import { TMP_BASE } from "./helpers/tmp_base.js"; + +const temp_dirs: string[] = []; + +afterAll(async () => { + for (const dir of temp_dirs) { + await rm(dir, { recursive: true, force: true }); + } +}); + +async function make_temp_dir(): Promise { + await mkdir(TMP_BASE, { recursive: true }); + const dir = await mkdtemp(path.join(TMP_BASE, "model-roles-")); + temp_dirs.push(dir); + return dir; +} + +const providers = [ + { kind: "openai_compat", name: "a", model: "m" }, + { kind: "openai_compat", name: "b", model: "m" }, + { kind: "ollama", name: "local", model: "m" }, +]; + +function config_with(name: string, hits: string[], status = 200): ProviderConfig { + const fetch_fn: typeof fetch = async () => { + hits.push(name); + const body = { + model: "m", + choices: [{ message: { role: "assistant", content: `from ${name}` }, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }; + return new Response(JSON.stringify(body), { status }); + }; + return { kind: "openai_compat", name, model: "m", api_key: "k", base_url: "http://mock.local/v1", fetch_fn }; +} + +function result(content: string): ChatResult { + return { + message: { role: "assistant", content }, + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + finish_reason: "stop", + model: "mock-model", + provider_name: "mock", + }; +} + +function is_compress_call(messages: readonly Message[]): boolean { + const first = messages[0]; + return first?.role === "system" && first.content.includes("compress"); +} + +describe("models config", () => { + it("is optional and absent by default", () => { + expect(parse_agent_config({ providers }).models).toBeUndefined(); + }); + + it("accepts known provider names per role", () => { + const config = parse_agent_config({ providers, models: { chat: ["b", "a"], compress: ["local"] } }); + expect(config.models).toEqual({ chat: ["b", "a"], compress: ["local"] }); + expect(Object.isFrozen(config.models)).toBe(true); + }); + + it("rejects unknown and repeated names, empty lists and unknown roles", () => { + expect(() => parse_agent_config({ providers, models: { chat: ["nope"] } })).toThrow(/unknown provider .*nope/); + expect(() => parse_agent_config({ providers, models: { compress: ["a", "a"] } })).toThrow(/listed twice/); + expect(() => parse_agent_config({ providers, models: { chat: [] } })).toThrow(); + expect(() => parse_agent_config({ providers, models: { embed: ["a"] } })).toThrow(); + }); +}); + +describe("ProviderRouter.for_role", () => { + it("keeps the given order and shares built clients", () => { + const router = new ProviderRouter(parse_agent_config({ providers }).providers); + const role = router.for_role(["local", "a"]); + expect(role.list().map((provider) => provider.name)).toEqual(["local", "a"]); + expect(role.get("a")).toBe(router.get("a")); + }); + + it("throws on an unknown name", () => { + const router = new ProviderRouter(parse_agent_config({ providers }).providers); + expect(() => router.for_role(["zzz"])).toThrow(/unknown provider/); + }); + + it("fails over within the role only", async () => { + const hits: string[] = []; + const router = new ProviderRouter([config_with("a", hits), config_with("b", hits, 401), config_with("c", hits)]); + const out = await router.for_role(["b", "c"]).chat_with_failover([{ role: "user", content: "hi" }], []); + expect(out.message.content).toBe("from c"); + expect(hits).toEqual(["b", "c"]); + }); +}); + +describe("compress role in the loop", () => { + const pad = "x".repeat(400); + const seed: Message[] = Array.from({ length: 11 }, (_unused, index) => ({ + role: "user" as const, + content: `note ${index} ${pad}`, + })); + const params = { max_turns: 1, context_budget_tokens: 100, compress_threshold: 0.8 }; + const runner: LoopDeps["tools"] = { execute: async () => ({ ok: true, output: "ok" }) }; + + it("summarizes with compress_chat when set", async () => { + const main_calls: string[] = []; + const chat: ChatFn = async (messages) => { + main_calls.push(is_compress_call(messages) ? "compress" : "turn"); + return result("done"); + }; + const compress_chat: ChatFn = async () => result("CHEAP SUMMARY"); + const outcome = await run_conversation({ chat, compress_chat, tools: runner, definitions: () => [] }, seed, params); + expect(main_calls).toEqual(["turn"]); + expect(outcome.messages.some((message) => message.content?.includes("CHEAP SUMMARY") === true)).toBe(true); + }); + + it("falls back to chat when compress_chat fails", async () => { + const chat: ChatFn = async (messages) => result(is_compress_call(messages) ? "MAIN SUMMARY" : "done"); + const compress_chat: ChatFn = async () => { + throw new Error("local model down"); + }; + const outcome = await run_conversation({ chat, compress_chat, tools: runner, definitions: () => [] }, seed, params); + expect(outcome.messages.some((message) => message.content?.includes("MAIN SUMMARY") === true)).toBe(true); + }); +}); + +describe("Agent model roles", () => { + it("routes the main loop through models.chat", async () => { + const work_dir = await make_temp_dir(); + const hits: string[] = []; + const agent = create_agent({ + providers: [config_with("a", hits), config_with("b", hits)], + models: { chat: ["b"] }, + work_dir, + session_dir: path.join(work_dir, "sessions"), + log_level: "error", + }); + const run = await agent.run({ input: "go" }); + expect(run.outcome.final?.content).toBe("from b"); + expect(hits).toEqual(["b"]); + }); + + it("keeps providers order without a models block", async () => { + const work_dir = await make_temp_dir(); + const hits: string[] = []; + const agent = create_agent({ + providers: [config_with("a", hits), config_with("b", hits)], + work_dir, + session_dir: path.join(work_dir, "sessions"), + log_level: "error", + }); + const run = await agent.run({ input: "go" }); + expect(run.outcome.final?.content).toBe("from a"); + expect(hits).toEqual(["a"]); + }); +}); diff --git a/test/tui.test.ts b/test/tui.test.ts index bb8776f..6dc0323 100644 --- a/test/tui.test.ts +++ b/test/tui.test.ts @@ -438,6 +438,17 @@ describe("notice blocks", () => { expect(unknown_command_block("wat").role).toBe("error"); }); + it("labels the first models.chat provider when roles are set", () => { + const config = { + providers: [ + { name: "a", model: "first", kind: "openai_compat" }, + { name: "b", model: "chosen", kind: "ollama" }, + ], + models: { chat: ["b"] }, + }; + expect(model_label_block(config).lines[0]).toContain("model: chosen"); + }); + it("formats tool result blocks with ok and error styling flags", () => { const ok = tool_result_block({ id: "1", name: "shell", args: { cmd: "ls" } }, true, "file.txt"); expect(ok.role).toBe("tool"); From c00a4c7bad46ee1c39ed8b39d8c78ed1e8aa190a Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 21:43:22 +0000 Subject: [PATCH 07/43] test, docs: cover models.compress through Agent; clarify omitted-role fallback Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/user-guide/cli.md | 2 +- docs/user-guide/library.md | 2 +- src/agent/config.ts | 2 +- test/model_roles.test.ts | 22 ++++++++++++++++++++++ 4 files changed, 25 insertions(+), 3 deletions(-) diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index 6a8bb0e..cedd961 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -130,7 +130,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid | `providers[].timeout_ms` | positive int | none | Per-request abort deadline. | | `providers[].think` | boolean | – | Ollama only: request thinking mode. | | `providers[].keep_alive` | string | – | Ollama only: model residency (e.g. `"10m"`). | -| `models` | object | omitted | Optional per-role provider chains by name. `chat` is the main loop's failover order; `compress` is the context-compression chain and falls back to the `chat` chain when it fails. An omitted role uses `providers` order. Names must exist in `providers` and appear once per role. Example: `"models": { "chat": ["claude", "local"], "compress": ["local"] }`. | +| `models` | object | omitted | Optional per-role provider chains by name. `chat` is the main loop's failover order; `compress` is the context-compression chain and falls back to the `chat` chain when it fails. Without `chat`, the main loop uses `providers` order; without `compress`, compression uses the chat chain. Names must exist in `providers` and appear once per role. Example: `"models": { "chat": ["claude", "local"], "compress": ["local"] }`. | | `agent_name` | string | `lich` | Wizard label. The TUI banner uses the active theme welcome string, not this field. | | `theme` | string | `lich` | Display theme name. See [Themes](https://github.com/Moikapy/lich/blob/main/README.md#themes). | | `gateway` | object | omitted | Optional. `platforms` (`webhook` \| `telegram` \| `discord` \| `twitch`) and `token_envs` (platform → env-var name). Secrets stay in the environment. | diff --git a/docs/user-guide/library.md b/docs/user-guide/library.md index 25e413a..4652cde 100644 --- a/docs/user-guide/library.md +++ b/docs/user-guide/library.md @@ -140,7 +140,7 @@ const config = { }; ``` -Listed providers form a failover chain tried in order: `rate_limit`/`network` errors retry with backoff (3 attempts) on the current provider before failing over; `auth`, `overflow`, and `bad_request` fail over immediately. The last error is rethrown when all providers fail. +Listed providers form a failover chain tried in order: `rate_limit`/`network` errors retry with backoff (3 attempts) on the current provider before failing over; `auth`, `overflow`, and `bad_request` fail over immediately. The last error is rethrown when all providers fail. An optional `models` block splits the chain by role: `models.chat` sets the main loop's order and `models.compress` the context-compression chain (see the [config reference](cli.md#config-file-reference)). `LICH_ALLOW_SELF_COMMIT` and `LICH_TEST_COMMAND` are process-env knobs, not config fields. See the [CLI environment](cli.md#self-improvement-environment). diff --git a/src/agent/config.ts b/src/agent/config.ts index b8fa84f..b7b66a1 100644 --- a/src/agent/config.ts +++ b/src/agent/config.ts @@ -98,7 +98,7 @@ const providers_schema = z const role_schema = z.array(z.string().min(1)).min(1).optional(); -/** Per-role provider chains by name. Omitted roles use `providers` order. */ +/** Per-role provider chains by name. Omitted `chat` uses `providers` order; omitted `compress` uses the chat chain. */ const models_schema = z .object({ /** Main loop failover order. */ diff --git a/test/model_roles.test.ts b/test/model_roles.test.ts index 9c2acc0..0d24c4a 100644 --- a/test/model_roles.test.ts +++ b/test/model_roles.test.ts @@ -149,6 +149,28 @@ describe("Agent model roles", () => { expect(hits).toEqual(["b"]); }); + it("sends compression to models.compress and turns to models.chat", async () => { + const work_dir = await make_temp_dir(); + const hits: string[] = []; + const pad = "x".repeat(400); + const agent = create_agent({ + providers: [config_with("main", hits), config_with("cheap", hits)], + models: { chat: ["main"], compress: ["cheap"] }, + context_budget_tokens: 100, + max_turns: 1, + work_dir, + session_dir: path.join(work_dir, "sessions"), + log_level: "error", + }); + const history: Message[] = Array.from({ length: 11 }, (_unused, index) => ({ + role: "user" as const, + content: `note ${index} ${pad}`, + })); + const run = await agent.run({ input: "go", history }); + expect(hits).toEqual(["cheap", "main"]); + expect(run.messages.some((message) => message.content?.includes("from cheap") === true)).toBe(true); + }); + it("keeps providers order without a models block", async () => { const work_dir = await make_temp_dir(); const hits: string[] = []; From b054cc76274bb8f1c094f465171aebcc5a272c7f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:20:47 +0000 Subject: [PATCH 08/43] feat(plugins): entry settings, host model access and before_llm_call hook Plugin entries may be { path, settings?, models? } beside a bare path. Hooks and plugin tools receive the frozen settings and models.chat(role, ...), which refuses roles the entry was not granted. A new before_llm_call hook may return a note that is capped, sent as a trailing system message on that one main-loop call, and never saved to history; throwing hooks fail open. Part of #149. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 4 + docs/user-guide/cli.md | 2 +- docs/user-guide/plugins.md | 36 +++++- src/agent/agent.ts | 48 ++++++-- src/agent/config.ts | 30 ++++- src/agent/loop.ts | 22 +++- src/plugins/hooks.ts | 55 +++++++-- src/plugins/loader.ts | 15 ++- src/plugins/types.ts | 41 +++++++ src/tools/types.ts | 5 + test/plugin_host.test.ts | 226 +++++++++++++++++++++++++++++++++++++ 11 files changed, 459 insertions(+), 25 deletions(-) create mode 100644 test/plugin_host.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index a2bac24..7a12a7e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased +- Plugin entries may be `{ path, settings?, models? }`. Hooks and plugin tools + get the frozen `settings` and `models.chat(role, …)`, which refuses roles not + granted. New `before_llm_call` hook can add a capped note to one model call; + it fails open (#149). - Optional `models` config block assigns provider chains per role: `chat` for the main loop and `compress` for context compression (falls back to `chat` on failure). Without it, behavior is unchanged (#149). diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index cedd961..957e514 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -134,7 +134,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid | `agent_name` | string | `lich` | Wizard label. The TUI banner uses the active theme welcome string, not this field. | | `theme` | string | `lich` | Display theme name. See [Themes](https://github.com/Moikapy/lich/blob/main/README.md#themes). | | `gateway` | object | omitted | Optional. `platforms` (`webhook` \| `telegram` \| `discord` \| `twitch`) and `token_envs` (platform → env-var name). Secrets stay in the environment. | -| `plugins` | string array | `[]` | Module paths relative to `work_dir` or absolute. Bare `lich`, one-shot, chat, tui, and gateway load them through `create_agent_with_plugins`. `run_agent` does too. `create_agent` does not. See the [plugins guide](plugins.md). | +| `plugins` | array | `[]` | Module paths relative to `work_dir` or absolute, or `{path, settings?, models?}` objects (free-form `settings`; granted model roles, default none). Bare `lich`, one-shot, chat, tui, and gateway load them through `create_agent_with_plugins`. `run_agent` does too. `create_agent` does not. See the [plugins guide](plugins.md). | | `mcp_servers` | object | omitted | Optional. Closed record of named servers. Each entry is stdio `{command, args, env?}` or loopback http `{url}`. `enabled` defaults to false. Unknown keys are rejected. See the [Redot guide](redot.md). | | `system_prompt` | string | built-in | Replaces the default system prompt. | | `max_turns` | int >= 1 | `25` | Turn budget per run. | diff --git a/docs/user-guide/plugins.md b/docs/user-guide/plugins.md index da7dbdc..b2e168c 100644 --- a/docs/user-guide/plugins.md +++ b/docs/user-guide/plugins.md @@ -40,6 +40,17 @@ Point your config at it and restart: Entry paths are relative to `work_dir` (or absolute). Restarting the agent reloads plugins — there is no hot reload. +An entry can also be an object with free-form `settings` and the model roles the plugin may call: + +```json +"plugins": [ + "./.lich/plugins/my-plugin.ts", + { "path": "./.lich/plugins/lane.mjs", "settings": { "mode": "shadow", "threshold": 0.75 }, "models": ["compress"] } +] +``` + +`models` lists roles from the [`models` config](cli.md#config-file-reference) (`chat`, `compress`); it defaults to none. Unknown keys and roles are rejected. + ## Hook reference All hooks are awaited. Hook errors are logged as warnings and skipped — a broken hook never breaks the run. @@ -48,10 +59,33 @@ All hooks are awaited. Hook errors are logged as warnings and skipped — a brok | --- | --- | --- | | `before_tool_call` | `(info: {tool_name, args}, ctx) => {block?: boolean, reason?: string} \| void` | Runs before each tool call in plugin registration order. Return `{block: true, reason}` to veto. | | `after_tool_call` | `(info: {tool_name, args, result_summary, ok, error?}, ctx) => void` | Runs after each tool call with a 300-char summary plus structured `ok`/`error`. | +| `before_llm_call` | `(info: {turn, messages}, ctx) => {note?: string} \| void` | Runs before each main-loop model call (not compression). A returned `note` is capped at 2000 chars, prefixed `[plugin ]`, and sent as a trailing system message on that one call only; it is never saved to history. | | `on_run_start` | `(info: {input_chars}, ctx) => void` | Runs once before the conversation loop starts. | | `on_run_end` | `(info: {stopped_reason, turns_used}, ctx) => void` | Runs once after the loop ends with the outcome. | -`ctx` is `{work_dir, state?}` — the agent's working directory plus that plugin's per-run bag. +`ctx` is `{work_dir, state?, settings?, models?}`: + +- `state`: that plugin's per-run bag. +- `settings`: that plugin's frozen `settings` from its config entry (`{}` for a bare path). +- `models.chat(role, messages, options?)`: calls the host's provider chain for `role`. A role not granted in the entry's `models` is refused with an error. `compress` uses the `chat` chain when no `models.compress` is configured. + +Plugin tools get the same `settings` and `models` on their `ToolContext`. + +```mjs +export default { + name: "lane", + hooks: { + async before_llm_call(info, ctx) { + if (ctx.settings.mode !== "act") return; + const last = info.messages.at(-1); + const reply = await ctx.models.chat("compress", [ + { role: "user", content: `One line: what is the user asking?\n\n${last?.content ?? ""}` }, + ]); + return { note: `Request summary: ${reply.message.content}` }; + }, + }, +}; +``` ## Tool authoring diff --git a/src/agent/agent.ts b/src/agent/agent.ts index 47dd825..15bf182 100644 --- a/src/agent/agent.ts +++ b/src/agent/agent.ts @@ -11,12 +11,12 @@ import { ToolRegistry } from "../tools/registry.js"; import { HookedToolRunner } from "../plugins/hooks.js"; import { gatekeeper_plugin } from "../plugins/builtin/gatekeeper.plugin.js"; import { load_plugins, plugin_errors_summary, type LoadedPlugin } from "../plugins/loader.js"; -import type { HookContext, Plugin } from "../plugins/types.js"; -import type { Message, Usage } from "../providers/types.js"; +import type { HookContext, ModelRole, Plugin, PluginAccess, PluginModels } from "../plugins/types.js"; +import type { ChatOptions, Message, Usage } from "../providers/types.js"; import { ProviderRouter } from "../providers/router.js"; import { create_session_recorder, type SessionRecorder } from "../session/recorder.js"; import { open_session, type SessionHandle } from "../session/store.js"; -import type { ToolContext } from "../tools/types.js"; +import type { Tool, ToolContext } from "../tools/types.js"; import type { AgentConfig } from "./config.js"; import { parse_agent_config } from "./config.js"; import { randomUUID } from "node:crypto"; @@ -86,14 +86,27 @@ function collect_usage(total: Usage): (event: AgentEventBody) => void { } /** Register plugin tools onto the final registry; duplicates warn and skip. */ -function register_plugin_tools(registry: ToolRegistry, plugins: readonly LoadedPlugin[]): void { +/** Plugin tool whose context also carries the owning plugin's settings and model access. */ +function with_plugin_access(tool: Tool, access: PluginAccess): Tool { + return { + ...tool, + execute: (args, context) => tool.execute(args, { ...context, ...access }), + }; +} + +function register_plugin_tools( + registry: ToolRegistry, + plugins: readonly LoadedPlugin[], + access: ReadonlyMap, +): void { for (const loaded of plugins) { + const plugin_access = access.get(loaded.plugin); for (const tool of loaded.plugin.tools ?? []) { if (registry.has(tool.name) === true) { logger.warn(`plugin ${loaded.plugin.name} tool ${tool.name} already registered; skipping`); continue; } - registry.register(tool); + registry.register(plugin_access === undefined ? tool : with_plugin_access(tool, plugin_access)); logger.info(`plugin ${loaded.plugin.name} registered tool ${tool.name}`); } } @@ -147,7 +160,14 @@ export class Agent { const allow_self_commit = process.env["LICH_ALLOW_SELF_COMMIT"] === "1"; const gatekeeper = gatekeeper_plugin(allow_self_commit); const gatekeeper_loaded: LoadedPlugin = { plugin: gatekeeper, entry: "builtin:gatekeeper" }; - register_plugin_tools(this.registry, [gatekeeper_loaded, ...plugins]); + const access = new Map(); + for (const loaded of plugins) { + access.set(loaded.plugin, { + settings: loaded.settings ?? Object.freeze({}), + models: this.plugin_models(loaded.plugin.name, loaded.models ?? []), + }); + } + register_plugin_tools(this.registry, [gatekeeper_loaded, ...plugins], access); const base_executor = new ToolExecutor(this.registry, { work_dir: config.work_dir, env: tool_env(config), @@ -155,7 +175,7 @@ export class Agent { // Tools and hooks share one synthetic LoadedPlugin so the gate is live. const hooked = hooked_plugins_of([gatekeeper_loaded, ...plugins]); if (hooked.length > 0) { - this.hook_runner = new HookedToolRunner(base_executor, hooked); + this.hook_runner = new HookedToolRunner(base_executor, hooked, access); this.executor = this.hook_runner; } else { this.hook_runner = undefined; @@ -285,11 +305,25 @@ export class Agent { this.mcp_sessions = await attach_enabled_mcp_tools(this.registry, this.config, this.mcp_runtime); } + /** Model access for one plugin: granted roles only; `compress` uses the chat chain when unset. */ + private plugin_models(plugin_name: string, granted: readonly ModelRole[]): PluginModels { + return Object.freeze({ + chat: async (role: ModelRole, messages: readonly Message[], options?: ChatOptions) => { + if (granted.includes(role) === false) { + throw new Error(`plugin "${plugin_name}" was not granted model role "${role}"`); + } + const router = role === "compress" ? (this.compress_router ?? this.router) : this.router; + return await router.chat_with_failover(messages, [], options); + }, + }); + } + /** Per-run deps: the built-once ToolContext threads through every tool execution. */ private loop_deps(tool_context: ToolContext, emitter: AgentEmitter): LoopDeps { return { chat: (messages, tools, chat_options) => this.router.chat_with_failover(messages, tools, chat_options), compress_chat: this.compress_router?.chat_with_failover.bind(this.compress_router), + before_llm_call: this.hook_runner?.call_before_llm.bind(this.hook_runner), tools: this.executor, definitions: () => this.registry.definitions(), emitter, diff --git a/src/agent/config.ts b/src/agent/config.ts index b7b66a1..963e619 100644 --- a/src/agent/config.ts +++ b/src/agent/config.ts @@ -6,6 +6,7 @@ import { z } from "zod"; import { DEFAULT_GATEWAY_TOOLS_ENABLED } from "../gateway/access.js"; import { ENV_VAR_NAME } from "../gateway/token_env.js"; import { refuse_mcp_entry } from "../mcp/mcp_pin.js"; +import { MODEL_ROLES } from "../plugins/types.js"; import type { ProviderConfig } from "../providers/types.js"; import { set_log_level } from "../util/log.js"; @@ -109,6 +110,19 @@ const models_schema = z .strict() .optional(); +/** Bare module path, or `{ path, settings?, models? }` with free-form settings and granted roles. */ +const plugin_entry_schema = z.union([ + z.string(), + z + .object({ + path: z.string().min(1), + settings: z.record(z.unknown()).optional(), + /** Model roles this plugin may call; default none (fail closed). */ + models: z.array(z.enum(MODEL_ROLES)).optional(), + }) + .strict(), +]); + const agent_config_schema = z .object({ /** Wizard label. The TUI banner uses the active theme welcome string. */ @@ -125,8 +139,8 @@ const agent_config_schema = z compress_threshold: z.number().min(0.1).max(0.95).default(0.8), session_dir: z.string().optional(), terminal_timeout_ms: z.number().int().positive().default(60000), - /** Plugin entry module specifiers, relative to work_dir or absolute. */ - plugins: z.array(z.string()).default([]), + /** Plugin entries: module paths relative to work_dir or absolute, optionally with settings and roles. */ + plugins: z.array(plugin_entry_schema).default([]), gateway: gateway_schema, log_level: z.enum(["debug", "info", "warn", "error"]).default("info"), theme: z.string().min(1).default("lich"), @@ -167,6 +181,16 @@ const agent_config_schema = z export type AgentConfig = z.infer; +function deep_freeze(value: unknown): void { + if (typeof value !== "object" || value === null || Object.isFrozen(value) === true) { + return; + } + Object.freeze(value); + for (const child of Object.values(value)) { + deep_freeze(child); + } +} + function freeze_config(config: AgentConfig): AgentConfig { Object.freeze(config); Object.freeze(config.providers); @@ -180,7 +204,7 @@ function freeze_config(config: AgentConfig): AgentConfig { } Object.freeze(config.plugins); for (const plugin of config.plugins) { - Object.freeze(plugin); + deep_freeze(plugin); } if (Array.isArray(config.tools_enabled) === true) { Object.freeze(config.tools_enabled); diff --git a/src/agent/loop.ts b/src/agent/loop.ts index d935980..4fd6021 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -18,6 +18,7 @@ import type { ToolMessage, Usage, } from "../providers/types.js"; +import type { BeforeLlmCallInfo, HookContext } from "../plugins/types.js"; import { ProviderError } from "../providers/types.js"; import type { ToolContext, ToolResult } from "../tools/types.js"; import { logger } from "../util/log.js"; @@ -37,6 +38,8 @@ export interface LoopDeps { chat: ChatFn; /** Context-compression chat; falls back to `chat` when unset or failing. */ compress_chat?: ChatFn; + /** Plugin before_llm_call fan-out; returned notes apply to that one main-loop call. */ + before_llm_call?: (info: BeforeLlmCallInfo, ctx: HookContext) => Promise; tools: ToolRunner; definitions: () => ToolDefinition[]; emitter?: AgentEmitter; @@ -133,14 +136,29 @@ async function run_tool_calls( return signal_aborted(signal) === true ? "aborted" : "continued"; } +/** History plus any before_llm_call notes as one trailing system message; history itself is untouched. */ +async function messages_for_call(deps: LoopDeps, history: readonly Message[], turn: number): Promise { + if (deps.before_llm_call === undefined) { + return history; + } + const ctx: HookContext = { work_dir: deps.tool_context?.work_dir ?? process.cwd() }; + const notes = await deps.before_llm_call({ turn, messages: Object.freeze([...history]) }, ctx); + if (notes.length === 0) { + return history; + } + return [...history, { role: "system", content: notes.join("\n\n") }]; +} + async function call_chat( deps: LoopDeps, history: readonly Message[], params: LoopParams, emitter: AgentEmitter | undefined, + turn: number, ): Promise { try { - return await deps.chat(history, deps.definitions(), { + const messages = await messages_for_call(deps, history, turn); + return await deps.chat(messages, deps.definitions(), { temperature: params.temperature, max_tokens: params.max_tokens, signal: params.signal, @@ -299,7 +317,7 @@ export async function run_conversation( emitter?.emit({ type: "llm_start", turn }); let result: ChatResult; try { - result = await call_chat(deps, history, params, emitter); + result = await call_chat(deps, history, params, emitter, turn); } catch (error) { if (signal_aborted(params.signal) === true) { return aborted_outcome(history, turn - 1); diff --git a/src/plugins/hooks.ts b/src/plugins/hooks.ts index 93532b4..5c24212 100644 --- a/src/plugins/hooks.ts +++ b/src/plugins/hooks.ts @@ -14,14 +14,19 @@ import type { ToolContext, ToolResult } from "../tools/types.js"; import { logger } from "../util/log.js"; import type { AfterToolCallInfo, + BeforeLlmCallInfo, + BeforeLlmCallResult, BeforeToolCallInfo, BeforeToolCallResult, HookContext, Plugin, + PluginAccess, RunEndInfo, } from "./types.js"; const SUMMARY_MAX_CHARS = 300; +/** Per-note cap for before_llm_call notes. */ +export const NOTE_MAX_CHARS = 2000; /** Structural ToolRunner shape accepted from the wrapped executor. */ export interface WrappedToolRunner { @@ -70,9 +75,9 @@ function reset_plugin_bag(plugin: Plugin): Map { return fresh; } -/** Per-invocation ctx: base plus ONLY this plugin's own sub-map (A8). */ -function with_hook_state(base: HookContext, plugin: Plugin): HookContext { - return { ...base, state: hook_state_for(plugin) }; +/** Per-invocation ctx: base plus ONLY this plugin's own sub-map, settings and model access (A8). */ +function with_hook_state(base: HookContext, plugin: Plugin, access: PluginAccess | undefined): HookContext { + return { ...base, ...access, state: hook_state_for(plugin) }; } function clamp_summary(text: string): string { @@ -87,10 +92,20 @@ function clamp_summary(text: string): string { export class HookedToolRunner { private readonly wrapped: WrappedToolRunner; private readonly hooked_plugins: Plugin[]; + private readonly access: ReadonlyMap; - constructor(wrapped: WrappedToolRunner, plugins: readonly Plugin[]) { + constructor( + wrapped: WrappedToolRunner, + plugins: readonly Plugin[], + access: ReadonlyMap = new Map(), + ) { this.wrapped = wrapped; this.hooked_plugins = plugins.filter((plugin) => plugin.hooks !== undefined); + this.access = access; + } + + private ctx_for(base: HookContext, plugin: Plugin): HookContext { + return with_hook_state(base, plugin, this.access.get(plugin)); } /** @@ -116,7 +131,7 @@ export class HookedToolRunner { continue; } try { - const verdict = (await hook(info, with_hook_state(base, plugin))) as BeforeToolCallResult | undefined; + const verdict = (await hook(info, this.ctx_for(base, plugin))) as BeforeToolCallResult | undefined; if (verdict?.block === true) { return verdict; } @@ -135,7 +150,7 @@ export class HookedToolRunner { continue; } try { - await hook(info, with_hook_state(base, plugin)); + await hook(info, this.ctx_for(base, plugin)); } catch (hook_error) { logger.warn(`plugin after_tool_call hook threw for ${info.tool_name}; continuing`, hook_error); } @@ -173,13 +188,37 @@ export class HookedToolRunner { continue; } try { - await hook(info, with_hook_state(base, plugin)); + await hook(info, this.ctx_for(base, plugin)); } catch (hook_error) { logger.warn("plugin on_run_start hook threw; continuing", hook_error); } } } + /** + * before_llm_call fan-out in plugin order; never throws. Returns each + * plugin's note, capped at NOTE_MAX_CHARS, for this one model call. + */ + async call_before_llm(info: BeforeLlmCallInfo, base: HookContext): Promise { + const notes: string[] = []; + for (const plugin of this.hooked_plugins) { + const hook = plugin.hooks?.before_llm_call; + if (hook === undefined) { + continue; + } + try { + const result = (await hook(info, this.ctx_for(base, plugin))) as BeforeLlmCallResult | undefined; + const note = result?.note; + if (typeof note === "string" && note.length > 0) { + notes.push(`[plugin ${plugin.name}] ${note.slice(0, NOTE_MAX_CHARS)}`); + } + } catch (hook_error) { + logger.warn("plugin before_llm_call hook threw; continuing", hook_error); + } + } + return notes; + } + /** Best-effort on_run_end fan-out used by Agent.run; never throws. */ async call_run_end(info: RunEndInfo, base: HookContext): Promise { for (const plugin of this.hooked_plugins) { @@ -188,7 +227,7 @@ export class HookedToolRunner { continue; } try { - await hook(info, with_hook_state(base, plugin)); + await hook(info, this.ctx_for(base, plugin)); } catch (hook_error) { logger.warn("plugin on_run_end hook threw; continuing", hook_error); } diff --git a/src/plugins/loader.ts b/src/plugins/loader.ts index d3d785b..a876fbf 100644 --- a/src/plugins/loader.ts +++ b/src/plugins/loader.ts @@ -5,12 +5,16 @@ */ import { pathToFileURL } from "node:url"; import path from "node:path"; -import type { Plugin } from "./types.js"; +import type { ModelRole, Plugin, PluginEntry } from "./types.js"; /** A successfully loaded plugin plus the entry path it came from. */ export interface LoadedPlugin { plugin: Plugin; entry: string; + /** Frozen `settings` from the config entry; absent for bare-path entries. */ + settings?: Readonly>; + /** Model roles granted in the config entry; absent means none. */ + models?: readonly ModelRole[]; } /** One failed entry: the specifier and why it failed. */ @@ -85,18 +89,23 @@ async function load_one_entry(entry: string, base_dir: string): Promise { const plugins: LoadedPlugin[] = []; const errors: PluginLoadError[] = []; const seen = new Set(); - for (const entry of entries) { + for (const raw of entries) { + const entry = typeof raw === "string" ? raw : raw.path; if (entry.length === 0) { continue; } try { const loaded = await load_one_entry(entry, base_dir); + if (typeof raw !== "string") { + loaded.settings = Object.freeze({ ...raw.settings }); + loaded.models = Object.freeze([...(raw.models ?? [])]); + } if (is_builtin_collision(loaded.plugin.name) === true) { errors.push({ entry, error_message: `builtin_plugin_name_collision: ${loaded.plugin.name}` }); continue; diff --git a/src/plugins/types.ts b/src/plugins/types.ts index 9df5a87..5f68edf 100644 --- a/src/plugins/types.ts +++ b/src/plugins/types.ts @@ -5,11 +5,36 @@ * (in the case of before_tool_call) veto tool executions; tools merge into the * agent registry after builtin filtering. */ +import type { ChatOptions, ChatResult, Message } from "../providers/types.js"; import type { Tool } from "../tools/types.js"; +/** Model roles a plugin entry may be granted (`models` in config). */ +export const MODEL_ROLES = ["chat", "compress"] as const; +export type ModelRole = (typeof MODEL_ROLES)[number]; + +/** One `plugins` config entry: a bare module path, or a path with settings and granted roles. */ +export type PluginEntry = + | string + | { path: string; settings?: Record; models?: readonly ModelRole[] }; + +/** Host-owned model access for one plugin; roles not granted in config are refused. */ +export interface PluginModels { + chat(role: ModelRole, messages: readonly Message[], options?: ChatOptions): Promise; +} + +/** What the host hands one plugin: its frozen config settings and model access. */ +export interface PluginAccess { + settings: Readonly>; + models: PluginModels; +} + /** Runtime info handed to every hook call. */ export interface HookContext { work_dir: string; + /** This plugin's frozen `settings` from its config entry (`{}` when none). */ + settings?: Readonly>; + /** Model access limited to the roles granted in this plugin's config entry. */ + models?: PluginModels; /** * This plugin's own per-run state sub-map. Every hook invocation receives * a ctx exposing only the invoking plugin's bag; the bag is scoped per @@ -39,6 +64,17 @@ export interface AfterToolCallInfo extends BeforeToolCallInfo { error?: string; } +/** Argument passed to before_llm_call hooks; `messages` is a copy of the history for this call. */ +export interface BeforeLlmCallInfo { + turn: number; + messages: readonly Message[]; +} + +/** A note added to this one model call only, never to the saved history. */ +export interface BeforeLlmCallResult { + note?: string; +} + export interface RunEndInfo { stopped_reason: string; turns_used: number; @@ -51,6 +87,11 @@ export interface PluginHooks { ctx: HookContext, ): Promise | BeforeToolCallResult | void; after_tool_call?(info: AfterToolCallInfo, ctx: HookContext): Promise | void; + /** Runs before each main-loop model call; a returned note is capped and fails open. */ + before_llm_call?( + info: BeforeLlmCallInfo, + ctx: HookContext, + ): Promise | BeforeLlmCallResult | void; on_run_start?(info: { input_chars: number }, ctx: HookContext): Promise | void; on_run_end?(info: RunEndInfo, ctx: HookContext): Promise | void; } diff --git a/src/tools/types.ts b/src/tools/types.ts index 1a39393..cb51f32 100644 --- a/src/tools/types.ts +++ b/src/tools/types.ts @@ -1,3 +1,4 @@ +import type { PluginModels } from "../plugins/types.js"; import type { JsonSchemaObject } from "../util/json_schema.js"; export interface ToolResult { @@ -10,6 +11,10 @@ export interface ToolContext { work_dir: string; env: Record; signal?: AbortSignal; + /** Plugin tools only: the owning plugin's frozen config settings. */ + settings?: Readonly>; + /** Plugin tools only: model access limited to the owning plugin's granted roles. */ + models?: PluginModels; } export interface Tool { diff --git a/test/plugin_host.test.ts b/test/plugin_host.test.ts new file mode 100644 index 0000000..b0665fc --- /dev/null +++ b/test/plugin_host.test.ts @@ -0,0 +1,226 @@ +/** + * #149 PR B: plugin entries with settings and granted model roles, host model + * access on HookContext/ToolContext, and the before_llm_call hook. + */ +import { mkdir, mkdtemp, rm } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { afterAll, describe, expect, it } from "vitest"; +import { Agent } from "../src/agent/agent.js"; +import { parse_agent_config } from "../src/agent/config.js"; +import { NOTE_MAX_CHARS } from "../src/plugins/hooks.js"; +import { load_plugins, type LoadedPlugin } from "../src/plugins/loader.js"; +import type { HookContext, Plugin } from "../src/plugins/types.js"; +import type { ToolContext } from "../src/tools/types.js"; +import { TMP_BASE } from "./helpers/tmp_base.js"; + +const FIXTURES = fileURLToPath(new URL("./fixtures/plugins/", import.meta.url)); +const temp_dirs: string[] = []; + +afterAll(async () => { + for (const dir of temp_dirs) { + await rm(dir, { recursive: true, force: true }); + } +}); + +async function make_temp_dir(): Promise { + await mkdir(TMP_BASE, { recursive: true }); + const dir = await mkdtemp(path.join(TMP_BASE, "plugin-host-")); + temp_dirs.push(dir); + return dir; +} + +interface SentMessage { + role: string; + content: string; +} + +/** Provider fetch that records each request's messages and replies from a script. */ +function recording_fetch(reply: (call: number) => Record): { + fetch_fn: typeof fetch; + requests: SentMessage[][]; +} { + const requests: SentMessage[][] = []; + const fetch_fn: typeof fetch = async (_url, init) => { + const body = JSON.parse(String(init?.body)) as { messages: SentMessage[] }; + requests.push(body.messages); + const message = reply(requests.length); + const payload = { + model: "m", + choices: [{ message, finish_reason: message["tool_calls"] === undefined ? "stop" : "tool_calls" }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }; + return new Response(JSON.stringify(payload), { status: 200 }); + }; + return { fetch_fn, requests }; +} + +function text(content: string): Record { + return { role: "assistant", content }; +} + +function tool_call(name: string): Record { + return { + role: "assistant", + content: "", + tool_calls: [{ id: "t1", type: "function", function: { name, arguments: "{}" } }], + }; +} + +async function make_agent( + fetch_fn: typeof fetch, + loaded: LoadedPlugin[], + extra: Record = {}, +): Promise { + const work_dir = await make_temp_dir(); + const config = parse_agent_config({ + providers: [{ kind: "openai_compat", name: "mock", model: "m", api_key: "k", base_url: "http://mock.local/v1", fetch_fn }], + work_dir, + session_dir: path.join(work_dir, "sessions"), + log_level: "error", + ...extra, + }); + return new Agent(config, loaded); +} + +describe("plugin entries in config", () => { + it("accepts bare paths and objects, deep-freezing settings", () => { + const config = parse_agent_config({ + providers: [{ kind: "openai_compat", name: "a", model: "m" }], + plugins: ["./bare.mjs", { path: "./lane.mjs", settings: { mode: "shadow", nested: { k: 1 } }, models: ["compress"] }], + }); + const entry = config.plugins[1]; + expect(typeof entry).toBe("object"); + if (typeof entry === "object") { + expect(Object.isFrozen(entry.settings)).toBe(true); + expect(Object.isFrozen(entry.settings?.["nested"])).toBe(true); + expect(entry.models).toEqual(["compress"]); + } + }); + + it("rejects unknown roles and unknown keys", () => { + const providers = [{ kind: "openai_compat", name: "a", model: "m" }]; + expect(() => parse_agent_config({ providers, plugins: [{ path: "./p.mjs", models: ["embed"] }] })).toThrow(); + expect(() => parse_agent_config({ providers, plugins: [{ path: "./p.mjs", optoins: {} }] })).toThrow(); + }); + + it("loader carries settings and roles for object entries only", async () => { + const { plugins, errors } = await load_plugins( + ["good.plugin.ts", { path: "named.plugin.ts", settings: { threshold: 0.75 }, models: ["chat"] }], + FIXTURES, + ); + expect(errors).toEqual([]); + expect(plugins[0]?.settings).toBeUndefined(); + expect(plugins[0]?.models).toBeUndefined(); + expect(plugins[1]?.entry).toBe("named.plugin.ts"); + expect(plugins[1]?.settings).toEqual({ threshold: 0.75 }); + expect(Object.isFrozen(plugins[1]?.settings)).toBe(true); + expect(plugins[1]?.models).toEqual(["chat"]); + }); +}); + +describe("plugin settings and model access", () => { + it("hands hooks frozen settings and refuses ungranted roles", async () => { + const seen: HookContext[] = []; + const errors: string[] = []; + const plugin: Plugin = { + name: "lane", + hooks: { + on_run_start: async (_info, ctx) => { + seen.push(ctx); + await ctx.models?.chat("chat", [{ role: "user", content: "x" }]).catch((error: Error) => { + errors.push(error.message); + }); + }, + }, + }; + const { fetch_fn } = recording_fetch(() => text("done")); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane", settings: Object.freeze({ mode: "shadow" }) }]); + await agent.run({ input: "go" }); + expect(seen[0]?.settings).toEqual({ mode: "shadow" }); + expect(Object.isFrozen(seen[0]?.settings)).toBe(true); + expect(errors).toEqual(['plugin "lane" was not granted model role "chat"']); + }); + + it("lets a granted role call the host chain, compress falling back to chat", async () => { + const answers: string[] = []; + const plugin: Plugin = { + name: "lane", + hooks: { + on_run_start: async (_info, ctx) => { + const result = await ctx.models?.chat("compress", [{ role: "user", content: "classify" }]); + answers.push(result?.message.content ?? ""); + }, + }, + }; + const { fetch_fn, requests } = recording_fetch((call) => text(call === 1 ? "side answer" : "done")); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane", models: ["compress"] }]); + await agent.run({ input: "go" }); + expect(answers).toEqual(["side answer"]); + expect(requests[0]?.at(-1)?.content).toBe("classify"); + }); + + it("gives plugin tools their own settings and model access", async () => { + const contexts: ToolContext[] = []; + const plugin: Plugin = { + name: "lane", + tools: [ + { + name: "lane_probe", + description: "probe", + parameters: { type: "object", properties: {} }, + execute: async (_args, context) => { + contexts.push(context); + return { ok: true, output: "probed" }; + }, + }, + ], + }; + const { fetch_fn } = recording_fetch((call) => (call === 1 ? tool_call("lane_probe") : text("done"))); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane", settings: Object.freeze({ k: "v" }) }]); + await agent.run({ input: "go" }); + expect(contexts[0]?.settings).toEqual({ k: "v" }); + expect(typeof contexts[0]?.models?.chat).toBe("function"); + expect(contexts[0]?.work_dir.length).toBeGreaterThan(0); + }); +}); + +describe("before_llm_call", () => { + it("adds a capped note to that one call and never to the history", async () => { + let calls = 0; + const plugin: Plugin = { + name: "lane", + hooks: { + before_llm_call: (info) => { + calls += 1; + expect(info.messages.some((message) => message.role === "user")).toBe(true); + return calls === 1 ? { note: "y".repeat(NOTE_MAX_CHARS + 50) } : undefined; + }, + }, + }; + const { fetch_fn, requests } = recording_fetch((call) => (call === 1 ? tool_call("read_file") : text("done"))); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane" }]); + const run = await agent.run({ input: "go" }); + const first_note = requests[0]?.at(-1); + expect(first_note?.role).toBe("system"); + expect(first_note?.content).toBe(`[plugin lane] ${"y".repeat(NOTE_MAX_CHARS)}`); + expect(requests[1]?.some((message) => message.content.startsWith("[plugin lane]"))).toBe(false); + expect(run.messages.some((message) => message.content?.startsWith("[plugin lane]") === true)).toBe(false); + }); + + it("fails open when the hook throws", async () => { + const plugin: Plugin = { + name: "lane", + hooks: { + before_llm_call: () => { + throw new Error("decision model down"); + }, + }, + }; + const { fetch_fn, requests } = recording_fetch(() => text("done")); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane" }]); + const run = await agent.run({ input: "go" }); + expect(run.outcome.final?.content).toBe("done"); + expect(requests[0]?.at(-1)?.role).toBe("user"); + }); +}); From d6b712adcdc64bee531d8c356a45f6270676fcaf Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:27:31 +0000 Subject: [PATCH 09/43] fix(plugins): deep-copy hook messages, honor run abort in models.chat, deep-freeze loader settings Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/agent/agent.ts | 16 +++++++++++-- src/agent/config.ts | 11 +-------- src/agent/loop.ts | 3 ++- src/plugins/loader.ts | 5 +++- src/util/freeze.ts | 10 ++++++++ test/plugin_host.test.ts | 51 ++++++++++++++++++++++++++++++++++++++++ 6 files changed, 82 insertions(+), 14 deletions(-) create mode 100644 src/util/freeze.ts diff --git a/src/agent/agent.ts b/src/agent/agent.ts index 15bf182..c729b39 100644 --- a/src/agent/agent.ts +++ b/src/agent/agent.ts @@ -19,6 +19,7 @@ import { open_session, type SessionHandle } from "../session/store.js"; import type { Tool, ToolContext } from "../tools/types.js"; import type { AgentConfig } from "./config.js"; import { parse_agent_config } from "./config.js"; +import { AsyncLocalStorage } from "node:async_hooks"; import { randomUUID } from "node:crypto"; import type { AgentEvent, AgentEventBody } from "./events.js"; import { AgentEmitter, EnvelopedAgentEmitter } from "./events.js"; @@ -131,6 +132,16 @@ function tool_env(config: AgentConfig): Record { }; } +/** The current run's abort signal, so plugin model calls stop when the run is aborted. */ +const run_signal = new AsyncLocalStorage(); + +function combine_signals(a: AbortSignal | undefined, b: AbortSignal | undefined): AbortSignal | undefined { + if (a === undefined) { + return b; + } + return b === undefined ? a : AbortSignal.any([a, b]); +} + export class Agent { readonly events: EnvelopedAgentEmitter; readonly config: AgentConfig; @@ -185,7 +196,7 @@ export class Agent { async run(options: AgentRunOptions): Promise { await this.attach_mcp_once(); - const body = (): Promise => this.run_body(options); + const body = (): Promise => run_signal.run(options.signal, () => this.run_body(options)); // Per-run ALS scope so concurrent Agent.run calls do not share gatekeeper state (M-6). if (this.hook_runner !== undefined) { return this.hook_runner.run_scope(body); @@ -313,7 +324,8 @@ export class Agent { throw new Error(`plugin "${plugin_name}" was not granted model role "${role}"`); } const router = role === "compress" ? (this.compress_router ?? this.router) : this.router; - return await router.chat_with_failover(messages, [], options); + const signal = combine_signals(run_signal.getStore(), options?.signal); + return await router.chat_with_failover(messages, [], { ...options, signal }); }, }); } diff --git a/src/agent/config.ts b/src/agent/config.ts index 963e619..8ae2ee2 100644 --- a/src/agent/config.ts +++ b/src/agent/config.ts @@ -8,6 +8,7 @@ import { ENV_VAR_NAME } from "../gateway/token_env.js"; import { refuse_mcp_entry } from "../mcp/mcp_pin.js"; import { MODEL_ROLES } from "../plugins/types.js"; import type { ProviderConfig } from "../providers/types.js"; +import { deep_freeze } from "../util/freeze.js"; import { set_log_level } from "../util/log.js"; const gateway_allowlist = z.record(z.string(), z.array(z.string())).default({}); @@ -181,16 +182,6 @@ const agent_config_schema = z export type AgentConfig = z.infer; -function deep_freeze(value: unknown): void { - if (typeof value !== "object" || value === null || Object.isFrozen(value) === true) { - return; - } - Object.freeze(value); - for (const child of Object.values(value)) { - deep_freeze(child); - } -} - function freeze_config(config: AgentConfig): AgentConfig { Object.freeze(config); Object.freeze(config.providers); diff --git a/src/agent/loop.ts b/src/agent/loop.ts index 4fd6021..f6f5503 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -142,7 +142,8 @@ async function messages_for_call(deps: LoopDeps, history: readonly Message[], tu return history; } const ctx: HookContext = { work_dir: deps.tool_context?.work_dir ?? process.cwd() }; - const notes = await deps.before_llm_call({ turn, messages: Object.freeze([...history]) }, ctx); + // Deep copy: hooks must not reach live history objects or the session transcript. + const notes = await deps.before_llm_call({ turn, messages: structuredClone(history) }, ctx); if (notes.length === 0) { return history; } diff --git a/src/plugins/loader.ts b/src/plugins/loader.ts index a876fbf..33c424f 100644 --- a/src/plugins/loader.ts +++ b/src/plugins/loader.ts @@ -5,6 +5,7 @@ */ import { pathToFileURL } from "node:url"; import path from "node:path"; +import { deep_freeze } from "../util/freeze.js"; import type { ModelRole, Plugin, PluginEntry } from "./types.js"; /** A successfully loaded plugin plus the entry path it came from. */ @@ -103,7 +104,9 @@ export async function load_plugins( try { const loaded = await load_one_entry(entry, base_dir); if (typeof raw !== "string") { - loaded.settings = Object.freeze({ ...raw.settings }); + const settings = structuredClone({ ...raw.settings }); + deep_freeze(settings); + loaded.settings = settings; loaded.models = Object.freeze([...(raw.models ?? [])]); } if (is_builtin_collision(loaded.plugin.name) === true) { diff --git a/src/util/freeze.ts b/src/util/freeze.ts new file mode 100644 index 0000000..a556c4b --- /dev/null +++ b/src/util/freeze.ts @@ -0,0 +1,10 @@ +/** Recursively freeze plain objects and arrays in place. */ +export function deep_freeze(value: unknown): void { + if (typeof value !== "object" || value === null || Object.isFrozen(value) === true) { + return; + } + Object.freeze(value); + for (const child of Object.values(value)) { + deep_freeze(child); + } +} diff --git a/test/plugin_host.test.ts b/test/plugin_host.test.ts index b0665fc..e772093 100644 --- a/test/plugin_host.test.ts +++ b/test/plugin_host.test.ts @@ -117,6 +117,14 @@ describe("plugin entries in config", () => { expect(Object.isFrozen(plugins[1]?.settings)).toBe(true); expect(plugins[1]?.models).toEqual(["chat"]); }); + + it("loader deep-freezes a copy of object-entry settings", async () => { + const nested = { level: 1 }; + const { plugins } = await load_plugins([{ path: "named.plugin.ts", settings: { nested } }], FIXTURES); + const settings = plugins[0]?.settings as { nested: { level: number } }; + expect(Object.isFrozen(settings.nested)).toBe(true); + expect(Object.isFrozen(nested)).toBe(false); + }); }); describe("plugin settings and model access", () => { @@ -208,6 +216,49 @@ describe("before_llm_call", () => { expect(run.messages.some((message) => message.content?.startsWith("[plugin lane]") === true)).toBe(false); }); + it("hands hooks a deep copy, so mutations never reach history", async () => { + const plugin: Plugin = { + name: "lane", + hooks: { + before_llm_call: (info) => { + const first = info.messages.find((message) => message.role === "user") as { content: string }; + first.content = "tampered"; + }, + }, + }; + const { fetch_fn, requests } = recording_fetch(() => text("done")); + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane" }]); + const run = await agent.run({ input: "go" }); + expect(run.messages.some((message) => message.content === "tampered")).toBe(false); + expect(requests[0]?.some((message) => message.content === "tampered")).toBe(false); + }); + + it("aborts plugin model calls with the run signal", async () => { + const controller = new AbortController(); + const seen: boolean[] = []; + let rejected = 0; + const plugin: Plugin = { + name: "lane", + hooks: { + before_llm_call: async (_info, ctx) => { + controller.abort(); + await ctx.models?.chat("chat", [{ role: "user", content: "x" }]).catch(() => { + rejected += 1; + }); + }, + }, + }; + const fetch_fn: typeof fetch = async (_url, init) => { + seen.push(init?.signal?.aborted === true); + throw Object.assign(new Error("aborted"), { name: "AbortError" }); + }; + const agent = await make_agent(fetch_fn, [{ plugin, entry: "lane", models: ["chat"] }]); + await agent.run({ input: "go", signal: controller.signal }); + expect(rejected).toBe(1); + // The router refuses an already-aborted signal before any request goes out. + expect(seen).toEqual([]); + }); + it("fails open when the hook throws", async () => { const plugin: Plugin = { name: "lane", From 2b01b6d9b24ee4f9346870ed3844a0aee281769a Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:32:41 +0000 Subject: [PATCH 10/43] wiki: ingest Ollama System One/Clef check; record model roles and plugin host features Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/comparisons/lich-vs-hermes.md | 8 +-- wiki/concepts/decision-models.md | 17 +++--- wiki/entities/lich-plugins-and-hooks.md | 12 +++- wiki/entities/lich-providers.md | 8 ++- wiki/log.md | 3 + .../2026-10-03-ollama-systemone-clef.md | 61 +++++++++++++++++++ 6 files changed, 92 insertions(+), 17 deletions(-) create mode 100644 wiki/raw/audits/2026-10-03-ollama-systemone-clef.md diff --git a/wiki/comparisons/lich-vs-hermes.md b/wiki/comparisons/lich-vs-hermes.md index d599658..9940074 100644 --- a/wiki/comparisons/lich-vs-hermes.md +++ b/wiki/comparisons/lich-vs-hermes.md @@ -1,7 +1,7 @@ --- title: Lich vs Hermes created: 2026-09-23 -updated: 2026-10-02 +updated: 2026-10-03 type: comparison tags: [hermes, ecosystem, research] sources: [raw/audits/2026-09-23-hermes-vs-lich.md, raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#113", "#149"] @@ -38,9 +38,9 @@ This compares Lich with [[hermes-agent]], each feature checked against the Lich | Cron | full scheduler + agent tool | none | **skip for now**: sim world ticks later | | File checkpoints | shadow git | none | **port later**: cheap, useful for coding games | | UIs | CLI, Ink TUI, Electron, web, ACP | CLI, Ink TUI, library | in progress ([[ossuary]]) | -| Model roles (2026-10-02) | `fallback_model` chain + per-task `auxiliary.` provider/model | one failover chain for everything, compression included | **adapt**: `models.chat` / `models.compress` naming providers (#149) | -| Plugin settings + model access (2026-10-02) | `plugins.entries..settings`, host-owned `ctx.llm` | module paths only, no model access | **port**: `{ path, settings, models }` + granted roles (#149) | -| LLM-call hooks (2026-10-02) | `pre_llm_call`, `post_llm_call`, `llm_request` middleware, aux-call hooks | tool and run hooks only | **port**: `before_llm_call` first (#149) | +| Model roles (2026-10-02) | `fallback_model` chain + per-task `auxiliary.` provider/model | `models.chat` / `models.compress` provider chains; compress falls back to chat (#154) | **adapted** (#154) | +| Plugin settings + model access (2026-10-02) | `plugins.entries..settings`, host-owned `ctx.llm` | `{ path, settings, models }` entries; `ctx.models.chat` limited to granted roles (#157) | **ported** (#157) | +| LLM-call hooks (2026-10-02) | `pre_llm_call`, `post_llm_call`, `llm_request` middleware, aux-call hooks | `before_llm_call` with one-call capped notes (#157); no post-call or middleware hooks | **ported** first hook (#157) | | Decision models (2026-10-02) | none in core; 15 community Jev plugins (+4 with Jev backends) | none | **adapt**: plugin-owned client, shadow first ([[decision-models]], #148) | ## What Lich does better diff --git a/wiki/concepts/decision-models.md b/wiki/concepts/decision-models.md index aef2e93..d959970 100644 --- a/wiki/concepts/decision-models.md +++ b/wiki/concepts/decision-models.md @@ -1,21 +1,21 @@ --- title: Decision models (typed, calibrated decisions) created: 2026-10-02 -updated: 2026-10-02 +updated: 2026-10-03 type: concept tags: [providers, performance, research, games, security] -sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#148", "#149"] +sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, raw/audits/2026-10-03-ollama-systemone-clef.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#148", "#149"] confidence: medium --- # Decision models -**What they are.** A decision model does not chat. It takes state plus typed questions and returns a typed answer with a probability for every option, in one forward pass. The API shape comes from TypeSafe's Jev (`POST /v1/systemone`). Cloudflare's Clef / Clef-flash (Apache 2.0) and Ollama 0.35's local endpoint speak the same API. ^[raw/audits/2026-10-02-decision-models-ollama-research.md] +**What they are.** A decision model does not chat. It takes state plus typed questions and returns a typed answer with a probability for every option, in one forward pass. The API shape comes from TypeSafe's Jev (`POST /v1/systemone`). Cloudflare's Clef / Clef-flash (Apache 2.0) and Ollama's local `/v1/systemone` endpoint (0.35+) serve the same question types; exact Jev wire compatibility is unverified. ^[raw/audits/2026-10-02-decision-models-ollama-research.md] ^[raw/audits/2026-10-03-ollama-systemone-clef.md] | Question type | Returns | Typical use | |---|---|---| | `noul` | probability of yes | gates, filters, guards | -| `choice` | one labelled option, per-option probabilities, confidence | routing, action pick, intent | +| `choice` | one labelled option, per-option probabilities, confidence | routing, action pick, intent (Ollama: 2-26 options) | | `score` | position on an ordered rubric, plus distribution | urgency, risk, quality | **Why Lich cares.** The [[tao-loop]] spends a full LLM call on every narrow decision. Decision models answer those in roughly 40-500 ms (Clef-flash fastest), locally or hosted, with a confidence value that tells the caller when to fall back to the LLM. That is the same latency problem [[action-terminal-mode]] attacks for game NPCs. @@ -30,7 +30,7 @@ confidence: medium ## How Hermes does it -[[hermes-agent]] core has no decision-model support. Fourteen community plugins add it on generic host features: per-plugin settings, host-owned model access, and hooks before tool and LLM calls. They route skills (`pre_llm_call`), gate tools (`pre_tool_call`, shadow by default), review approvals (failures escalate), pick per-turn model and effort (request middleware), and skip idle cron runs. ^[raw/audits/2026-10-02-hermes-models-memory-decisions.md] +[[hermes-agent]] core has no decision-model support. Fifteen dedicated community plugins (plus four with Jev backends) add it on generic host features: per-plugin settings, host-owned model access, and hooks before tool and LLM calls. They route skills (`pre_llm_call`), gate tools (`pre_tool_call`, shadow by default), review approvals (failures escalate), pick per-turn model and effort (request middleware), and skip idle cron runs. ^[raw/audits/2026-10-02-hermes-models-memory-decisions.md] ## Rules of use @@ -41,7 +41,8 @@ confidence: medium ## Getting the models -- **Ollama (local or ollama.com):** 0.35 adds `/v1/systemone` with `nimble` (9B) and `tev1` (4B, 0.8B). Lich's [[lich-providers]] Ollama client only speaks `/api/chat` (`src/providers/ollama.ts:155-158@e9bdd82`), so the decision client lives in the plugin, not core, as in Hermes. The plugin gets its endpoint from per-plugin settings (#149). -- **Cloudflare Workers AI:** `clef` (27B) and `clef-flash` (9B). Clef is not in the Ollama library; Ollaya runs it locally. +- **Ollama, local only.** `/v1/systemone` shipped in 0.35.0 with `nimble` and `tev1`; 0.35.1 adds Cloudflare's `clef` (27B) and `clef-flash` (9B), which also take images. The server refuses cloud models on this endpoint, so ollama.com hosting does not apply. Limits: 1-64 questions, 2-26 criteria, 64 KiB body without images. `confidence` is 1 minus normalised entropy, not a calibration guarantee. `clef-flash` is broken on `/v1/systemone` in 0.35.1 (ollama/ollama#18769); prefer `clef:27b` or `nimble` until it is fixed. ^[raw/audits/2026-10-03-ollama-systemone-clef.md] +- **Cloudflare Workers AI:** hosted `clef` and `clef-flash`. +- **Lich side.** The [[lich-providers]] Ollama client only builds `/api/chat` URLs (`src/providers/ollama.ts:155-158@e9bdd82`), and a chat call would lose the probabilities. The decision client therefore lives in the plugin (#148), as in Hermes. Since #157 a plugin entry carries its endpoint and thresholds in `settings`, can call a granted host role through `ctx.models.chat`, and can add a one-call note through `before_llm_call` ([[lich-plugins-and-hooks]]). -Numbers above come from search summaries, not the primary pages (blocked during research); re-verify before relying on them. +Model sizes and latency figures still come from search summaries; re-verify before relying on them. diff --git a/wiki/entities/lich-plugins-and-hooks.md b/wiki/entities/lich-plugins-and-hooks.md index f177966..ab33ae1 100644 --- a/wiki/entities/lich-plugins-and-hooks.md +++ b/wiki/entities/lich-plugins-and-hooks.md @@ -1,10 +1,10 @@ --- title: Lich plugins, hooks and the gatekeeper created: 2026-09-23 -updated: 2026-10-02 +updated: 2026-10-03 type: entity tags: [plugins, security, runtime] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149"] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149", "#157"] confidence: high --- @@ -17,7 +17,13 @@ A plugin is `{name, tools?, hooks?}`, loaded from paths listed in `config.plugin - **`before_tool_call`** can veto. A veto becomes a tool error whose text starts with `blocked_by_plugin:`. - **`after_tool_call`**, **`on_run_start`** and **`on_run_end`**. -There is **no hook that sees the prompt or the messages**. A plugin therefore cannot inject memory, skills or world state before an LLM call. Adding `before_llm_call` and `build_system_prompt` hooks is item 7 of #113 §2c. Hermes gets the same effect with prompt tiers; see [[prompt-cache-tiers]]. #149 plans a `before_llm_call` hook, per-plugin `settings`, and model calls limited to granted roles, matching what [[hermes-agent]]'s decision and memory plugins rely on. +Since #157 (`1567638`) plugins get three host features that [[hermes-agent]]'s decision and memory plugins rely on: + +- **Entry objects.** A `plugins` entry is a bare path or a strict `{ path, settings?, models? }` (`src/agent/config.ts:115@1567638`). Settings are deep-frozen; `models` lists granted roles and defaults to none. +- **Context.** `HookContext` and `ToolContext` carry that plugin's `settings` and `models.chat(role, …)` (`src/plugins/types.ts:32-37@1567638`). An ungranted role is refused; the call honours the run's abort signal (`src/agent/agent.ts:136,320@1567638`). Plugin tools get the same fields through a wrapper at registration. +- **`before_llm_call`.** Runs before each main-loop model call with a deep copy of the history (`src/agent/loop.ts:140@1567638`). A returned note is capped at 2000 chars, labelled with the plugin name, and sent as one trailing system message on that call only; it is never saved. Throwing hooks fail open (`src/plugins/hooks.ts:202@1567638`). + +There is still no `build_system_prompt` hook (item 7 of #113 §2c); notes are per call, not a stable prompt tier. Hermes gets that effect with prompt tiers; see [[prompt-cache-tiers]]. Hook state was a module-global WeakMap in v0.8.0. It is **per run via AsyncLocalStorage on v0.9.0**, which fixes the case where concurrent runs clobbered each other's state. diff --git a/wiki/entities/lich-providers.md b/wiki/entities/lich-providers.md index 9a1fdce..f7424b8 100644 --- a/wiki/entities/lich-providers.md +++ b/wiki/entities/lich-providers.md @@ -1,10 +1,10 @@ --- title: Lich providers and failover created: 2026-09-23 -updated: 2026-10-02 +updated: 2026-10-03 type: entity tags: [providers, runtime, performance] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-decision-models-ollama-research.md, "#148", "#152"] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-10-02-decision-models-ollama-research.md, "#148", "#152", "#154"] confidence: high --- @@ -14,6 +14,10 @@ confidence: high - `rate_limit` and `network` errors get 3 attempts with deterministic backoff. - `auth`, `overflow` and `bad_request` errors move to the next provider immediately. +## Model roles + +Since #154, an optional `models` block names provider chains per role (`src/agent/config.ts:104@1567638`). `models.chat` sets the main loop's failover order; `models.compress` routes context compression and falls back to the chat chain when it fails (`src/agent/loop.ts:206@1567638`). `ProviderRouter.for_role` builds each chain over the shared client cache (`src/providers/router.ts:32@1567638`). Without `models`, behavior is unchanged. Plugins reach these roles only when granted ([[lich-plugins-and-hooks]]). + ## Strengths - The errors are classified correctly. diff --git a/wiki/log.md b/wiki/log.md index be3d6a7..ac4a709 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -31,3 +31,6 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-02 ingest | Hermes upstream at bed0d535: per-task auxiliary models and fallback chain, memory-plugin embedders, 14 community Jev plugins and the host features they use | raw/audits/2026-10-02-hermes-models-memory-decisions.md - 2026-10-02 update | Hermes model roles, memory and decision plugins; comparison rows for model roles, plugin settings, LLM-call hooks and decision models; decision models stay plugin-owned with shadow mode first; #148/#149 revised to match | entities/hermes-agent.md, comparisons/lich-vs-hermes.md, concepts/decision-models.md, entities/lich-plugins-and-hooks.md, index.md - 2026-10-02 update | Fix Hermes Jev plugin count (15 dedicated plus 4 with Jev backends; raw source left as captured); providers page now reflects merged #152 (qwen3:8b default, Ollama cloud documented) | entities/hermes-agent.md, comparisons/lich-vs-hermes.md, entities/lich-providers.md +- 2026-10-03 ingest | Ollama System One and Cloudflare Clef from the ollama/ollama repo at 42e911bc: Clef in 0.35.1, local only, 2-26 criteria, clef-flash bug #18769 | raw/audits/2026-10-03-ollama-systemone-clef.md +- 2026-10-03 update | Decision models: correct Clef availability, local-only endpoint, option cap; plugin path now uses #157 host features | concepts/decision-models.md +- 2026-10-03 update | Model roles (#154) and plugin entries, settings, model access, before_llm_call (#157) pinned at 1567638; Hermes comparison rows marked done | entities/lich-providers.md, entities/lich-plugins-and-hooks.md, comparisons/lich-vs-hermes.md diff --git a/wiki/raw/audits/2026-10-03-ollama-systemone-clef.md b/wiki/raw/audits/2026-10-03-ollama-systemone-clef.md new file mode 100644 index 0000000..628087b --- /dev/null +++ b/wiki/raw/audits/2026-10-03-ollama-systemone-clef.md @@ -0,0 +1,61 @@ +--- +source_url: https://github.com/ollama/ollama/tree/42e911bc +ingested: 2026-10-03 +sha256: 601fcbfed0f92c39031d989f26ac3b3e73c99ee600ca5c7fc098b7d50035ab6d +--- +# Ollama System One and Cloudflare Clef: primary-source check (2026-10-03) + +Research by a Claude Code subagent from a shallow clone of `ollama/ollama` at +`42e911bc` (2026-10-02), GitHub release/issue pages, and web-search summaries. +ollama.com and blog.cloudflare.com were blocked by the egress proxy. + +## Verified (primary source) + +1. Decision models listed: Nimble, Tev1, Clef, Clef Flash + (`docs/capabilities/decision.mdx:16-22`). Clef and Clef Flash need v0.35.1+ + and also accept images. `/v1/systemone` first shipped in v0.35.0 + (`ollama pull nimble`). Release notes v0.35.1 + (github.com/ollama/ollama/releases/tag/v0.35.1) add Clef support and make + decision models report only the `decision` capability. +2. Local only: "Currently available locally" (`decision.mdx:6-8`). The server + rejects cloud model refs with "System One requires a local decision model" + and accepts only GGUF models (`server/routes.go:874-891`). +3. API: route at `server/routes.go:2076`. + - Request: `{model, state (string or JSON), images?: [base64 PNG/JPEG/WebP], + questions: {name: {type, instructions, criteria}}}`. + - Response: `{model, answers: {name: {type, choice, probabilities, + confidence} | {type: "noul", noul}}, usage}` (`decision.mdx:28-113`). + - Types: `choice` (2-26 options), `noul` (`false`/`true` descriptions, + returns P(true)), `score` (2-26 ordered levels, probability-weighted). + - Limits: 1-64 questions (`decision/systemone.go:41`), 2-26 criteria + (`:198`); body 64 KiB without images, 32 MiB with + (`server/routes.go:846-864`). + - `confidence` = 1 - normalised entropy (`systemone.go:247`); not a + calibration guarantee (`decision.mdx:129`). + - Decision models hide other capabilities in show/list + (`server/images.go:162-167`). +4. ollama.com auth: `https://ollama.com/api` and `/v1` with + `Authorization: Bearer $OLLAMA_API_KEY` + (`docs/api/authentication.mdx:11-22`). Not applicable to System One. +5. Open bug github.com/ollama/ollama/issues/18769: on v0.35.1 `clef-flash` + fails on `/v1/systemone` ("non-finite logit" on CUDA, "cannot open model" + on CPU); it works on `/v1/chat/completions`; `clef:27b` works. + +## Search summaries only (unverified) + +- Ollama on X announced `ollama pull clef` / `ollama pull clef-flash`. +- `clef` / `clef:27b` Q4_K_M ~18 GB; `clef:27b-q8_0` ~30 GB; `clef-flash:9b` + Q8_0, 9.1B params, ~9.5-11 GB (ollama.com library tags, modelfit.io). +- Clef fine-tuned from Qwen3.8-27B, Clef Flash from Qwen3.5-9B. +- MLX System One support (PR #18701) reportedly merged 2026-10-03. +- No new latency benchmarks found. + +## Against the 2026-10-02 notes + +- Confirmed: v0.35 `/v1/systemone`; `nimble` and `tev1` exist; noul/choice/score; + ollama.com Bearer auth. +- Contradicted: "Clef is not in the Ollama library" (added in v0.35.1); + "Ollama (local or ollama.com)" for decision models (local only); + "up to 255 options" (Ollama caps criteria at 26). +- Still unverified: `nimble`/`tev1` sizes, the ~91 ms latency figure, exact + Jev wire compatibility, Ollaya. From e010ef8ddce634194b140afc39f0900a3f6272a7 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:36:08 +0000 Subject: [PATCH 11/43] wiki: index one-liners match model roles and plugin host features Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/index.md | 4 ++-- wiki/log.md | 1 + 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/wiki/index.md b/wiki/index.md index 8cb97d2..d448862 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -15,9 +15,9 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) ## Entities: Lich subsystems - [[lich-agent-loop]]: `run_conversation` + `Agent`. A dependency-injected TAO loop. Its P0 gaps are that events aren't scoped to a run, there is no streaming, it has global state, and each agent is heavyweight. -- [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover. They have no streaming, `tool_choice` or cache_control, and ~150 lines of their helpers are duplicated. +- [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover and per-role chains (`models.chat` / `models.compress`, #154). They have no streaming, `tool_choice` or cache_control, and ~150 lines of their helpers are duplicated. - [[lich-tools-and-guardrails]]: builtins, an executor that never throws, and the wards. Known holes: `tools_enabled` doesn't restrict plugin tools, and `terminal` isn't sandboxed. -- [[lich-plugins-and-hooks]]: tool-call hooks with veto, and the gatekeeper's single gated `git_commit`. There are no prompt-level hooks or plugin settings yet (#149), and hooks fail open. +- [[lich-plugins-and-hooks]]: tool-call hooks with veto, the gatekeeper's single gated `git_commit`, and (#157) per-plugin settings, granted model roles and a `before_llm_call` note hook. There is no `build_system_prompt` hook yet, and hooks fail open. - [[lich-sessions]]: JSONL phylacteries used as combat logs. There is no search, and gateway files are supersets of each other. - [[lich-mcp]]: an MCP client and catalog. Redot is a real entry and Godot has none. The code is spread over 21 micro-files. - [[lich-gateway]]: familiars routed into one shared Agent. The per-chat bus and read-only defaults make it a good hub. diff --git a/wiki/log.md b/wiki/log.md index ac4a709..f657ab8 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -34,3 +34,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-03 ingest | Ollama System One and Cloudflare Clef from the ollama/ollama repo at 42e911bc: Clef in 0.35.1, local only, 2-26 criteria, clef-flash bug #18769 | raw/audits/2026-10-03-ollama-systemone-clef.md - 2026-10-03 update | Decision models: correct Clef availability, local-only endpoint, option cap; plugin path now uses #157 host features | concepts/decision-models.md - 2026-10-03 update | Model roles (#154) and plugin entries, settings, model access, before_llm_call (#157) pinned at 1567638; Hermes comparison rows marked done | entities/lich-providers.md, entities/lich-plugins-and-hooks.md, comparisons/lich-vs-hermes.md +- 2026-10-03 update | Index one-liners for providers and plugins match #154 and #157 | index.md From 4640db2764fdcfe5c7c5573905d2138846114c9c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:38:23 +0000 Subject: [PATCH 12/43] wiki: bump index updated date Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/index.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/wiki/index.md b/wiki/index.md index d448862..59d95e0 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -1,7 +1,7 @@ --- title: Wiki index type: index -updated: 2026-10-02 +updated: 2026-10-03 --- # Lich Wiki: Index From b8266bcc213fae630ca31d02e6a303139eac0abc Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 22:59:49 +0000 Subject: [PATCH 13/43] feat(examples): decision_lane plugin for game_bridge action selection A plugin-owned client for Ollama's local /v1/systemone endpoint asks one choice question per enemy action (and target) before each LLM turn. Shadow mode logs one JSONL line per decision; act mode queues orders through game_bridge's enemy_actions tool and meteor veto when every answer clears the threshold, else the LLM decides. Requests are checked against Ollama's limits before sending; failures fall back and are logged. bench.mjs replays saved snapshots for latency, coverage and agreement with the LLM. Part of #148. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 5 + examples/decision_lane/README.md | 68 +++++ examples/decision_lane/bench.mjs | 106 ++++++++ .../decision_lane/decision_lane.plugin.mjs | 117 +++++++++ examples/decision_lane/round.mjs | 107 ++++++++ examples/decision_lane/systemone.mjs | 99 +++++++ test/decision_lane.test.ts | 242 ++++++++++++++++++ 7 files changed, 744 insertions(+) create mode 100644 examples/decision_lane/README.md create mode 100644 examples/decision_lane/bench.mjs create mode 100644 examples/decision_lane/decision_lane.plugin.mjs create mode 100644 examples/decision_lane/round.mjs create mode 100644 examples/decision_lane/systemone.mjs create mode 100644 test/decision_lane.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 7a12a7e..7557c18 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,11 @@ ## Unreleased +- New `examples/decision_lane/`: a prototype plugin that asks a local Ollama + decision model (`/v1/systemone`) to pick each game_bridge enemy's action + before the LLM turn. Shadow mode (default) only logs; act mode queues orders + through game_bridge's own checks when every answer clears the threshold. + Includes a replay benchmark (#148). - Plugin entries may be `{ path, settings?, models? }`. Hooks and plugin tools get the frozen `settings` and `models.chat(role, …)`, which refuses roles not granted. New `before_llm_call` hook can add a capped note to one model call; diff --git a/examples/decision_lane/README.md b/examples/decision_lane/README.md new file mode 100644 index 0000000..043e48f --- /dev/null +++ b/examples/decision_lane/README.md @@ -0,0 +1,68 @@ +# decision_lane + +Prototype plugin for #148: before each LLM turn, ask a local **decision model** to pick every enemy's action for the current [game_bridge](../game_bridge/README.md) round. A decision model returns one labelled answer per question with a confidence, in one forward pass, instead of generating text. See the wiki page `wiki/concepts/decision-models.md` for background. + +It needs the game_bridge plugin loaded too, and Ollama 0.35 or later with a decision model pulled: + +```sh +ollama pull nimble # or clef:27b (Ollama 0.35.1+) +``` + +```json +{ + "plugins": [ + "./examples/game_bridge/game_bridge.plugin.mjs", + { + "path": "./examples/decision_lane/decision_lane.plugin.mjs", + "settings": { "mode": "shadow", "model": "nimble", "threshold": 0.75 } + } + ] +} +``` + +## How it works + +1. The `before_llm_call` hook reads `.lich/game/state.json`. Without a usable snapshot (a round, at least one hero and one enemy) it does nothing. +2. It decides each round once per run. For every enemy it asks a `choice` question for the action (that enemy's `kit` abilities plus `attack`, `defend`, `flee`) and, when there is more than one hero, a `choice` question for the target. +3. It sends them to `POST {base_url}/v1/systemone` with a 2 s timeout. The request is checked against Ollama's limits before it is sent: 1-64 questions, 2-26 options per question, 64 KiB body. +4. It writes one line to `.lich/game/decisions.jsonl` per decision: mode, model, round, each enemy's picks with confidence, the lowest confidence, latency, the outcome (`shadow`, `acted` or `fallback`) and any fallback reason. + +## Modes + +| `mode` | Effect | +| --- | --- | +| `shadow` (default) | Only logs. Use it to compare the model's picks with the LLM's `enemy_actions` calls before trusting it. | +| `act` | When every answer's confidence is at least `threshold`, queues the round through game_bridge's own `enemy_actions` tool and adds a one-call note telling the LLM the round is handled. Otherwise the LLM turn runs as usual. | + +In `act` mode the orders still go through game_bridge's validation and its meteor veto. A decision-model verdict can only add an order that those checks accept; it never bypasses them. A timeout, HTTP error, bad response, low confidence, veto or rejected order is logged as a fallback and the LLM decides the round. The hook never throws. + +The LLM call itself still happens in `act` mode (hooks cannot skip it); the note keeps that turn short. Skipping the call outright would need a new hook result, which is out of scope for this prototype. + +## Settings + +| Setting | Default | Meaning | +| --- | --- | --- | +| `mode` | `"shadow"` | `shadow` or `act`. | +| `base_url` | `"http://localhost:11434"` | System One endpoint root. | +| `model` | `"nimble"` | Decision model. Avoid `clef-flash` on Ollama 0.35.1 (ollama/ollama#18769); `clef:27b` works. | +| `threshold` | `0.75` | Minimum confidence for every answer before `act` queues orders. | +| `timeout_ms` | `2000` | Request timeout. | +| `log_file` | `"decisions.jsonl"` | File under `.lich/game/`. | +| `api_key_env` | none | Env var holding a Bearer key, for a hosted endpoint that needs one. | + +## What leaves the machine + +Each decision sends the battle snapshot (round, hero ids, enemy ids and kits) and the last user message, cut to 2000 characters, to `base_url`. + +- **Local Ollama (default):** nothing leaves the machine. Ollama serves decision models locally only; it refuses cloud models on `/v1/systemone`. +- **Any other `base_url`:** that host receives the payload above, plus the key from `api_key_env` if set. Only local Ollama has been checked against this wire format. + +## Benchmark + +`bench.mjs` replays saved snapshots through one or more models and reports latency, and, at thresholds 0.6, 0.75 and 0.9, how often the lane would act (coverage) and how often it agrees with the LLM's orders: + +```sh +node examples/decision_lane/bench.mjs cases.jsonl --models tev1:0.8b,nimble +``` + +Each line of `cases.jsonl` is `{"battle": , "request": "...", "llm_actions": []}`. Build it from a shadow run: pair each `state.json` with the `actions` from the matching `orders.jsonl` line the LLM wrote. diff --git a/examples/decision_lane/bench.mjs b/examples/decision_lane/bench.mjs new file mode 100644 index 0000000..4c335bf --- /dev/null +++ b/examples/decision_lane/bench.mjs @@ -0,0 +1,106 @@ +/** + * Replays saved battle snapshots through System One and compares the picks with + * what the LLM chose. Needs a local Ollama 0.35+ with the models pulled. + * + * node examples/decision_lane/bench.mjs cases.jsonl [--base-url URL] [--models tev1:0.8b,nimble] + * + * Each cases.jsonl line: {"battle": , "request"?: "...", + * "llm_actions"?: [{"enemy_id","action","target_ref"}]}. + */ +import { readFile } from "node:fs/promises"; +import { build_questions, normalize_battle, pick_actions } from "./round.mjs"; +import { systemone } from "./systemone.mjs"; + +const THRESHOLDS = [0.6, 0.75, 0.9]; + +function parse_args(argv) { + const args = { file: undefined, base_url: "http://localhost:11434", models: ["tev1:0.8b", "nimble"] }; + for (let index = 0; index < argv.length; index += 1) { + const value = argv[index]; + if (value === "--base-url") { + args.base_url = argv[(index += 1)]; + } else if (value === "--models") { + args.models = argv[(index += 1)].split(","); + } else { + args.file = value; + } + } + return args; +} + +async function read_cases(file) { + const lines = (await readFile(file, "utf8")).split("\n").filter((line) => line.trim().length > 0); + return lines + .map((line) => JSON.parse(line)) + .map((item) => ({ ...item, battle: normalize_battle(item.battle) })) + .filter((item) => item.battle !== undefined); +} + +/** Share of enemies whose action and target both match the LLM's order; undefined without one. */ +function agreement(actions, llm_actions) { + if (Array.isArray(llm_actions) === false || llm_actions.length === 0) { + return undefined; + } + const by_enemy = new Map(llm_actions.map((action) => [action.enemy_id, action])); + const matches = actions.filter((action) => { + const llm = by_enemy.get(action.enemy_id); + return llm !== undefined && llm.action === action.action && llm.target_ref === action.target_ref; + }); + return matches.length / actions.length; +} + +async function run_model(model, cases, base_url) { + const rows = []; + for (const item of cases) { + try { + const { answers, latency_ms } = await systemone({ + base_url, + model, + state: { battle: item.battle, request: item.request ?? "" }, + questions: build_questions(item.battle), + timeout_ms: 10000, + }); + const picked = pick_actions(item.battle, answers); + rows.push({ latency_ms, min_confidence: picked.min_confidence, agree: agreement(picked.actions, item.llm_actions) }); + } catch (error) { + rows.push({ error: error?.reason ?? String(error) }); + } + } + return rows; +} + +function mean(values) { + return values.length === 0 ? undefined : values.reduce((sum, value) => sum + value, 0) / values.length; +} + +function fmt(value, digits = 2) { + return value === undefined ? "-" : value.toFixed(digits); +} + +function report(model, rows) { + const ok = rows.filter((row) => row.error === undefined); + const latencies = ok.map((row) => row.latency_ms).sort((a, b) => a - b); + const p50 = latencies[Math.floor(latencies.length / 2)]; + console.log(`\n${model}: ${ok.length}/${rows.length} ok, p50 ${p50 ?? "-"} ms, mean ${fmt(mean(latencies), 0)} ms`); + console.log("| threshold | coverage | agreement when covered |"); + console.log("| --- | --- | --- |"); + for (const threshold of THRESHOLDS) { + const covered = ok.filter((row) => row.min_confidence >= threshold); + const agree = mean(covered.map((row) => row.agree).filter((value) => value !== undefined)); + console.log(`| ${threshold} | ${fmt(covered.length / Math.max(rows.length, 1))} | ${fmt(agree)} |`); + } + const errors = rows.filter((row) => row.error !== undefined).map((row) => row.error); + if (errors.length > 0) { + console.log(`errors: ${[...new Set(errors)].join(", ")}`); + } +} + +const args = parse_args(process.argv.slice(2)); +if (args.file === undefined) { + console.error("usage: node examples/decision_lane/bench.mjs cases.jsonl [--base-url URL] [--models a,b]"); + process.exit(2); +} +const cases = await read_cases(args.file); +for (const model of args.models) { + report(model, await run_model(model, cases, args.base_url)); +} diff --git a/examples/decision_lane/decision_lane.plugin.mjs b/examples/decision_lane/decision_lane.plugin.mjs new file mode 100644 index 0000000..ed13ffa --- /dev/null +++ b/examples/decision_lane/decision_lane.plugin.mjs @@ -0,0 +1,117 @@ +/** + * Decision lane: asks a local decision model (Ollama System One) to pick each + * enemy's action for the current game_bridge round before the LLM turn. + * + * mode "shadow" (default) only logs. mode "act" queues the orders when every + * answer clears `threshold`; the orders still pass game_bridge's validation and + * meteor veto, so the model's verdict never bypasses a deterministic check. + * Any failure, low confidence, or veto falls back to the normal LLM turn. + */ +import { appendFile, mkdir } from "node:fs/promises"; +import path from "node:path"; +import { enemy_actions_tool } from "../game_bridge/enemy_actions.mjs"; +import { game_file } from "../game_bridge/bridge_paths.mjs"; +import { meteor_veto } from "../game_bridge/meteor_veto.mjs"; +import { build_questions, pick_actions, read_battle } from "./round.mjs"; +import { systemone } from "./systemone.mjs"; + +export const DEFAULT_SETTINGS = Object.freeze({ + mode: "shadow", + base_url: "http://localhost:11434", + model: "nimble", + threshold: 0.75, + timeout_ms: 2000, + log_file: "decisions.jsonl", + api_key_env: undefined, +}); + +/** The last user message is sent as context; capped so the payload stays small. */ +const REQUEST_MAX_CHARS = 2000; + +async function decide_round(info, ctx) { + const settings = { ...DEFAULT_SETTINGS, ...ctx.settings }; + const battle = await read_battle(ctx.work_dir); + if (battle === undefined || ctx.state?.get("decided_round") === battle.round) { + return undefined; + } + ctx.state?.set("decided_round", battle.round); + const record = { ts: new Date().toISOString(), mode: settings.mode, model: settings.model, round: battle.round }; + try { + const { answers, latency_ms } = await systemone({ + base_url: settings.base_url, + model: settings.model, + state: { battle, request: last_user_text(info.messages) }, + questions: build_questions(battle), + timeout_ms: settings.timeout_ms, + api_key: settings.api_key_env === undefined ? undefined : process.env[settings.api_key_env], + }); + const picked = pick_actions(battle, answers); + Object.assign(record, { latency_ms, min_confidence: picked.min_confidence, answers: picked.log }); + return await act_on(picked, battle, settings, ctx, record); + } catch (error) { + record.outcome = "fallback"; + record.fallback_reason = typeof error?.reason === "string" ? error.reason : "error"; + return undefined; + } finally { + await write_log(ctx.work_dir, settings.log_file, record); + } +} + +async function act_on(picked, battle, settings, ctx, record) { + if (settings.mode !== "act") { + record.outcome = "shadow"; + return undefined; + } + if (picked.min_confidence < settings.threshold) { + return fallback(record, "low_confidence"); + } + const args = { + round: battle.round, + actions: picked.actions, + rationale: `decision_lane ${settings.model} min_confidence=${picked.min_confidence.toFixed(2)}`, + }; + const veto = await meteor_veto({ tool_name: "enemy_actions", args }); + if (veto?.block === true) { + return fallback(record, `vetoed:${veto.reason}`); + } + const result = await enemy_actions_tool.execute(args, { work_dir: ctx.work_dir, env: {} }); + if (result.ok === false) { + return fallback(record, `order_rejected:${result.error}`); + } + record.outcome = "acted"; + const summary = picked.actions.map((action) => `${action.enemy_id}:${action.action}->${action.target_ref}`).join(", "); + return { + note: `The decision lane already queued round ${battle.round} orders (${summary}). Do not call enemy_actions again for round ${battle.round}; reply briefly.`, + }; +} + +function fallback(record, reason) { + record.outcome = "fallback"; + record.fallback_reason = reason; + return undefined; +} + +function last_user_text(messages) { + const last = [...messages].reverse().find((message) => message.role === "user"); + return typeof last?.content === "string" ? last.content.slice(0, REQUEST_MAX_CHARS) : ""; +} + +/** One JSONL line per decision; a log failure never breaks the run. */ +async function write_log(work_dir, log_file, record) { + try { + const file = game_file(work_dir, log_file); + await mkdir(path.dirname(file), { recursive: true }); + await appendFile(file, `${JSON.stringify(record)}\n`, "utf8"); + } catch { + // Logging is best-effort. + } +} + +const decision_lane = { + name: "decision_lane", + hooks: { + before_llm_call: decide_round, + }, +}; + +export default decision_lane; diff --git a/examples/decision_lane/round.mjs b/examples/decision_lane/round.mjs new file mode 100644 index 0000000..a90717c --- /dev/null +++ b/examples/decision_lane/round.mjs @@ -0,0 +1,107 @@ +/** + * Turns the game_bridge battle snapshot into System One questions and maps the + * answers back to `enemy_actions` orders. + */ +import { readFile } from "node:fs/promises"; +import { STATE_FILE, game_file } from "../game_bridge/bridge_paths.mjs"; +import { DecisionError, SYSTEMONE_LIMITS } from "./systemone.mjs"; + +const BASIC_ACTIONS = Object.freeze({ + attack: "Basic attack on one hero", + defend: "Defend this round", + flee: "Try to flee the battle", +}); + +/** `{ round, heroes: [id], enemies: [{ id, kit: [ability] }] }`, or undefined when unusable. */ +export async function read_battle(work_dir) { + try { + return normalize_battle(JSON.parse(await readFile(game_file(work_dir, STATE_FILE), "utf8"))); + } catch { + return undefined; + } +} + +/** Same shape check as read_battle, for snapshots already in memory (bench replay). */ +export function normalize_battle(parsed) { + if (typeof parsed?.round !== "number" || Number.isFinite(parsed.round) === false) { + return undefined; + } + const heroes = ids_of(parsed.heroes); + const enemies = (Array.isArray(parsed.enemies) ? parsed.enemies : []) + .filter((enemy) => typeof enemy?.id === "string" && enemy.id.length > 0) + .map((enemy) => ({ id: enemy.id, kit: strings_of(enemy.kit) })); + if (heroes.length === 0 || enemies.length === 0) { + return undefined; + } + return { round: parsed.round, heroes, enemies }; +} + +/** One `choice` per enemy for its action, plus one for its target when there is more than one hero. */ +export function build_questions(battle) { + const questions = {}; + battle.enemies.forEach((enemy, index) => { + questions[`e${index}_action`] = { + type: "choice", + instructions: `Which action should enemy "${enemy.id}" take this round?`, + criteria: action_criteria(enemy.kit), + }; + if (battle.heroes.length > 1) { + questions[`e${index}_target`] = { + type: "choice", + instructions: `Which hero should enemy "${enemy.id}" target this round?`, + criteria: Object.fromEntries(battle.heroes.map((id) => [`hero:${id}`, `Hero ${id}`])), + }; + } + }); + if (Object.keys(questions).length > SYSTEMONE_LIMITS.max_questions) { + throw new DecisionError("too_many_enemies"); + } + return questions; +} + +/** Orders plus the lowest confidence across every answer; a missing answer counts as 0. */ +export function pick_actions(battle, answers) { + let min_confidence = 1; + const log = {}; + const actions = battle.enemies.map((enemy, index) => { + const action = read_choice(answers[`e${index}_action`]); + const target = + battle.heroes.length > 1 + ? read_choice(answers[`e${index}_target`]) + : { choice: `hero:${battle.heroes[0]}`, confidence: 1 }; + min_confidence = Math.min(min_confidence, action.confidence, target.confidence); + log[enemy.id] = { action, target }; + return { enemy_id: enemy.id, action: action.choice, target_ref: target.choice }; + }); + return { actions, min_confidence, log }; +} + +function action_criteria(kit) { + const criteria = {}; + for (const ability of kit) { + criteria[ability] = `Use ability ${ability}`; + } + for (const [action, description] of Object.entries(BASIC_ACTIONS)) { + criteria[action] ??= description; + } + if (Object.keys(criteria).length > SYSTEMONE_LIMITS.max_criteria) { + throw new DecisionError("kit_too_large"); + } + return criteria; +} + +function read_choice(answer) { + const choice = typeof answer?.choice === "string" ? answer.choice : ""; + const confidence = typeof answer?.confidence === "number" ? answer.confidence : 0; + return { choice, confidence: choice.length > 0 ? confidence : 0 }; +} + +function ids_of(list) { + return (Array.isArray(list) ? list : []) + .map((item) => item?.id) + .filter((id) => typeof id === "string" && id.length > 0); +} + +function strings_of(list) { + return (Array.isArray(list) ? list : []).filter((item) => typeof item === "string" && item.length > 0); +} diff --git a/examples/decision_lane/systemone.mjs b/examples/decision_lane/systemone.mjs new file mode 100644 index 0000000..675b893 --- /dev/null +++ b/examples/decision_lane/systemone.mjs @@ -0,0 +1,99 @@ +/** + * Minimal client for Ollama's System One decision endpoint (`POST /v1/systemone`). + * Wire format and limits follow the Ollama docs (docs/capabilities/decision.mdx, + * decision/systemone.go at ollama/ollama@42e911bc). Requests are checked before + * they are sent, so an oversized payload never leaves the machine. + */ + +export const SYSTEMONE_LIMITS = Object.freeze({ + max_questions: 64, + min_criteria: 2, + max_criteria: 26, + max_body_bytes: 64 * 1024, +}); + +const DEFAULT_TIMEOUT_MS = 2000; + +/** A failed decision with a short machine-readable `reason` for the decision log. */ +export class DecisionError extends Error { + constructor(reason, message) { + super(message ?? reason); + this.name = "DecisionError"; + this.reason = reason; + } +} + +/** + * Ask typed questions about `state`. Returns `{ answers, latency_ms }`. + * Throws DecisionError on invalid input, timeout, HTTP error, or a bad response. + */ +export async function systemone({ + base_url, + model, + state, + questions, + timeout_ms = DEFAULT_TIMEOUT_MS, + api_key, + fetch_fn = fetch, + signal, +}) { + check_questions(questions); + const body = JSON.stringify({ model, state, questions }); + if (Buffer.byteLength(body, "utf8") > SYSTEMONE_LIMITS.max_body_bytes) { + throw new DecisionError("body_too_large"); + } + const headers = { "Content-Type": "application/json" }; + if (typeof api_key === "string" && api_key.length > 0) { + headers.Authorization = `Bearer ${api_key}`; + } + const timeout = AbortSignal.timeout(timeout_ms); + const started = performance.now(); + let response; + try { + response = await fetch_fn(`${base_url.replace(/\/+$/, "")}/v1/systemone`, { + method: "POST", + headers, + body, + signal: signal === undefined ? timeout : AbortSignal.any([timeout, signal]), + }); + } catch (error) { + if (timeout.aborted === true) { + throw new DecisionError("timeout"); + } + if (signal?.aborted === true) { + throw new DecisionError("aborted"); + } + throw new DecisionError("network", error instanceof Error ? error.message : String(error)); + } + if (response.ok === false) { + throw new DecisionError(`http_${response.status}`); + } + let parsed; + try { + parsed = await response.json(); + } catch { + throw new DecisionError("bad_response"); + } + const answers = parsed?.answers; + if (typeof answers !== "object" || answers === null) { + throw new DecisionError("bad_response"); + } + return { answers, latency_ms: Math.round(performance.now() - started) }; +} + +function check_questions(questions) { + const entries = Object.entries(questions ?? {}); + if (entries.length === 0 || entries.length > SYSTEMONE_LIMITS.max_questions) { + throw new DecisionError("question_count"); + } + for (const [, question] of entries) { + if (question.type === "noul") { + continue; + } + const criteria = question.criteria; + const count = Array.isArray(criteria) ? criteria.length : Object.keys(criteria ?? {}).length; + if (count < SYSTEMONE_LIMITS.min_criteria || count > SYSTEMONE_LIMITS.max_criteria) { + throw new DecisionError("criteria_count"); + } + } +} diff --git a/test/decision_lane.test.ts b/test/decision_lane.test.ts new file mode 100644 index 0000000..149c3a9 --- /dev/null +++ b/test/decision_lane.test.ts @@ -0,0 +1,242 @@ +/** + * decision_lane example (#148 part B): System One client limits and wire shape, + * plus the plugin's shadow / act / fallback paths against a fake local server. + */ +import { createServer, type Server } from "node:http"; +import type { AddressInfo } from "node:net"; +import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { afterAll, beforeAll, describe, expect, it } from "vitest"; +import { load_plugins } from "../src/plugins/loader.js"; +import type { HookContext, Plugin } from "../src/plugins/types.js"; +import type { Message } from "../src/providers/types.js"; +import { TMP_BASE } from "./helpers/tmp_base.js"; +// @ts-expect-error plain .mjs example without type declarations +import { DecisionError, systemone } from "../examples/decision_lane/systemone.mjs"; + +const REPO = fileURLToPath(new URL("..", import.meta.url)); +const PLUGIN = "./examples/decision_lane/decision_lane.plugin.mjs"; +const temp_dirs: string[] = []; + +interface Received { + url: string; + body: { model: string; questions: Record }> }; +} + +let server: Server; +let base_url = ""; +let received: Received[] = []; +let reply: (body: Received["body"]) => { status: number; json: unknown } = () => ({ status: 200, json: {} }); + +beforeAll(async () => { + server = createServer((request, response) => { + let raw = ""; + request.on("data", (chunk) => (raw += chunk)); + request.on("end", () => { + const body = JSON.parse(raw) as Received["body"]; + received.push({ url: request.url ?? "", body }); + const { status, json } = reply(body); + response.writeHead(status, { "Content-Type": "application/json" }); + response.end(JSON.stringify(json)); + }); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + base_url = `http://127.0.0.1:${(server.address() as AddressInfo).port}`; +}); + +afterAll(async () => { + await new Promise((resolve) => server.close(() => resolve())); + for (const dir of temp_dirs) { + await rm(dir, { recursive: true, force: true }); + } +}); + +async function game_dir(state: unknown): Promise { + await mkdir(TMP_BASE, { recursive: true }); + const dir = await mkdtemp(path.join(TMP_BASE, "decision-lane-")); + temp_dirs.push(dir); + if (state !== undefined) { + await mkdir(path.join(dir, ".lich", "game"), { recursive: true }); + await writeFile(path.join(dir, ".lich", "game", "state.json"), JSON.stringify(state)); + } + return dir; +} + +/** Answer every choice with its first criterion at the given confidence. */ +function confident(confidence: number) { + return (body: Received["body"]) => ({ + status: 200, + json: { + model: body.model, + answers: Object.fromEntries( + Object.entries(body.questions).map(([name, question]) => [ + name, + { type: "choice", choice: Object.keys(question.criteria)[0], confidence }, + ]), + ), + }, + }); +} + +async function load_lane(settings: Record): Promise<{ plugin: Plugin; settings: Readonly> }> { + const { plugins, errors } = await load_plugins([{ path: PLUGIN, settings: { base_url, ...settings } }], REPO); + expect(errors).toEqual([]); + const loaded = plugins[0]; + if (loaded === undefined || loaded.settings === undefined) { + throw new Error("plugin did not load"); + } + return { plugin: loaded.plugin, settings: loaded.settings }; +} + +const messages: Message[] = [{ role: "user", content: "Round starts. Decide the enemy turn." }]; + +async function run_hook(work_dir: string, settings: Record, state = new Map()) { + const lane = await load_lane(settings); + const ctx: HookContext = { work_dir, settings: lane.settings, state }; + return await lane.plugin.hooks?.before_llm_call?.({ turn: 1, messages }, ctx); +} + +async function read_lines(file: string): Promise[]> { + try { + return (await readFile(file, "utf8")) + .trim() + .split("\n") + .map((line) => JSON.parse(line) as Record); + } catch { + return []; + } +} + +const battle = { + round: 1, + heroes: [{ id: "hero1" }, { id: "hero2" }], + enemies: [{ id: "goblin", kit: ["slash"] }, { id: "slime" }], +}; + +describe("systemone client", () => { + it("posts the documented shape with an optional Bearer key", async () => { + let seen: { url: string; init: RequestInit | undefined } | undefined; + const fetch_fn: typeof fetch = async (url, init) => { + seen = { url: String(url), init }; + return new Response(JSON.stringify({ answers: { a: { type: "noul", noul: 0.9 } } }), { status: 200 }); + }; + const out = await systemone({ + base_url: "http://h:1/", + model: "nimble", + state: { x: 1 }, + questions: { a: { type: "noul", instructions: "?" } }, + api_key: "k", + fetch_fn, + }); + expect(seen?.url).toBe("http://h:1/v1/systemone"); + expect(JSON.parse(String(seen?.init?.body))).toEqual({ + model: "nimble", + state: { x: 1 }, + questions: { a: { type: "noul", instructions: "?" } }, + }); + expect((seen?.init?.headers as Record)["Authorization"]).toBe("Bearer k"); + expect(out.answers.a.noul).toBe(0.9); + }); + + it("rejects out-of-limit requests before sending", async () => { + const fetch_fn: typeof fetch = async () => { + throw new Error("must not be called"); + }; + const base = { base_url: "http://h", model: "m", state: "s", fetch_fn }; + const choice = (count: number) => ({ + type: "choice", + criteria: Object.fromEntries(Array.from({ length: count }, (_unused, index) => [`o${index}`, "x"])), + }); + await expect(systemone({ ...base, questions: {} })).rejects.toMatchObject({ reason: "question_count" }); + await expect(systemone({ ...base, questions: { q: choice(1) } })).rejects.toMatchObject({ reason: "criteria_count" }); + await expect(systemone({ ...base, questions: { q: choice(27) } })).rejects.toMatchObject({ reason: "criteria_count" }); + await expect( + systemone({ ...base, state: "x".repeat(70 * 1024), questions: { q: choice(2) } }), + ).rejects.toMatchObject({ reason: "body_too_large" }); + }); + + it("maps HTTP errors and timeouts to reasons", async () => { + const questions = { q: { type: "noul", instructions: "?" } }; + const http_404: typeof fetch = async () => new Response("{}", { status: 404 }); + await expect(systemone({ base_url: "http://h", model: "m", state: "s", questions, fetch_fn: http_404 })).rejects.toMatchObject({ + reason: "http_404", + }); + const hang: typeof fetch = (_url, init) => + new Promise((_resolve, reject) => init?.signal?.addEventListener("abort", () => reject(new Error("aborted")))); + const timed = systemone({ base_url: "http://h", model: "m", state: "s", questions, fetch_fn: hang, timeout_ms: 20 }); + await expect(timed).rejects.toBeInstanceOf(DecisionError); + await expect(timed).rejects.toMatchObject({ reason: "timeout" }); + }); +}); + +describe("decision_lane plugin", () => { + it("shadow mode logs the decision and changes nothing", async () => { + received = []; + reply = confident(0.95); + const dir = await game_dir(battle); + const result = await run_hook(dir, {}); + expect(result).toBeUndefined(); + expect(received[0]?.url).toBe("/v1/systemone"); + expect(Object.keys(received[0]?.body.questions ?? {})).toEqual(["e0_action", "e0_target", "e1_action", "e1_target"]); + expect(Object.keys(received[0]?.body.questions["e0_action"]?.criteria ?? {})).toEqual(["slash", "attack", "defend", "flee"]); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log).toHaveLength(1); + expect(log[0]).toMatchObject({ mode: "shadow", model: "nimble", round: 1, outcome: "shadow", min_confidence: 0.95 }); + expect(await read_lines(path.join(dir, ".lich/game/orders.jsonl"))).toEqual([]); + }); + + it("act mode queues validated orders and returns a note when every answer clears the threshold", async () => { + reply = confident(0.9); + const dir = await game_dir(battle); + const result = await run_hook(dir, { mode: "act", threshold: 0.75 }); + expect(result?.note).toContain("already queued round 1 orders"); + const orders = await read_lines(path.join(dir, ".lich/game/orders.jsonl")); + expect(orders[0]).toMatchObject({ + round: 1, + actions: [ + { enemy_id: "goblin", action: "slash", target_ref: "hero:hero1" }, + { enemy_id: "slime", action: "attack", target_ref: "hero:hero1" }, + ], + }); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "acted" }); + }); + + it("falls back to the LLM on low confidence", async () => { + reply = confident(0.5); + const dir = await game_dir(battle); + expect(await run_hook(dir, { mode: "act", threshold: 0.75 })).toBeUndefined(); + expect(await read_lines(path.join(dir, ".lich/game/orders.jsonl"))).toEqual([]); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "low_confidence" }); + }); + + it("never bypasses the meteor veto", async () => { + reply = confident(0.99); + const dir = await game_dir({ round: 1, heroes: [{ id: "hero1" }], enemies: [{ id: "mage", kit: ["meteor"] }] }); + expect(await run_hook(dir, { mode: "act" })).toBeUndefined(); + expect(await read_lines(path.join(dir, ".lich/game/orders.jsonl"))).toEqual([]); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "vetoed:meteor_gates_closed_until_round_3" }); + }); + + it("fails open when the server is unreachable", async () => { + const dir = await game_dir(battle); + expect(await run_hook(dir, { mode: "act", base_url: "http://127.0.0.1:1" })).toBeUndefined(); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "network" }); + }); + + it("skips without a snapshot and decides each round once per run", async () => { + received = []; + reply = confident(0.9); + expect(await run_hook(await game_dir(undefined), {})).toBeUndefined(); + expect(received).toHaveLength(0); + const dir = await game_dir(battle); + const state = new Map(); + await run_hook(dir, {}, state); + await run_hook(dir, {}, state); + expect(received).toHaveLength(1); + }); +}); From bab48508d32375da2936a0db22c70919a62a4012 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 23:05:13 +0000 Subject: [PATCH 14/43] fix(decision_lane): zero off-criteria answers, require numeric threshold, keep log under .lich/game Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- examples/decision_lane/README.md | 4 +-- examples/decision_lane/bench.mjs | 5 ++-- .../decision_lane/decision_lane.plugin.mjs | 12 ++++++--- examples/decision_lane/round.mjs | 16 +++++++----- test/decision_lane.test.ts | 26 +++++++++++++++++++ 5 files changed, 50 insertions(+), 13 deletions(-) diff --git a/examples/decision_lane/README.md b/examples/decision_lane/README.md index 043e48f..070ce67 100644 --- a/examples/decision_lane/README.md +++ b/examples/decision_lane/README.md @@ -34,7 +34,7 @@ ollama pull nimble # or clef:27b (Ollama 0.35.1+) | `shadow` (default) | Only logs. Use it to compare the model's picks with the LLM's `enemy_actions` calls before trusting it. | | `act` | When every answer's confidence is at least `threshold`, queues the round through game_bridge's own `enemy_actions` tool and adds a one-call note telling the LLM the round is handled. Otherwise the LLM turn runs as usual. | -In `act` mode the orders still go through game_bridge's validation and its meteor veto. A decision-model verdict can only add an order that those checks accept; it never bypasses them. A timeout, HTTP error, bad response, low confidence, veto or rejected order is logged as a fallback and the LLM decides the round. The hook never throws. +In `act` mode the orders still go through game_bridge's validation and its meteor veto. A decision-model verdict can only add an order that those checks accept; it never bypasses them. An answer that is not one of the offered options counts as confidence 0. A timeout, HTTP error, bad response, low confidence, non-numeric `threshold`, veto or rejected order is logged as a fallback and the LLM decides the round. The hook never throws. The LLM call itself still happens in `act` mode (hooks cannot skip it); the note keeps that turn short. Skipping the call outright would need a new hook result, which is out of scope for this prototype. @@ -47,7 +47,7 @@ The LLM call itself still happens in `act` mode (hooks cannot skip it); the note | `model` | `"nimble"` | Decision model. Avoid `clef-flash` on Ollama 0.35.1 (ollama/ollama#18769); `clef:27b` works. | | `threshold` | `0.75` | Minimum confidence for every answer before `act` queues orders. | | `timeout_ms` | `2000` | Request timeout. | -| `log_file` | `"decisions.jsonl"` | File under `.lich/game/`. | +| `log_file` | `"decisions.jsonl"` | File name under `.lich/game/`; anything with a directory part falls back to the default. | | `api_key_env` | none | Env var holding a Bearer key, for a hosted endpoint that needs one. | ## What leaves the machine diff --git a/examples/decision_lane/bench.mjs b/examples/decision_lane/bench.mjs index 4c335bf..bf9ccab 100644 --- a/examples/decision_lane/bench.mjs +++ b/examples/decision_lane/bench.mjs @@ -53,14 +53,15 @@ async function run_model(model, cases, base_url) { const rows = []; for (const item of cases) { try { + const questions = build_questions(item.battle); const { answers, latency_ms } = await systemone({ base_url, model, state: { battle: item.battle, request: item.request ?? "" }, - questions: build_questions(item.battle), + questions, timeout_ms: 10000, }); - const picked = pick_actions(item.battle, answers); + const picked = pick_actions(item.battle, questions, answers); rows.push({ latency_ms, min_confidence: picked.min_confidence, agree: agreement(picked.actions, item.llm_actions) }); } catch (error) { rows.push({ error: error?.reason ?? String(error) }); diff --git a/examples/decision_lane/decision_lane.plugin.mjs b/examples/decision_lane/decision_lane.plugin.mjs index ed13ffa..f85d02b 100644 --- a/examples/decision_lane/decision_lane.plugin.mjs +++ b/examples/decision_lane/decision_lane.plugin.mjs @@ -37,15 +37,16 @@ async function decide_round(info, ctx) { ctx.state?.set("decided_round", battle.round); const record = { ts: new Date().toISOString(), mode: settings.mode, model: settings.model, round: battle.round }; try { + const questions = build_questions(battle); const { answers, latency_ms } = await systemone({ base_url: settings.base_url, model: settings.model, state: { battle, request: last_user_text(info.messages) }, - questions: build_questions(battle), + questions, timeout_ms: settings.timeout_ms, api_key: settings.api_key_env === undefined ? undefined : process.env[settings.api_key_env], }); - const picked = pick_actions(battle, answers); + const picked = pick_actions(battle, questions, answers); Object.assign(record, { latency_ms, min_confidence: picked.min_confidence, answers: picked.log }); return await act_on(picked, battle, settings, ctx, record); } catch (error) { @@ -62,6 +63,9 @@ async function act_on(picked, battle, settings, ctx, record) { record.outcome = "shadow"; return undefined; } + if (Number.isFinite(settings.threshold) === false) { + return fallback(record, "invalid_threshold"); + } if (picked.min_confidence < settings.threshold) { return fallback(record, "low_confidence"); } @@ -98,8 +102,10 @@ function last_user_text(messages) { /** One JSONL line per decision; a log failure never breaks the run. */ async function write_log(work_dir, log_file, record) { + // A bare file name only, so the log always stays under .lich/game. + const name = typeof log_file === "string" && log_file === path.basename(log_file) ? log_file : DEFAULT_SETTINGS.log_file; try { - const file = game_file(work_dir, log_file); + const file = game_file(work_dir, name); await mkdir(path.dirname(file), { recursive: true }); await appendFile(file, `${JSON.stringify(record)}\n`, "utf8"); } catch { diff --git a/examples/decision_lane/round.mjs b/examples/decision_lane/round.mjs index a90717c..d537002 100644 --- a/examples/decision_lane/round.mjs +++ b/examples/decision_lane/round.mjs @@ -59,15 +59,18 @@ export function build_questions(battle) { return questions; } -/** Orders plus the lowest confidence across every answer; a missing answer counts as 0. */ -export function pick_actions(battle, answers) { +/** + * Orders plus the lowest confidence across every answer. A missing answer, or a + * choice that was not one of the question's criteria, counts as confidence 0. + */ +export function pick_actions(battle, questions, answers) { let min_confidence = 1; const log = {}; const actions = battle.enemies.map((enemy, index) => { - const action = read_choice(answers[`e${index}_action`]); + const action = read_choice(answers[`e${index}_action`], questions[`e${index}_action`]); const target = battle.heroes.length > 1 - ? read_choice(answers[`e${index}_target`]) + ? read_choice(answers[`e${index}_target`], questions[`e${index}_target`]) : { choice: `hero:${battle.heroes[0]}`, confidence: 1 }; min_confidence = Math.min(min_confidence, action.confidence, target.confidence); log[enemy.id] = { action, target }; @@ -90,10 +93,11 @@ function action_criteria(kit) { return criteria; } -function read_choice(answer) { +function read_choice(answer, question) { const choice = typeof answer?.choice === "string" ? answer.choice : ""; + const offered = Object.hasOwn(question?.criteria ?? {}, choice); const confidence = typeof answer?.confidence === "number" ? answer.confidence : 0; - return { choice, confidence: choice.length > 0 ? confidence : 0 }; + return { choice, confidence: offered === true ? confidence : 0 }; } function ids_of(list) { diff --git a/test/decision_lane.test.ts b/test/decision_lane.test.ts index 149c3a9..f538019 100644 --- a/test/decision_lane.test.ts +++ b/test/decision_lane.test.ts @@ -212,6 +212,32 @@ describe("decision_lane plugin", () => { expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "low_confidence" }); }); + it("treats a choice outside the offered options as confidence 0", async () => { + reply = (body) => ({ + status: 200, + json: { + answers: Object.fromEntries( + Object.keys(body.questions).map((name) => [name, { type: "choice", choice: "enemy:boss", confidence: 0.99 }]), + ), + }, + }); + const dir = await game_dir(battle); + expect(await run_hook(dir, { mode: "act" })).toBeUndefined(); + expect(await read_lines(path.join(dir, ".lich/game/orders.jsonl"))).toEqual([]); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "low_confidence", min_confidence: 0 }); + }); + + it("falls back on a non-numeric threshold and keeps the log under .lich/game", async () => { + reply = confident(0.99); + const dir = await game_dir(battle); + expect(await run_hook(dir, { mode: "act", threshold: "high", log_file: "../escape.jsonl" })).toBeUndefined(); + expect(await read_lines(path.join(dir, ".lich/game/orders.jsonl"))).toEqual([]); + expect(await read_lines(path.join(dir, ".lich/escape.jsonl"))).toEqual([]); + const log = await read_lines(path.join(dir, ".lich/game/decisions.jsonl")); + expect(log[0]).toMatchObject({ outcome: "fallback", fallback_reason: "invalid_threshold" }); + }); + it("never bypasses the meteor veto", async () => { reply = confident(0.99); const dir = await game_dir({ round: 1, heroes: [{ id: "hero1" }], enemies: [{ id: "mage", kit: ["meteor"] }] }); From c59f5d1f3f4ba4a269688900f9abc2a58bf33677 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 01:49:22 +0000 Subject: [PATCH 15/43] wiki: record the decision_lane prototype (#159) Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/concepts/decision-models.md | 6 +++--- wiki/entities/decision-lane-example.md | 28 ++++++++++++++++++++++++++ wiki/entities/game-bridge-example.md | 6 ++++-- wiki/index.md | 3 ++- wiki/log.md | 2 ++ 5 files changed, 39 insertions(+), 6 deletions(-) create mode 100644 wiki/entities/decision-lane-example.md diff --git a/wiki/concepts/decision-models.md b/wiki/concepts/decision-models.md index d959970..a752d99 100644 --- a/wiki/concepts/decision-models.md +++ b/wiki/concepts/decision-models.md @@ -1,10 +1,10 @@ --- title: Decision models (typed, calibrated decisions) created: 2026-10-02 -updated: 2026-10-03 +updated: 2026-10-04 type: concept tags: [providers, performance, research, games, security] -sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, raw/audits/2026-10-03-ollama-systemone-clef.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#148", "#149"] +sources: [raw/audits/2026-10-02-decision-models-ollama-research.md, raw/audits/2026-10-03-ollama-systemone-clef.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#148", "#149", "#159"] confidence: medium --- @@ -22,7 +22,7 @@ confidence: medium ## Where it could fit -1. **Game action selection**, the first prototype in #148: a `choice` over the legal actions in [[game-bridge-example]], with a scripted or LLM fallback below a confidence threshold. +1. **Game action selection**, prototyped as [[decision-lane-example]] (#159): a `choice` per enemy action and target in [[game-bridge-example]], shadow by default, falling back to the LLM below a confidence threshold. Not yet measured. 2. **Gateway triage** in [[lich-gateway]]: "should the agent answer this message?" before a run starts. 3. **Persona routing** in [[persona-orchestrator-example]], from message content. 4. **Tool-list narrowing** before an LLM turn, which cuts prompt tokens. diff --git a/wiki/entities/decision-lane-example.md b/wiki/entities/decision-lane-example.md new file mode 100644 index 0000000..ea3f389 --- /dev/null +++ b/wiki/entities/decision-lane-example.md @@ -0,0 +1,28 @@ +--- +title: decision_lane example (decision model for game_bridge rounds) +created: 2026-10-04 +updated: 2026-10-04 +type: entity +tags: [games, npc, plugins, providers] +sources: [raw/audits/2026-10-03-ollama-systemone-clef.md, "#148", "#159"] +confidence: medium +--- + +# decision_lane example + +`examples/decision_lane/*` (#159, `cbdf58e`) is the first prototype of [[decision-models]] in Lich. A `before_llm_call` hook asks a local Ollama decision model to pick each enemy's action for the current [[game-bridge-example]] round. It is not shipped in the npm package. + +## How it works + +- **Plugin-owned client.** `systemone.mjs` posts to `{base_url}/v1/systemone` with a 2 s timeout and checks Ollama's limits before sending: 1-64 questions, 2-26 criteria, 64 KiB body (`examples/decision_lane/systemone.mjs:8,30@cbdf58e`). Core providers are untouched, as [[0008-idea-agnostic-extensible-harness]] asks. +- **Questions.** One `choice` per enemy for its action (kit plus `attack`/`defend`/`flee`) and one for its target when there is more than one hero (`examples/decision_lane/round.mjs:40@cbdf58e`). An answer outside the offered criteria counts as confidence 0 (`round.mjs:96@cbdf58e`). +- **Modes.** `shadow` (default) only logs. `act` queues orders when every answer clears `threshold`, and returns a one-call note ([[lich-plugins-and-hooks]]). It decides each round once per run (`examples/decision_lane/decision_lane.plugin.mjs:31-61@cbdf58e`). +- **Guard rule.** Act-mode orders go through game_bridge's own `meteor_veto` and `enemy_actions` validation (`decision_lane.plugin.mjs:77,81@cbdf58e`). The verdict can add an order those checks accept; it cannot bypass them. Low confidence, a non-numeric threshold, a veto, a rejected order or any request failure falls back to the LLM, and the hook never throws. +- **Log.** One JSONL line per decision in `.lich/game/decisions.jsonl`: picks, confidences, latency, outcome, fallback reason. The file name must be bare, so it stays under `.lich/game` (`decision_lane.plugin.mjs:104@cbdf58e`). +- **Egress.** With the default local Ollama nothing leaves the machine; Ollama serves decision models locally only ^[raw/audits/2026-10-03-ollama-systemone-clef.md]. Another `base_url` receives the snapshot and the last user message (capped at 2000 chars). + +## Limits + +- **The LLM call still happens in act mode.** Hooks cannot skip it, so act mode saves the decision, not the call; the note keeps that turn short. Skipping it needs a new hook result. This is the same latency problem [[action-terminal-mode]] targets. +- **No measured result yet.** `bench.mjs` replays saved snapshots and reports latency, coverage and agreement with the LLM at thresholds 0.6 / 0.75 / 0.9. #148 stays open until a shadow run and bench numbers (`tev1:0.8b`, `nimble`) are written up here as keep or drop. +- **Model choice.** `clef-flash` is broken on `/v1/systemone` in Ollama 0.35.1 (ollama/ollama#18769); default is `nimble`, and `clef:27b` works. diff --git a/wiki/entities/game-bridge-example.md b/wiki/entities/game-bridge-example.md index 511edff..d6c7965 100644 --- a/wiki/entities/game-bridge-example.md +++ b/wiki/entities/game-bridge-example.md @@ -1,7 +1,7 @@ --- title: game_bridge example (file-bus combat commander) created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-04 type: entity tags: [games, npc, plugins] sources: [raw/audits/2026-09-23-game-surface-audit.md] @@ -27,4 +27,6 @@ confidence: high Deferred in #114, item 4: retire the file bus once [[client-executed-tools]] and the GDScript SDK exist. The proposed replacement for the double LLM call is [[action-terminal-mode]]. Until then, keep the docs, because this is the only working Godot path. -Related: [[godot-and-redot]], [[persona-orchestrator-example]], [[lich-sessions]]. +[[decision-lane-example]] can pick a round's orders with a local decision model before the LLM turn; its act-mode orders still go through `enemy_actions` and the meteor veto. + +Related: [[godot-and-redot]], [[persona-orchestrator-example]], [[decision-lane-example]], [[lich-sessions]]. diff --git a/wiki/index.md b/wiki/index.md index 59d95e0..ee571cf 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -1,7 +1,7 @@ --- title: Wiki index type: index -updated: 2026-10-03 +updated: 2026-10-04 --- # Lich Wiki: Index @@ -30,6 +30,7 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) - [[godot-and-redot]]: the game→Lich direction (webhook plus file bus) and the Lich→editor direction (Redot MCP). `WebSocketPeer` is the path to a GDScript SDK. - [[game-bridge-example]]: the file-bus enemy commander. It's racy, needs 2 LLM calls per decision, and will retire once client tools exist. - [[persona-orchestrator-example]]: one Agent per NPC persona, which collapses to ~20 lines once Profiles and Sessions exist. +- [[decision-lane-example]]: a local decision model picks game_bridge enemy orders before the LLM turn; shadow by default, guarded by game_bridge's checks, not yet measured (#148). - [[roadmap-issues]]: what #113, #114, #79 (closed) and #46/#47 (closed) are each for; north star + verticals + six phases. Status lives on the kanban, not here. ## Concepts diff --git a/wiki/log.md b/wiki/log.md index f657ab8..695d2ce 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -35,3 +35,5 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-03 update | Decision models: correct Clef availability, local-only endpoint, option cap; plugin path now uses #157 host features | concepts/decision-models.md - 2026-10-03 update | Model roles (#154) and plugin entries, settings, model access, before_llm_call (#157) pinned at 1567638; Hermes comparison rows marked done | entities/lich-providers.md, entities/lich-plugins-and-hooks.md, comparisons/lich-vs-hermes.md - 2026-10-03 update | Index one-liners for providers and plugins match #154 and #157 | index.md +- 2026-10-04 create | decision_lane example merged in #159: plugin-owned System One client, shadow/act modes, game_bridge checks still apply, bench pending | entities/decision-lane-example.md, index.md +- 2026-10-04 update | Decision models and game_bridge pages link the decision_lane prototype | concepts/decision-models.md, entities/game-bridge-example.md From ca1c05fa9f63a07a3ec2223073ac76b80d7206bc Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 01:52:26 +0000 Subject: [PATCH 16/43] wiki: decision_lane egress mentions the Bearer key Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/entities/decision-lane-example.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/wiki/entities/decision-lane-example.md b/wiki/entities/decision-lane-example.md index ea3f389..22b5733 100644 --- a/wiki/entities/decision-lane-example.md +++ b/wiki/entities/decision-lane-example.md @@ -19,7 +19,7 @@ confidence: medium - **Modes.** `shadow` (default) only logs. `act` queues orders when every answer clears `threshold`, and returns a one-call note ([[lich-plugins-and-hooks]]). It decides each round once per run (`examples/decision_lane/decision_lane.plugin.mjs:31-61@cbdf58e`). - **Guard rule.** Act-mode orders go through game_bridge's own `meteor_veto` and `enemy_actions` validation (`decision_lane.plugin.mjs:77,81@cbdf58e`). The verdict can add an order those checks accept; it cannot bypass them. Low confidence, a non-numeric threshold, a veto, a rejected order or any request failure falls back to the LLM, and the hook never throws. - **Log.** One JSONL line per decision in `.lich/game/decisions.jsonl`: picks, confidences, latency, outcome, fallback reason. The file name must be bare, so it stays under `.lich/game` (`decision_lane.plugin.mjs:104@cbdf58e`). -- **Egress.** With the default local Ollama nothing leaves the machine; Ollama serves decision models locally only ^[raw/audits/2026-10-03-ollama-systemone-clef.md]. Another `base_url` receives the snapshot and the last user message (capped at 2000 chars). +- **Egress.** With the default local Ollama nothing leaves the machine; Ollama serves decision models locally only ^[raw/audits/2026-10-03-ollama-systemone-clef.md]. Another `base_url` receives the snapshot and the last user message (capped at 2000 chars), plus `Authorization: Bearer ` when `api_key_env` is set (`decision_lane.plugin.mjs:47@cbdf58e`). ## Limits From 2074964f217eb42333ba1d906e1566a4e6f674a0 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 03:54:46 +0000 Subject: [PATCH 17/43] fix(agent): tools_enabled filters plugin tools; guard hooks fail closed; no tools on the last turn - A tools_enabled list (top-level or gateway) now applies to plugin tools, including the gatekeeper's git_commit. The commander persona lists its three game_bridge tools explicitly. - A before_tool_call hook that throws blocks the call (blocked_by_plugin: hook_error) instead of allowing it. - Tool calls on the max_turns turn are not executed; they are closed with a turn_budget_exhausted result so history stays valid. Part of #133. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 9 ++++++ docs/architecture/tools.md | 4 +-- docs/user-guide/cli.md | 2 +- docs/user-guide/gateway.md | 2 +- docs/user-guide/library.md | 4 +-- docs/user-guide/plugins.md | 4 +-- docs/user-guide/redot.md | 2 +- examples/persona_orchestrator/README.md | 6 ++-- examples/persona_orchestrator/personas.ts | 2 +- src/agent/agent.ts | 13 ++++++-- src/agent/loop.ts | 16 ++++++++-- src/plugins/hooks.ts | 9 ++++-- test/cli_plugins.test.ts | 3 +- test/loop.test.ts | 18 ++++++++++++ test/mcp_client.test.ts | 2 +- test/persona_orchestrator.test.ts | 4 +-- test/plugins.test.ts | 36 +++++++++++++++++++++-- test/session_persist.test.ts | 2 +- 18 files changed, 108 insertions(+), 30 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 15a30a2..977eb31 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,15 @@ ## Unreleased +- **Breaking:** a `tools_enabled` list (top-level or `gateway.tools_enabled`) + now filters plugin tools too, including the gatekeeper's `git_commit`. List + each plugin tool you want exposed; `"all"` is unchanged. Previously plugin + tools always registered, so gateway chat users could reach every plugin tool. +- A `before_tool_call` hook that throws now blocks that call + (`blocked_by_plugin: hook_error`) instead of silently allowing it. +- Tools the model requests on the `max_turns` turn no longer run, since the + model never sees their results; they are closed with a + `turn_budget_exhausted` tool result so the history stays valid (#133). - `prompt.abort` cancels a `prompt.submit` already accepted on the same socket but still waiting behind another frame, so a cancel issued during resume, list, or another in-flight call cannot miss that run. diff --git a/docs/architecture/tools.md b/docs/architecture/tools.md index be25797..260e093 100644 --- a/docs/architecture/tools.md +++ b/docs/architecture/tools.md @@ -219,8 +219,8 @@ instructions. See the [plugins guide](../user-guide/plugins.md#skills-and-memory `filter_registry` (src/agent/agent.ts) iterates `base.list()` and registers only allowed names onto a new `ToolRegistry` when `tools_enabled` is a list (`"all"` returns the base registry untouched). Plugin tools, including the - gatekeeper's `git_commit`, register after that filter, so `tools_enabled: []` - still leaves `git_commit`. MCP tools register later, on first `run()`, and + gatekeeper's `git_commit`, register after that filter and are skipped when a + list does not name them. MCP tools register later, on first `run()`, and only when the allowlist is `"all"` or names an `mcp_` tool. For building your own tool, see [extending](./extending.md#add-a-builtin-tool). \ No newline at end of file diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index 957e514..a2a5f7d 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -139,7 +139,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid | `system_prompt` | string | built-in | Replaces the default system prompt. | | `max_turns` | int >= 1 | `25` | Turn budget per run. | | `work_dir` | string | cwd | Root for all file tools; paths outside are rejected. | -| `tools_enabled` | `"all"` or name array | `"all"` | Restrict the builtin registry to these names. `[]` drops builtins and MCP tools and does not connect to MCP servers. Plugin tools still register afterward, including the gatekeeper's `git_commit`. | +| `tools_enabled` | `"all"` or name array | `"all"` | Restrict every tool to these names: builtins, MCP tools and plugin tools, including the gatekeeper's `git_commit`. `[]` exposes no tools and does not connect to MCP servers. | | `temperature` | 0–2 | – | Sampling temperature. | | `max_tokens` | positive int | – | Completion cap. | | `context_budget_tokens` | positive int | `100000` | Estimated budget before compression triggers. | diff --git a/docs/user-guide/gateway.md b/docs/user-guide/gateway.md index e699493..94f5829 100644 --- a/docs/user-guide/gateway.md +++ b/docs/user-guide/gateway.md @@ -152,7 +152,7 @@ The gateway agent ignores top-level `tools_enabled` and uses `gateway.tools_enab `read_file`, `list_dir`, `grep_files`, `fetch_url`, `web_search`, `docs_read`, `docs_search` -Set `"gateway": { "tools_enabled": "all" }` (or an explicit name list) only when you intentionally want writes / `terminal` on chat platforms. +Set `"gateway": { "tools_enabled": "all" }` (or an explicit name list) only when you intentionally want writes / `terminal` on chat platforms. The list applies to plugin tools too: add each plugin tool name (for example `enemy_actions`) that chat users may trigger. ## Running multiple platforms at once diff --git a/docs/user-guide/library.md b/docs/user-guide/library.md index 4652cde..a98f92d 100644 --- a/docs/user-guide/library.md +++ b/docs/user-guide/library.md @@ -47,7 +47,7 @@ const result = await run_agent( ## Agent class -`new Agent(config)` (or `create_agent(raw)`) builds the provider router, registers the builtin tools (filtered by `tools_enabled`) plus the gatekeeper's `git_commit`, and exposes: +`new Agent(config)` (or `create_agent(raw)`) builds the provider router, registers the builtin tools and the gatekeeper's `git_commit` (all filtered by `tools_enabled`), and exposes: | Member | Type | Purpose | | --- | --- | --- | @@ -146,7 +146,7 @@ Listed providers form a failover chain tried in order: `rate_limit`/`network` er ## Custom tool filtering -`tools_enabled` accepts `"all"` (default) or an array of tool names to register; everything else stays unregistered and invisible to the model. MCP tools, when a named server is `enabled`, use the same allowlist and stay off when the list is `[]` (that empty list does not connect). The filter does not apply to plugin tools: they register afterward, including the gatekeeper's `git_commit` (fail-closed unless `LICH_ALLOW_SELF_COMMIT=1`). `[]` strips every builtin and every MCP tool and does not throw. +`tools_enabled` accepts `"all"` (default) or an array of tool names to register; everything else stays unregistered and invisible to the model. MCP tools, when a named server is `enabled`, use the same allowlist and stay off when the list is `[]` (that empty list does not connect). The same list applies to plugin tools, including the gatekeeper's `git_commit` (fail-closed unless `LICH_ALLOW_SELF_COMMIT=1`), so list each plugin tool you want exposed. `[]` exposes no tools and does not throw. `mcp_servers` and `catalog_client_entry` ship in this package (since 0.7.0). See the [Redot guide](redot.md). diff --git a/docs/user-guide/plugins.md b/docs/user-guide/plugins.md index b2e168c..6cd8ccf 100644 --- a/docs/user-guide/plugins.md +++ b/docs/user-guide/plugins.md @@ -53,7 +53,7 @@ An entry can also be an object with free-form `settings` and the model roles the ## Hook reference -All hooks are awaited. Hook errors are logged as warnings and skipped — a broken hook never breaks the run. +All hooks are awaited. A `before_tool_call` hook that throws blocks that call (`blocked_by_plugin: hook_error`) so a broken guard never silently allows it; other hook errors are logged as warnings and skipped. Neither ends the run. | Hook | Signature | Purpose | | --- | --- | --- | @@ -135,7 +135,7 @@ Failures are contained at every layer: - **Broken import** (missing file, syntax error, missing export): the loader records the entry as an error, logs one warning with `plugin_errors_summary`, and starts the agent without that plugin. - **Duplicate plugin names**: later duplicates become error entries; the first registration wins. -- **Throwing hook**: logged as a warning; the run continues as if the hook did not exist. +- **Throwing hook**: a throwing `before_tool_call` blocks that call with `blocked_by_plugin: hook_error`; other hooks are logged as warnings and skipped. The run continues either way. - **Throwing tool**: the executor captures it and returns `{ok: false, error}` to the loop. ## Naming rules diff --git a/docs/user-guide/redot.md b/docs/user-guide/redot.md index 69c5a23..19c9740 100644 --- a/docs/user-guide/redot.md +++ b/docs/user-guide/redot.md @@ -26,7 +26,7 @@ stdio is a local binary you named, plus `args`, plus optional `env`. The process HTTP is `{ "url": "http://127.0.0.1:9/mcp" }`. The host must be `127.0.0.1` or `localhost`. `0.0.0.0` and any other host are refused. There is no remote MCP in v1. -On connect the client sends `initialize`, then `notifications/initialized`, then `tools/list`. `tools/call` runs only when the model invokes a registered tool. Registered names are `mcp__`, so two servers cannot collide. They appear only when that server is `enabled` and `tools_enabled` is `"all"` or lists the prefixed name. `tools_enabled: []` drops them even when the server is enabled, and does not connect. Plugin tools still register, including the gatekeeper's `git_commit`. The commander persona keeps `tools_enabled: []` and the `game_bridge` plugin only — `persona_config` does not copy `mcp_servers`. +On connect the client sends `initialize`, then `notifications/initialized`, then `tools/list`. `tools/call` runs only when the model invokes a registered tool. Registered names are `mcp__`, so two servers cannot collide. They appear only when that server is `enabled` and `tools_enabled` is `"all"` or lists the prefixed name. `tools_enabled: []` drops them even when the server is enabled, and does not connect. A list applies to plugin tools too, so list each plugin tool you want exposed. The commander persona lists only its three `game_bridge` tools — `persona_config` does not copy `mcp_servers`. This client has shipped since 0.7.0 (`lich mcp`, `mcp_servers`; current package 0.8.0). From a clone, use `bun src/cli.ts mcp ...`. diff --git a/examples/persona_orchestrator/README.md b/examples/persona_orchestrator/README.md index 8f55fad..24544f7 100644 --- a/examples/persona_orchestrator/README.md +++ b/examples/persona_orchestrator/README.md @@ -19,7 +19,7 @@ There is no basic-attack fallback and no tunable veto table. The meteor gate is Copy `personas.ts`, `history_queue.ts`, `reply.ts`, and `orchestrator.ts`. In the game repo, load agents with `create_agent_with_plugins` from `@moikapy/lich` (this checkout's `run.ts` imports `../../src/index.js`). -- **Factory.** `persona_config` merges a shared provider list with one persona's `system_prompt`, `tools_enabled`, `max_turns`, `max_tokens`, `context_budget_tokens`, and `plugins`. It does not copy `mcp_servers`. One `create_agent_with_plugins` call per persona, cached by `persona_id`. `tools_enabled` filters builtins and editor MCP tools. Plugin tools register after the filter, so they are not stripped. An empty `tools_enabled` drops MCP tools even when a server is enabled, and does not throw — the model just replies, plus the gatekeeper's `git_commit` (always registered, fail-closed unless `LICH_ALLOW_SELF_COMMIT=1`). Duplicate tool names warn and skip; first wins. +- **Factory.** `persona_config` merges a shared provider list with one persona's `system_prompt`, `tools_enabled`, `max_turns`, `max_tokens`, `context_budget_tokens`, and `plugins`. It does not copy `mcp_servers`. One `create_agent_with_plugins` call per persona, cached by `persona_id`. `tools_enabled` filters every tool by name: builtins, editor MCP tools and plugin tools, including the gatekeeper's `git_commit` (fail-closed unless `LICH_ALLOW_SELF_COMMIT=1`). An empty `tools_enabled` exposes no tools and does not throw — the model just replies. Duplicate tool names warn and skip; first wins. - **Serialization.** `enqueue` is the GatewayBus promise chain, keyed by `chat_id`. Concurrent posts to one conversation cannot interleave history. Different `chat_id`s run concurrently. The plugin itself still makes no concurrency guarantee; one bridge per `chat_id` is the intended pattern. - **History cap.** `cap_history` keeps the newest N messages (default 40). Oldest conversations drop at 200. The loop re-seeds `system_prompt` on every run, so losing a stored system line is safe. A cap of N does not keep the whole battle: the oldest overflow is dropped. Restate facts the digest still needs, or write them with `dungeon_memory_write`. - **Budgets.** Each persona has its own `max_turns`, `max_tokens`, and `context_budget_tokens`. Budget exhaustion returns `stopped_reason: "budget"` and `round_fate` is `game_repo_decides`. The HTTP body is still `{reply, usage}`. This example does not write an order line on that path. @@ -30,8 +30,8 @@ Copy `personas.ts`, `history_queue.ts`, `reply.ts`, and `orchestrator.ts`. In th | `persona_id` | `tools_enabled` | Plugins | What the model sees | | --- | --- | --- | --- | -| `commander` | `[]` | `game_bridge.plugin.mjs` | No builtin file/terminal tools. No `mcp_*` editor tools. Plugin tools `enemy_actions`, `dungeon_memory_read`, `dungeon_memory_write`, plus `git_commit`. | -| `chronicler` | `["read_file"]` | none | Builtin `read_file` only (plus `git_commit`). No game-bridge tools. | +| `commander` | `["enemy_actions", "dungeon_memory_read", "dungeon_memory_write"]` | `game_bridge.plugin.mjs` | Those three plugin tools only. No builtin file/terminal tools, no `mcp_*` editor tools, no `git_commit`. | +| `chronicler` | `["read_file"]` | none | Builtin `read_file` only. No game-bridge tools. | `config.plugins` is loaded only by `create_agent_with_plugins`. `create_agent` does not load plugins. Paths are relative to `work_dir`. If `work_dir` is the game repo, copy `examples/game_bridge/` there and point `plugins` at that copy. Restart to reload; there is no hot reload. diff --git a/examples/persona_orchestrator/personas.ts b/examples/persona_orchestrator/personas.ts index 0a7c4ed..9468d77 100644 --- a/examples/persona_orchestrator/personas.ts +++ b/examples/persona_orchestrator/personas.ts @@ -18,7 +18,7 @@ export const PERSONA_TABLE: readonly PersonaEntry[] = [ { persona_id: "commander", system_prompt: COMMANDER_PROMPT, - tools_enabled: [], + tools_enabled: ["enemy_actions", "dungeon_memory_read", "dungeon_memory_write"], max_turns: 8, max_tokens: 800, context_budget_tokens: 12000, diff --git a/src/agent/agent.ts b/src/agent/agent.ts index c729b39..146b329 100644 --- a/src/agent/agent.ts +++ b/src/agent/agent.ts @@ -86,7 +86,6 @@ function collect_usage(total: Usage): (event: AgentEventBody) => void { }; } -/** Register plugin tools onto the final registry; duplicates warn and skip. */ /** Plugin tool whose context also carries the owning plugin's settings and model access. */ function with_plugin_access(tool: Tool, access: PluginAccess): Tool { return { @@ -95,14 +94,24 @@ function with_plugin_access(tool: Tool, access: PluginAccess): Tool { }; } +/** + * Register plugin tools onto the final registry; duplicates warn and skip. + * A tools_enabled list applies here too, so a plugin tool (gatekeeper's + * git_commit included) is exposed only when listed. + */ function register_plugin_tools( registry: ToolRegistry, plugins: readonly LoadedPlugin[], access: ReadonlyMap, + enabled: "all" | readonly string[], ): void { for (const loaded of plugins) { const plugin_access = access.get(loaded.plugin); for (const tool of loaded.plugin.tools ?? []) { + if (enabled !== "all" && enabled.includes(tool.name) === false) { + logger.info(`plugin ${loaded.plugin.name} tool ${tool.name} not in tools_enabled; skipping`); + continue; + } if (registry.has(tool.name) === true) { logger.warn(`plugin ${loaded.plugin.name} tool ${tool.name} already registered; skipping`); continue; @@ -178,7 +187,7 @@ export class Agent { models: this.plugin_models(loaded.plugin.name, loaded.models ?? []), }); } - register_plugin_tools(this.registry, [gatekeeper_loaded, ...plugins], access); + register_plugin_tools(this.registry, [gatekeeper_loaded, ...plugins], access, config.tools_enabled); const base_executor = new ToolExecutor(this.registry, { work_dir: config.work_dir, env: tool_env(config), diff --git a/src/agent/loop.ts b/src/agent/loop.ts index f555eab..606efdc 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -178,6 +178,13 @@ async function call_chat( } } +/** Close final-turn tool calls with not-run results so the history stays valid for resume. */ +function skip_tool_calls(history: Message[], calls: readonly ToolCall[]): void { + for (const call of calls) { + history.push(tool_message_from_result(call, { ok: false, output: "", error: "turn_budget_exhausted" })); + } +} + function replace_history(history: Message[], next: readonly Message[]): void { history.length = 0; for (const message of next) { @@ -379,13 +386,16 @@ export async function run_conversation( emitter?.emit({ type: "turn_end", turn }); return { messages: history, final: result.message, result, turns_used: turn, stopped_reason: "final" }; } + if (turn === params.max_turns) { + // The model gets no turn to read results, so do not run side effects it cannot see. + skip_tool_calls(history, calls); + break; + } const tool_status = await run_tool_calls(deps, history, turn, calls, emitter, params.signal); if (tool_status === "aborted") { return aborted_outcome(history, turn); } - if (turn < params.max_turns) { - emitter?.emit({ type: "turn_end", turn }); - } + emitter?.emit({ type: "turn_end", turn }); } emitter?.emit({ type: "budget_exhausted", turns_used: params.max_turns }); emitter?.emit({ type: "turn_end", turn: params.max_turns }); diff --git a/src/plugins/hooks.ts b/src/plugins/hooks.ts index 5c24212..a1f5f38 100644 --- a/src/plugins/hooks.ts +++ b/src/plugins/hooks.ts @@ -2,8 +2,9 @@ * HookedToolRunner: wraps the ToolExecutor with plugin hooks. * * before_tool_call hooks run in registration order and may veto a call (first - * blocker wins; the wrapped executor is never called). Hook errors are warned - * and skipped, never fatal. after_tool_call hooks observe the result summary + * blocker wins; the wrapped executor is never called). A before_tool_call hook + * that throws blocks the call (fail closed); other hook errors are warned and + * skipped, never fatal. after_tool_call hooks observe the result summary * plus the executor's structured ok/error fields. Every hook invocation * receives a ctx exposing only its own plugin's state sub-map: bags are keyed * per AsyncLocalStorage run scope (M-6) with a WeakMap fallback for tests that @@ -136,7 +137,9 @@ export class HookedToolRunner { return verdict; } } catch (hook_error) { - logger.warn(`plugin before_tool_call hook threw for ${info.tool_name}; continuing`, hook_error); + // Fail closed: a broken guard must not silently allow the call. + logger.warn(`plugin ${plugin.name} before_tool_call hook threw for ${info.tool_name}; blocking`, hook_error); + return { block: true, reason: "hook_error" }; } } return {}; diff --git a/test/cli_plugins.test.ts b/test/cli_plugins.test.ts index 0dd89ef..8553063 100644 --- a/test/cli_plugins.test.ts +++ b/test/cli_plugins.test.ts @@ -174,7 +174,8 @@ describe("cli plugin load", () => { ? completion_body(tool_call_body("t1", "plugin_echo", { text: calls === 1 ? "hello" : "bus" }), "tool_calls") : completion_body({ role: "assistant", content: "done" }, "stop"); }); - const { agent, bus } = await create_gateway_bus(parse_agent_config(mock_config(work_dir, fetch_fn))); + const config = mock_config(work_dir, fetch_fn, { gateway: { tools_enabled: ["plugin_echo"] } }); + const { agent, bus } = await create_gateway_bus(parse_agent_config(config)); expect(tool_content(await agent.run({ input: "echo" }))).toBe("plugin_echo: hello"); const ended: string[] = []; agent.events.on((event) => { diff --git a/test/loop.test.ts b/test/loop.test.ts index 1e9bb46..9566488 100644 --- a/test/loop.test.ts +++ b/test/loop.test.ts @@ -124,6 +124,24 @@ describe("run_conversation", () => { expect(events.at(-1)?.type).toBe("turn_end"); }); + it("does not run tools requested on the max_turns turn and closes them as not run", async () => { + const emitter = new AgentEmitter(); + const events: AgentEventBody[] = []; + emitter.on((event) => events.push(event)); + const chat: ChatFn = async () => result("", [{ id: "late", name: "noop", args: {} }]); + const { runner, calls } = make_tool_runner("noop output"); + const deps: LoopDeps = { chat, tools: runner, definitions: () => [], emitter }; + + const outcome = await run_conversation(deps, [{ role: "user", content: "go" }], { max_turns: 2 }); + + expect(outcome.stopped_reason).toBe("budget"); + expect(calls).toHaveLength(1); + expect(event_types(events).filter((type) => type === "tool_call_start")).toHaveLength(1); + const last = outcome.messages.at(-1); + expect(last).toMatchObject({ role: "tool", tool_call_id: "late", is_error: true }); + expect(last?.content).toContain("turn_budget_exhausted"); + }); + it("scenario C: abort signal stops the loop before the next LLM call", async () => { const controller = new AbortController(); const emitter = new AgentEmitter(); diff --git a/test/mcp_client.test.ts b/test/mcp_client.test.ts index 98654c3..b6072fa 100644 --- a/test/mcp_client.test.ts +++ b/test/mcp_client.test.ts @@ -425,7 +425,7 @@ describe("mcp tool disablement", () => { throw new Error("missing commander"); } const raw = persona_config({ providers, work_dir }, commander); - expect(raw.tools_enabled).toEqual([]); + expect(raw.tools_enabled).toEqual(["enemy_actions", "dungeon_memory_read", "dungeon_memory_write"]); expect(raw).not.toHaveProperty("mcp_servers"); expect(raw.plugins).toEqual(["./examples/game_bridge/game_bridge.plugin.mjs"]); const seen = await visible_tools({ ...raw, work_dir }, boom_spawn()); diff --git a/test/persona_orchestrator.test.ts b/test/persona_orchestrator.test.ts index fc20d52..8acd721 100644 --- a/test/persona_orchestrator.test.ts +++ b/test/persona_orchestrator.test.ts @@ -126,9 +126,7 @@ describe("persona orchestrator", () => { expect(system_prompt_of(seen[1])).toBe(persona_by_id(PERSONA_TABLE, "chronicler")?.system_prompt); const commander_tools = tool_names(seen[0]); expect(commander_tools.filter((name) => BUILTIN_NAMES.includes(name))).toEqual([]); - expect(commander_tools).toEqual( - expect.arrayContaining(["enemy_actions", "dungeon_memory_read", "dungeon_memory_write", "git_commit"]), - ); + expect([...commander_tools].sort()).toEqual(["dungeon_memory_read", "dungeon_memory_write", "enemy_actions"]); const chronicler_tools = tool_names(seen[1]); expect(chronicler_tools).toContain("read_file"); expect(chronicler_tools).not.toContain("terminal"); diff --git a/test/plugins.test.ts b/test/plugins.test.ts index 9bba4db..5d324b5 100644 --- a/test/plugins.test.ts +++ b/test/plugins.test.ts @@ -172,7 +172,7 @@ describe("HookedToolRunner", () => { expect(log).toEqual(["before:read_file", "after:read_file:tool output"]); }); - it("continues when a hook throws", async () => { + it("blocks the call when before_tool_call throws, and survives a throwing after hook", async () => { const { runner, calls } = spy_runner("still works"); const throwing: PluginHooks = { before_tool_call: async () => { @@ -184,8 +184,13 @@ describe("HookedToolRunner", () => { }; const hooked = new HookedToolRunner(runner, [plugin_with(throwing)]); const result = await hooked.execute("read_file", {}); - expect(result.ok).toBe(true); - expect(result.output).toBe("still works"); + expect(result).toEqual({ ok: false, output: "", error: "blocked_by_plugin: hook_error" }); + expect(calls).toEqual([]); + + const after_only = new HookedToolRunner(runner, [plugin_with({ after_tool_call: throwing.after_tool_call })]); + const passed = await after_only.execute("read_file", {}); + expect(passed.ok).toBe(true); + expect(passed.output).toBe("still works"); expect(calls).toEqual([{ name: "read_file", args: {} }]); }); @@ -338,6 +343,31 @@ describe("HookedToolRunner", () => { }); describe("Agent with plugins", () => { + it("applies a tools_enabled list to plugin tools and git_commit", async () => { + const work_dir = await make_temp_dir(); + const seen: string[][] = []; + const fetch_fn: typeof fetch = async (_url, init) => { + const body = JSON.parse(String(init?.body)) as { tools?: Array<{ function: { name: string } }> }; + seen.push((body.tools ?? []).map((tool) => tool.function.name)); + return new Response(JSON.stringify(completion_body({ role: "assistant", content: "ok" }, "stop")), { status: 200 }); + }; + const listed = await create_agent_with_plugins({ + ...base_config(work_dir, fetch_fn), + tools_enabled: ["read_file", "plugin_echo"], + plugins: [path.join(FIXTURES, "good.plugin.ts")], + }); + await listed.run({ input: "go" }); + expect([...(seen[0] ?? [])].sort()).toEqual(["plugin_echo", "read_file"]); + + const unlisted = await create_agent_with_plugins({ + ...base_config(work_dir, fetch_fn), + tools_enabled: ["read_file"], + plugins: [path.join(FIXTURES, "good.plugin.ts")], + }); + await unlisted.run({ input: "go" }); + expect(seen[1]).toEqual(["read_file"]); + }); + it("registers a plugin tool and the loop can call it end to end", async () => { const work_dir = await make_temp_dir(); const fetch_script = scripted_fetch((call_count) => { diff --git a/test/session_persist.test.ts b/test/session_persist.test.ts index 1236ff9..f875072 100644 --- a/test/session_persist.test.ts +++ b/test/session_persist.test.ts @@ -155,7 +155,7 @@ describe("session run_end", () => { ], work_dir, session_dir: path.join(work_dir, "sessions"), - max_turns: 1, + max_turns: 2, log_level: "error", }); const stop = agent.events.on((event) => { From 4f97ea297a8f0e30c1feff45a616e4560369711b Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 04:00:20 +0000 Subject: [PATCH 18/43] fix(loop): emit tool_call_end for last-turn skipped calls so transcripts persist them Also document the last-turn skip in docs/architecture/agent-loop.md. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/architecture/agent-loop.md | 12 ++++++++---- src/agent/loop.ts | 18 ++++++++++++++---- test/loop.test.ts | 2 ++ test/session_persist.test.ts | 7 +++++++ 4 files changed, 31 insertions(+), 8 deletions(-) diff --git a/docs/architecture/agent-loop.md b/docs/architecture/agent-loop.md index 9b28910..b0d50cd 100644 --- a/docs/architecture/agent-loop.md +++ b/docs/architecture/agent-loop.md @@ -36,8 +36,9 @@ stateDiagram-v2 Aborted : error event, stopped_reason = aborted Aborted --> [*] note right of Final - budget path: after the last allowed turn - budget_exhausted then turn_end, stopped_reason = budget + budget path: tool calls on the last allowed turn are not run + (cancelled tool_call_end each), then budget_exhausted then turn_end, + stopped_reason = budget end note ``` @@ -62,7 +63,10 @@ export interface LoopOutcome { **`max_turns` is an LLM-call budget, not a tool budget.** Each iteration makes exactly one LLM call; the tool executions between turns are free - a turn that calls three tools still consumes one turn. A model that always asks for tools -will run out of budget even though the tools all succeeded. +will run out of budget even though the tools all succeeded. Tool calls the model +requests on the last allowed turn are not executed, since no turn is left to read +their results: each is closed with a `turn_budget_exhausted` tool result and a +`tool_call_end` with `cancelled: true`. **Stopping conditions**, exactly as implemented: @@ -95,7 +99,7 @@ swallowed; handlers may unsubscribe mid-emit (the emitter iterates a snapshot). | `llm_start` | `{ turn }` | Just before the chat call | Exactly one per LLM call | | `llm_end` | `{ turn, result }` | Chat call resolved | Pairs with `llm_start`; never fires if the call throws | | `tool_call_start` | `{ turn, call }` | Before each tool executes | After `llm_end`, sequential per call | -| `tool_call_end` | `{ turn, call, result, cancelled? }` | After that tool resolves, or when abort skips a pending call | Normally pairs with `tool_call_start`; cancelled skips emit `cancelled: true` with no start | +| `tool_call_end` | `{ turn, call, result, cancelled? }` | After that tool resolves, or when abort or the last-turn budget skips a call | Normally pairs with `tool_call_start`; skipped calls emit `cancelled: true` with no start | | `final` | `{ message, result }` | A turn produced no tool calls | At most once per run; only on a real final | | `budget_exhausted` | `{ turns_used }` | Loop exits without a final | Follows the last `tool_call_end` | | `turn_end` | `{ turn }` | Last event of a turn | After `final` **or** after `budget_exhausted` | diff --git a/src/agent/loop.ts b/src/agent/loop.ts index 606efdc..5be46c7 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -178,10 +178,20 @@ async function call_chat( } } -/** Close final-turn tool calls with not-run results so the history stays valid for resume. */ -function skip_tool_calls(history: Message[], calls: readonly ToolCall[]): void { +/** + * Close final-turn tool calls with not-run results so the history stays valid + * for resume; tool_call_end (as on the abort path) lets the recorder persist them. + */ +function skip_tool_calls( + history: Message[], + turn: number, + calls: readonly ToolCall[], + emitter: AgentEmitter | undefined, +): void { + const skipped: ToolResult = { ok: false, output: "", error: "turn_budget_exhausted" }; for (const call of calls) { - history.push(tool_message_from_result(call, { ok: false, output: "", error: "turn_budget_exhausted" })); + history.push(tool_message_from_result(call, skipped)); + emitter?.emit({ type: "tool_call_end", turn, call, result: skipped, cancelled: true }); } } @@ -388,7 +398,7 @@ export async function run_conversation( } if (turn === params.max_turns) { // The model gets no turn to read results, so do not run side effects it cannot see. - skip_tool_calls(history, calls); + skip_tool_calls(history, turn, calls, emitter); break; } const tool_status = await run_tool_calls(deps, history, turn, calls, emitter, params.signal); diff --git a/test/loop.test.ts b/test/loop.test.ts index 9566488..8fde990 100644 --- a/test/loop.test.ts +++ b/test/loop.test.ts @@ -137,6 +137,8 @@ describe("run_conversation", () => { expect(outcome.stopped_reason).toBe("budget"); expect(calls).toHaveLength(1); expect(event_types(events).filter((type) => type === "tool_call_start")).toHaveLength(1); + const skipped = events.filter((event) => event.type === "tool_call_end").at(-1); + expect(skipped).toMatchObject({ type: "tool_call_end", turn: 2, cancelled: true, result: { error: "turn_budget_exhausted" } }); const last = outcome.messages.at(-1); expect(last).toMatchObject({ role: "tool", tool_call_id: "late", is_error: true }); expect(last?.content).toContain("turn_budget_exhausted"); diff --git a/test/session_persist.test.ts b/test/session_persist.test.ts index f875072..f923d5d 100644 --- a/test/session_persist.test.ts +++ b/test/session_persist.test.ts @@ -109,6 +109,13 @@ describe("session run_end", () => { expect(events.some((record) => record.meta?.event === "budget_exhausted")).toBe(true); const run_end = events.find((record) => record.meta?.event === "run_end"); expect(run_end?.meta).toMatchObject({ event: "run_end", stopped_reason: "budget", usage: result.usage_total }); + // Raw JSONL, not read_session_messages (which would fill a missing result itself). + const raw = await readFile(result.session_path ?? "", "utf8"); + const tool_lines = raw + .split("\n") + .filter((line) => line.includes('"role":"tool"')); + expect(tool_lines).toHaveLength(1); + expect(tool_lines[0]).toContain("turn_budget_exhausted"); }); it("appends run_end when the run is aborted before a turn", async () => { From 4840336935aeb83e27d24d06c64e424c10dad9ce Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 04:08:49 +0000 Subject: [PATCH 19/43] wiki: record #161 (plugin-tool allowlist, fail-closed guard hooks, last-turn tool skip) Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/entities/lich-agent-loop.md | 5 +++-- wiki/entities/lich-plugins-and-hooks.md | 10 +++++----- wiki/entities/lich-tools-and-guardrails.md | 7 +++---- wiki/index.md | 4 ++-- wiki/log.md | 1 + 5 files changed, 14 insertions(+), 13 deletions(-) diff --git a/wiki/entities/lich-agent-loop.md b/wiki/entities/lich-agent-loop.md index ecc0457..da79343 100644 --- a/wiki/entities/lich-agent-loop.md +++ b/wiki/entities/lich-agent-loop.md @@ -1,10 +1,10 @@ --- title: Lich agent loop (run_conversation + Agent) created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-04 type: entity tags: [core, events, context] -sources: [raw/audits/2026-09-23-core-engine-audit.md] +sources: [raw/audits/2026-09-23-core-engine-audit.md, "#161"] confidence: high --- @@ -22,6 +22,7 @@ This is the heart of Lich. `run_conversation` (`src/agent/loop.ts@77bc148`, abou 4. push the assistant message 5. no tool calls → `final`; otherwise run the tool calls **sequentially** (`loop.ts:110-121@77bc148`) - **Stop reasons:** `final`, `budget` (default `max_turns` 25), `aborted`. Other provider errors are thrown. +- **Last turn:** tool calls requested on the `max_turns` turn are not run; each gets a `turn_budget_exhausted` result and a cancelled `tool_call_end`, so transcripts stay valid (#161, `src/agent/loop.ts:185,399@029b7e8`). - **Events:** 11 synchronous event types (`events.ts:12-24@77bc148`) with **no run or session id** and no token deltas. See [[event-envelope]] and [[streaming-deltas]]. - **System prompt:** one static string, `config.system_prompt ?? DEFAULT_AGENT_SYSTEM_PROMPT`. Memory and skills are not injected; see [[memory-vs-skills]]. diff --git a/wiki/entities/lich-plugins-and-hooks.md b/wiki/entities/lich-plugins-and-hooks.md index ab33ae1..8278911 100644 --- a/wiki/entities/lich-plugins-and-hooks.md +++ b/wiki/entities/lich-plugins-and-hooks.md @@ -1,10 +1,10 @@ --- title: Lich plugins, hooks and the gatekeeper created: 2026-09-23 -updated: 2026-10-03 +updated: 2026-10-04 type: entity tags: [plugins, security, runtime] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149", "#157"] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, raw/audits/2026-10-02-hermes-models-memory-decisions.md, "#149", "#157", "#161"] confidence: high --- @@ -38,10 +38,10 @@ It also denies certain git operations. It is **not** Hermes-style learning (comp ## Holes -- Hooks fail open when they throw, and have no timeout. -- `git_commit` is always registered, even with `tools_enabled: []`. +- Hooks have no timeout. Since #161 a throwing `before_tool_call` blocks the call (`blocked_by_plugin: hook_error`, `src/plugins/hooks.ts:142@029b7e8`); other hooks still fail open. +- Since #161 a `tools_enabled` list filters plugin tools too, so `git_commit` registers only when listed (`src/agent/agent.ts:111@029b7e8`). - Plugins run in-process with full privileges. -All three have to change before plugins can be trusted in a shipped game ([[embedded-safety-profile]]). +The timeout and in-process privileges still have to change before plugins can be trusted in a shipped game ([[embedded-safety-profile]]). Related: [[lich-tools-and-guardrails]], [[game-bridge-example]]. diff --git a/wiki/entities/lich-tools-and-guardrails.md b/wiki/entities/lich-tools-and-guardrails.md index 2b08291..392eba4 100644 --- a/wiki/entities/lich-tools-and-guardrails.md +++ b/wiki/entities/lich-tools-and-guardrails.md @@ -1,10 +1,10 @@ --- title: Lich tools, executor and guardrails created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-04 type: entity tags: [tools, security, runtime] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, "#161"] confidence: high --- @@ -38,8 +38,7 @@ It also: ## Known holes (open on v0.9.0) -- **`tools_enabled` doesn't restrict plugin tools.** It is applied before plugin tools and `git_commit` are registered. -- **A throwing `before_tool_call` hook lets the call through** (fail-open), and hooks have no timeout. +- **Fixed in #161:** a `tools_enabled` list (including `gateway.tools_enabled`) now filters plugin tools and `git_commit` (`src/agent/agent.ts:111@029b7e8`), and a throwing `before_tool_call` blocks the call. Hooks still have no timeout. - **S-11:** `http_request` always returns `ok:true`; `guard.ts:12` checks `startsWith("..")`; `terminal_timeout_ms` is dead code. - **`docs_read` resolves its root from `process.cwd()`** and memoizes it globally. diff --git a/wiki/index.md b/wiki/index.md index ee571cf..48d6dd9 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -16,8 +16,8 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) - [[lich-agent-loop]]: `run_conversation` + `Agent`. A dependency-injected TAO loop. Its P0 gaps are that events aren't scoped to a run, there is no streaming, it has global state, and each agent is heavyweight. - [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover and per-role chains (`models.chat` / `models.compress`, #154). They have no streaming, `tool_choice` or cache_control, and ~150 lines of their helpers are duplicated. -- [[lich-tools-and-guardrails]]: builtins, an executor that never throws, and the wards. Known holes: `tools_enabled` doesn't restrict plugin tools, and `terminal` isn't sandboxed. -- [[lich-plugins-and-hooks]]: tool-call hooks with veto, the gatekeeper's single gated `git_commit`, and (#157) per-plugin settings, granted model roles and a `before_llm_call` note hook. There is no `build_system_prompt` hook yet, and hooks fail open. +- [[lich-tools-and-guardrails]]: builtins, an executor that never throws, and the wards. Known holes: `terminal` isn't sandboxed and hooks have no timeout; `tools_enabled` covers plugin tools since #161. +- [[lich-plugins-and-hooks]]: tool-call hooks with veto, the gatekeeper's single gated `git_commit`, and (#157) per-plugin settings, granted model roles and a `before_llm_call` note hook. There is no `build_system_prompt` hook yet; a throwing `before_tool_call` blocks the call (#161), other hooks fail open. - [[lich-sessions]]: JSONL phylacteries used as combat logs. There is no search, and gateway files are supersets of each other. - [[lich-mcp]]: an MCP client and catalog. Redot is a real entry and Godot has none. The code is spread over 21 micro-files. - [[lich-gateway]]: familiars routed into one shared Agent. The per-chat bus and read-only defaults make it a good hub. diff --git a/wiki/log.md b/wiki/log.md index 695d2ce..d137507 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -37,3 +37,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-03 update | Index one-liners for providers and plugins match #154 and #157 | index.md - 2026-10-04 create | decision_lane example merged in #159: plugin-owned System One client, shadow/act modes, game_bridge checks still apply, bench pending | entities/decision-lane-example.md, index.md - 2026-10-04 update | Decision models and game_bridge pages link the decision_lane prototype | concepts/decision-models.md, entities/game-bridge-example.md +- 2026-10-04 update | #161 merged: tools_enabled filters plugin tools and git_commit, throwing before_tool_call blocks, last-turn tool calls not run; holes lists and index updated, code pinned at 029b7e8 | entities/lich-plugins-and-hooks.md, entities/lich-tools-and-guardrails.md, entities/lich-agent-loop.md, index.md From 0b3d48595a73f57711f1099d9e43d5915d01c307 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 04:11:46 +0000 Subject: [PATCH 20/43] wiki: move the #161 fixed note out of the open-holes list Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/entities/lich-tools-and-guardrails.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/wiki/entities/lich-tools-and-guardrails.md b/wiki/entities/lich-tools-and-guardrails.md index 392eba4..0377dd8 100644 --- a/wiki/entities/lich-tools-and-guardrails.md +++ b/wiki/entities/lich-tools-and-guardrails.md @@ -36,9 +36,11 @@ It also: - **Network tools:** an SSRF guard blocks private and loopback URLs unless `LICH_ALLOW_PRIVATE_URLS=1`, and redirects are re-checked. - **`terminal` is not sandboxed.** It runs `bash -lc` in `work_dir` with secret env vars scrubbed. The docs say this plainly. +**Fixed in #161:** a `tools_enabled` list (including `gateway.tools_enabled`) now filters plugin tools and `git_commit` (`src/agent/agent.ts:111@029b7e8`), and a throwing `before_tool_call` blocks the call. + ## Known holes (open on v0.9.0) -- **Fixed in #161:** a `tools_enabled` list (including `gateway.tools_enabled`) now filters plugin tools and `git_commit` (`src/agent/agent.ts:111@029b7e8`), and a throwing `before_tool_call` blocks the call. Hooks still have no timeout. +- **Hooks have no timeout.** - **S-11:** `http_request` always returns `ok:true`; `guard.ts:12` checks `startsWith("..")`; `terminal_timeout_ms` is dead code. - **`docs_read` resolves its root from `process.cwd()`** and memoizes it globally. From 0930c7d4c8e93cf65ec580382ea16331304a3fb5 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 04:51:10 +0000 Subject: [PATCH 21/43] fix: http_request status, terminal_timeout_ms, .. path guard, OpenAI tool ids, MCP name collisions - http_request returns ok:false (error http_) for non-2xx, keeping status and body in output. - terminal defaults its command timeout to config terminal_timeout_ms. - The path guard rejects only .. and ../ prefixes, so names like ..notes.txt inside work_dir are allowed. - OpenAI-compatible tool calls without an id get a unique generated id. - MCP names that collide after sanitizing log a warning; first wins. - Backoff comment and provider docs no longer claim the deterministic jitter desynchronizes simultaneous callers. Part of #133. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 10 ++++++++ README.md | 2 +- docs/architecture/providers.md | 6 ++--- docs/architecture/tools.md | 8 +++---- docs/user-guide/cli.md | 4 ++-- src/mcp/mcp_register.ts | 8 ++++++- src/providers/failover.ts | 6 +++-- src/providers/openai.ts | 10 +++++++- src/tools/builtin/http_request.ts | 4 +++- src/tools/builtin/terminal.ts | 10 ++++++-- src/tools/guard.ts | 3 ++- test/mcp_register.test.ts | 38 +++++++++++++++++++++++++++++++ test/providers.test.ts | 29 +++++++++++++++++++++++ test/tools.test.ts | 13 +++++++++++ test/tools_extra.test.ts | 9 ++++++++ 15 files changed, 142 insertions(+), 18 deletions(-) create mode 100644 test/mcp_register.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 977eb31..d595e0a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,16 @@ ## Unreleased +- `http_request` reports a non-2xx status as `ok: false` (`error: http_`) + and still returns the status and body, like `fetch_url`. +- `terminal` uses config `terminal_timeout_ms` as its default command timeout + (it was ignored); a `timeout_ms` argument still overrides it. +- File tools accept names that only start with `..` (such as `..notes.txt`); + `..` and `../x` are still rejected. +- OpenAI-compatible tool calls without an id get a unique generated id + instead of `""`. +- MCP tools whose names collide after sanitizing log a warning; the first one + registered wins (#133). - **Breaking:** a `tools_enabled` list (top-level or `gateway.tools_enabled`) now filters plugin tools too, including the gatekeeper's `git_commit`. List each plugin tool you want exposed; `"all"` is unchanged. Previously plugin diff --git a/README.md b/README.md index b16d2d4..6a8ff95 100644 --- a/README.md +++ b/README.md @@ -141,7 +141,7 @@ the next self-commit. One gated `git_commit` per run requires | `LICH_ALLOW_PRIVATE_URLS` | set to exactly `1` to let `fetch_url` / `http_request` reach private or loopback URLs; unset or any other value is fail-closed (blocked) | | `LICH_TEST_COMMAND` | command `run_tests` runs (default: `node node_modules/vitest/vitest.mjs run`) | | `LICH_DOCS_DIR` | optional docs root for `docs_read` / `docs_search` (directory with `index.md`, or a parent containing `docs/`); else `/docs` or the package docs | -| `LICH_TERMINAL_TIMEOUT_MS` | injected into tool context from config `terminal_timeout_ms`; the `terminal` tool does **not** read it yet — use the tool's `timeout_ms` arg (default 60000) | +| `LICH_TERMINAL_TIMEOUT_MS` | injected into tool context from config `terminal_timeout_ms`; the `terminal` tool's default command timeout (60000), overridden by the tool's `timeout_ms` arg | Gateway-only vars are listed under [Gateway](#gateway). diff --git a/docs/architecture/providers.md b/docs/architecture/providers.md index 4ef9915..24e52e3 100644 --- a/docs/architecture/providers.md +++ b/docs/architecture/providers.md @@ -194,9 +194,9 @@ jitter = floor((500 * attempt) / 2) delay = min(exponential + jitter, 8000) ``` -Attempt 1 waits 1250 ms, attempt 2 waits 2500 ms. The fixed jitter term -spreads simultaneous callers without nondeterminism (tests assert exact -values). +Attempt 1 waits 1250 ms, attempt 2 waits 2500 ms. Because the formula is +deterministic, callers that fail at the same moment also retry at the same +moment; it does not desynchronize them (tests assert exact values). **`Retry-After` floor.** If the failed call produced `retry_after_ms` larger than the computed backoff, the server value wins (`delay_for_error`). diff --git a/docs/architecture/tools.md b/docs/architecture/tools.md index 260e093..0fbe8c0 100644 --- a/docs/architecture/tools.md +++ b/docs/architecture/tools.md @@ -42,9 +42,9 @@ confinement root), a process environment map (the agent injects from `Agent.run` so tools cancel on caller abort **or** the executor deadline (`tool.timeout_ms`, else `DEFAULT_TOOL_TIMEOUT_MS` = 30000). `terminal` sets 300000, `run_tests` sets -600000, and registered MCP tools set 120000. Note: `LICH_TERMINAL_TIMEOUT_MS` -is injected from config `terminal_timeout_ms` but the `terminal` tool ignores -it today — use the tool's `timeout_ms` argument (default 60000). +600000, and registered MCP tools set 120000. The `terminal` tool's own +command timeout defaults to config `terminal_timeout_ms` (injected as +`LICH_TERMINAL_TIMEOUT_MS`, default 60000); a `timeout_ms` argument overrides it. **Parameter schemas.** `parameters` is a `JsonSchemaObject` (`src/util/json_schema.ts`) passed through verbatim into provider requests. @@ -157,7 +157,7 @@ Docs tools join the list only when a docs root resolves. | `grep_files` | `pattern`, `path?`, `glob?`, `max_results?` | Explicit stack walk (no recursion), skips `SKIP_DIRS` entries and symbolic links, per-file `assert_file_tool_access` check (`.lich/config.json` is denied), binary sniff (NUL byte in first 1000 bytes), 1 MB file cap, `*.ext` suffix-glob matcher, overcollect-by-one to report suppressed counts. | | `fetch_url` | `url`, `max_chars?`, `timeout_ms?` | GET only; rejects non-http(s) protocols; refuses images/octet-stream; tags HTML bodies with `[html content]`; status/type header line first. | | `web_search` | `query`, `max_results?` | Scrapes DuckDuckGo's HTML endpoint (no API key); unwraps `uddg=` redirect links; decodes the handful of entities DDG emits. | -| `http_request` | `url`, `method?`, `headers?`, `body?`, ... | Method allowlist (GET/POST/PUT/PATCH/DELETE/HEAD/OPTIONS); stringified caller headers; reports `content-length`, `ratelimit-remaining`, `retry-after`. | +| `http_request` | `url`, `method?`, `headers?`, `body?`, ... | Method allowlist (GET/POST/PUT/PATCH/DELETE/HEAD/OPTIONS); stringified caller headers; reports `content-length`, `ratelimit-remaining`, `retry-after`. A non-2xx status returns `ok: false` with `error: http_` and keeps the status and body in `output`. | | `process_list` | `filter?`, `max_results?` | Reads `/proc` synchronously: numeric dirs are pids, `cmdline` is NUL-separated; missing entries (process died mid-scan) read as empty. | | `disk_usage` | `path?`, `max_entries?` | One `du -sb` subprocess per depth-1 entry with a 10 s timeout; sorted desc with a `TOTAL` row; `du` missing yields `du_unavailable`. | | `env_get` | `keys?`, `prefix?`, `reveal?` | Values are hidden unless `reveal`; names matching `/(secret\|token\|password\|key\|credential\|auth)/i` are **always** masked as ``. | diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index a2a5f7d..eba74a3 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -145,7 +145,7 @@ Validated by zod (top-level unknown keys are silently stripped; extra keys insid | `context_budget_tokens` | positive int | `100000` | Estimated budget before compression triggers. | | `compress_threshold` | 0.1–0.95 | `0.8` | Compress when usage >= this fraction of the budget. | | `session_dir` | string | `/.lich/sessions` | Transcript directory. | -| `terminal_timeout_ms` | positive int | `60000` | Written into tool context as `LICH_TERMINAL_TIMEOUT_MS`. The `terminal` tool does **not** read it yet; pass `timeout_ms` on the tool call (default 60000, max 300000). | +| `terminal_timeout_ms` | positive int | `60000` | Default command timeout for the `terminal` tool (passed in tool context as `LICH_TERMINAL_TIMEOUT_MS`). A `timeout_ms` on the tool call overrides it (max 300000). | | `log_level` | enum | `info` | Logger verbosity. | Minimal per-provider examples: @@ -175,7 +175,7 @@ never accepted as a config passthrough. | `LICH_ALLOW_PRIVATE_URLS` | Set to exactly `1` to let `fetch_url` / `http_request` reach private or loopback URLs. Unset or any other value is fail-closed (they are blocked). | | `LICH_TEST_COMMAND` | Command `run_tests` runs in `work_dir` (default `node node_modules/vitest/vitest.mjs run`). An optional `filter` argument is appended. | | `LICH_DOCS_DIR` | Optional docs root for `docs_read` / `docs_search` (dir with `index.md`, or a parent containing `docs/`). Else `/docs` or package docs. | -| `LICH_TERMINAL_TIMEOUT_MS` | Injected from config `terminal_timeout_ms` into tool context. Unused by `terminal` today — use the tool's `timeout_ms` arg. | +| `LICH_TERMINAL_TIMEOUT_MS` | Injected from config `terminal_timeout_ms` into tool context; the `terminal` tool's default command timeout. | Veto reasons, the terminal git denylist, skills, and `MEMORY.md` are in the [plugins guide](plugins.md#self-improvement-loop). diff --git a/src/mcp/mcp_register.ts b/src/mcp/mcp_register.ts index cb67314..c5c9eaa 100644 --- a/src/mcp/mcp_register.ts +++ b/src/mcp/mcp_register.ts @@ -6,6 +6,7 @@ import { mcp_tool_name } from "./mcp_names.js"; import { pin_for } from "./mcp_pin.js"; import type { ListedTool } from "./mcp_result.js"; import type { McpSession } from "./mcp_session.js"; +import { logger } from "../util/log.js"; function name_allowed(enabled: AgentConfig["tools_enabled"], name: string): boolean { if (enabled === "all") { @@ -52,7 +53,12 @@ export function register_listed( continue; } const registered = mcp_tool_name(server, spec.name); - if (name_allowed(enabled, registered) === false || registry.has(registered) === true) { + if (name_allowed(enabled, registered) === false) { + continue; + } + if (registry.has(registered) === true) { + // Sanitizing can map different names to one (e.g. "a-b" and "a_b"); first wins. + logger.warn(`mcp tool ${server}/${spec.name} maps to ${registered}, which is already registered; skipping`); continue; } registry.register(tool_for(registered, spec.name, spec, session)); diff --git a/src/providers/failover.ts b/src/providers/failover.ts index 9545f54..df0a985 100644 --- a/src/providers/failover.ts +++ b/src/providers/failover.ts @@ -21,8 +21,10 @@ export function classify_error(error: unknown): ProviderErrorKind { } /** - * Deterministic exponential backoff: base * 2**attempt capped at max, plus a - * fixed jitter term so simultaneous callers spread out without Math.random. + * Deterministic exponential backoff: base * 2**attempt plus a fixed term, + * capped at max. It is deterministic on purpose (testable, reproducible), so + * callers that fail at the same moment also retry at the same moment; it does + * not desynchronize them. */ export function compute_backoff_ms( attempt: number, diff --git a/src/providers/openai.ts b/src/providers/openai.ts index f54b64d..cd95e0e 100644 --- a/src/providers/openai.ts +++ b/src/providers/openai.ts @@ -15,6 +15,14 @@ import type { Usage, } from "./types.js"; +/** Tool calls the server sent without an id still need a unique one to pair with their results. */ +let tool_call_counter = 0; + +function next_tool_call_id(): string { + tool_call_counter += 1; + return `openai_${Date.now().toString(36)}_${tool_call_counter}`; +} + const DEFAULT_BASE_URL = "https://api.openai.com/v1"; const WELL_KNOWN_HOST = "api.openai.com"; const WELL_KNOWN_KEY_ENV = "OPENAI_API_KEY"; @@ -383,7 +391,7 @@ function parse_assistant_message( raw_arguments.length === 0 ? {} : safe_json_parse>(raw_arguments); if (is_record(parsed_arguments) === true) { tool_calls.push({ - id: raw_call.id ?? "", + id: raw_call.id !== undefined && raw_call.id.length > 0 ? raw_call.id : next_tool_call_id(), name: raw_call.function?.name ?? "", args: parsed_arguments, }); diff --git a/src/tools/builtin/http_request.ts b/src/tools/builtin/http_request.ts index 6b80d13..905832e 100644 --- a/src/tools/builtin/http_request.ts +++ b/src/tools/builtin/http_request.ts @@ -89,7 +89,9 @@ async function run_http(args: Record, external?: AbortSignal): "", clamp_output(clamped.text, max_chars), ]; - return { ok: true, output: sections.join("\n") }; + const output = sections.join("\n"); + // Non-2xx keeps the status and body in output but reports failure, like fetch_url. + return response.ok ? { ok: true, output } : { ok: false, output, error: `http_${response.status}` }; } export const http_request_tool: Tool = { diff --git a/src/tools/builtin/terminal.ts b/src/tools/builtin/terminal.ts index f7605dc..2ecdb79 100644 --- a/src/tools/builtin/terminal.ts +++ b/src/tools/builtin/terminal.ts @@ -25,7 +25,7 @@ const parameters: JsonSchemaObject = { type: "object", properties: { command: { type: "string", description: "Shell command to run via bash -lc" }, - timeout_ms: { type: "number", description: "Kill the command after this many ms (default 60000, max 300000)" }, + timeout_ms: { type: "number", description: "Kill the command after this many ms (default: config terminal_timeout_ms, 60000; max 300000)" }, }, required: ["command"], additionalProperties: false, @@ -41,6 +41,12 @@ function stream_chunk(current: { text: string }, chunk: Buffer): void { } } +/** `terminal_timeout_ms` from config (passed as LICH_TERMINAL_TIMEOUT_MS), else the built-in default. */ +function configured_timeout(env: Record): number { + const configured = Number(env["LICH_TERMINAL_TIMEOUT_MS"]); + return Number.isFinite(configured) && configured > 0 ? configured : DEFAULT_TIMEOUT_MS; +} + function clamp_timeout(raw: number): number { return Math.min(MAX_TIMEOUT_MS, Math.max(1, Math.floor(raw))); } @@ -148,7 +154,7 @@ export const terminal_tool: Tool = { execute: async (args, context) => capture_errors(async () => { const command = require_string_arg(args, "command"); - const timeout_ms = clamp_timeout(optional_number_arg(args, "timeout_ms", DEFAULT_TIMEOUT_MS)); + const timeout_ms = clamp_timeout(optional_number_arg(args, "timeout_ms", configured_timeout(context.env))); const outcome = await run_command(command, context.work_dir, context.env, timeout_ms, context.signal); return terminal_result(outcome); }), diff --git a/src/tools/guard.ts b/src/tools/guard.ts index a86c4c3..438d73e 100644 --- a/src/tools/guard.ts +++ b/src/tools/guard.ts @@ -9,7 +9,8 @@ export const DEFAULT_TOOL_TIMEOUT_MS = 30000; /** True when `candidate` is `base` or a descendant (lexical). */ function is_inside(base: string, candidate: string): boolean { const relative = path.relative(base, candidate); - return relative.startsWith("..") === false && path.isAbsolute(relative) === false; + const escapes = relative === ".." || relative.startsWith(`..${path.sep}`); + return escapes === false && path.isAbsolute(relative) === false; } /** Walk up from `target` until an existing path is found (non-recursive). */ diff --git a/test/mcp_register.test.ts b/test/mcp_register.test.ts new file mode 100644 index 0000000..e69f838 --- /dev/null +++ b/test/mcp_register.test.ts @@ -0,0 +1,38 @@ +/** + * register_listed: MCP names that sanitize to the same registered name keep + * the first tool and warn instead of dropping the second silently (#133). + */ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { register_listed } from "../src/mcp/mcp_register.js"; +import type { McpSession } from "../src/mcp/mcp_session.js"; +import { ToolRegistry } from "../src/tools/registry.js"; +import { logger } from "../src/util/log.js"; + +afterEach(() => { + vi.restoreAllMocks(); +}); + +const session = { call_tool: async (name: string) => `called ${name}` } as unknown as McpSession; +const parameters = { type: "object", properties: {} } as const; + +describe("register_listed", () => { + it("keeps the first tool when two names sanitize to the same registered name, and warns", async () => { + const warn = vi.spyOn(logger, "warn").mockImplementation(() => undefined); + const registry = new ToolRegistry(); + register_listed( + registry, + "editor", + [ + { name: "open-file", description: "first", parameters }, + { name: "open_file", description: "second", parameters }, + ], + "all", + session, + ); + expect(registry.list().map((tool) => tool.name)).toEqual(["mcp_editor_open_file"]); + const tool = registry.get("mcp_editor_open_file"); + expect(tool?.description).toBe("first"); + expect(warn).toHaveBeenCalledTimes(1); + expect(String(warn.mock.calls[0]?.[0])).toContain("editor/open_file"); + }); +}); diff --git a/test/providers.test.ts b/test/providers.test.ts index d40179e..06c37ef 100644 --- a/test/providers.test.ts +++ b/test/providers.test.ts @@ -182,6 +182,35 @@ describe("openai compat provider", () => { expect(result.provider_name).toBe("openai-main"); }); + it("gives tool calls without an id a unique non-empty id", async () => { + const { fetch_fn } = mock_fetch(() => ({ + status: 200, + body: { + model: "gpt-test", + choices: [ + { + message: { + role: "assistant", + content: "", + tool_calls: [ + { type: "function", function: { name: "list_dir", arguments: "{}" } }, + { id: "", type: "function", function: { name: "read_file", arguments: "{}" } }, + ], + }, + finish_reason: "tool_calls", + }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }, + })); + const provider = new OpenAICompatProvider(openai_config({ fetch_fn })); + const result = await provider.chat([{ role: "user", content: "go" }], [SAMPLE_TOOL]); + const ids = (result.message.tool_calls ?? []).map((call) => call.id); + expect(ids).toHaveLength(2); + expect(ids.every((id) => id.length > 0)).toBe(true); + expect(new Set(ids).size).toBe(2); + }); + it("omits executable tool calls and appends a note when arguments do not parse", async () => { const { fetch_fn } = mock_fetch(() => ({ status: 200, diff --git a/test/tools.test.ts b/test/tools.test.ts index 54819ad..13df948 100644 --- a/test/tools.test.ts +++ b/test/tools.test.ts @@ -108,6 +108,11 @@ describe("guard", () => { expect(() => resolve_safe_path(tmp_root, "/etc/passwd")).toThrow(/path_escape/); }); + it("resolve_safe_path accepts a name that only starts with two dots", () => { + expect(resolve_safe_path(tmp_root, "..notes.txt")).toBe(path.join(tmp_root, "..notes.txt")); + expect(() => resolve_safe_path(tmp_root, "..")).toThrow(/path_escape/); + }); + it("resolve_safe_path accepts inside paths", () => { const relative = resolve_safe_path(tmp_root, "./sub/file.txt"); expect(relative.startsWith(tmp_root) === true).toBe(true); @@ -246,6 +251,14 @@ describe("executor", () => { expect(Date.now() - started < 2000).toBe(true); }); + it("terminal defaults its timeout to config terminal_timeout_ms", async () => { + const context: ToolContext = { work_dir: tmp_root, env: { LICH_TERMINAL_TIMEOUT_MS: "200" } }; + const started = Date.now(); + const result = await terminal_tool.execute({ command: "sleep 5" }, context); + expect(result.error).toBe("timeout"); + expect(Date.now() - started < 4000).toBe(true); + }); + it("terminal declares a 300000ms executor timeout", () => { expect(terminal_tool.timeout_ms).toBe(300000); }); diff --git a/test/tools_extra.test.ts b/test/tools_extra.test.ts index 6bf7d00..c552edb 100644 --- a/test/tools_extra.test.ts +++ b/test/tools_extra.test.ts @@ -326,6 +326,15 @@ describe("http_request", () => { expect(result.output.includes('{"ok":true}')).toBe(true); }); + it("reports a non-2xx status as a failure but keeps status and body", async () => { + stub_fetch(vi.fn(async () => fake_response('{"error":"nope"}', { status: 404, headers: { "content-type": "application/json" } }))); + const result = await executor.execute("http_request", { url: "https://api.example.com/missing" }); + expect(result.ok).toBe(false); + expect(result.error).toBe("http_404"); + expect(result.output).toContain("# status 404"); + expect(result.output).toContain('{"error":"nope"}'); + }); + it("sends no body for GET and rejects invalid methods", async () => { const fetch_mock = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => fake_response("fine")); stub_fetch(fetch_mock); From 09fc44406ec2a073125cdab31a3f93c4df8b1a2c Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 04:56:46 +0000 Subject: [PATCH 22/43] fix(openai): treat a null tool-call id as missing; correct http_request note Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 2 +- src/providers/openai.ts | 2 +- src/tools/builtin/http_request.ts | 3 ++- test/providers.test.ts | 7 ++++--- 4 files changed, 8 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d595e0a..891fcc0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,7 +3,7 @@ ## Unreleased - `http_request` reports a non-2xx status as `ok: false` (`error: http_`) - and still returns the status and body, like `fetch_url`. + and still returns the status and body in `output`. - `terminal` uses config `terminal_timeout_ms` as its default command timeout (it was ignored); a `timeout_ms` argument still overrides it. - File tools accept names that only start with `..` (such as `..notes.txt`); diff --git a/src/providers/openai.ts b/src/providers/openai.ts index cd95e0e..4a32884 100644 --- a/src/providers/openai.ts +++ b/src/providers/openai.ts @@ -391,7 +391,7 @@ function parse_assistant_message( raw_arguments.length === 0 ? {} : safe_json_parse>(raw_arguments); if (is_record(parsed_arguments) === true) { tool_calls.push({ - id: raw_call.id !== undefined && raw_call.id.length > 0 ? raw_call.id : next_tool_call_id(), + id: typeof raw_call.id === "string" && raw_call.id.length > 0 ? raw_call.id : next_tool_call_id(), name: raw_call.function?.name ?? "", args: parsed_arguments, }); diff --git a/src/tools/builtin/http_request.ts b/src/tools/builtin/http_request.ts index 905832e..c28c724 100644 --- a/src/tools/builtin/http_request.ts +++ b/src/tools/builtin/http_request.ts @@ -90,7 +90,8 @@ async function run_http(args: Record, external?: AbortSignal): clamp_output(clamped.text, max_chars), ]; const output = sections.join("\n"); - // Non-2xx keeps the status and body in output but reports failure, like fetch_url. + // Non-2xx reports failure but keeps status and body in output on purpose (API error + // bodies matter); unlike fetch_url, which returns an empty output. return response.ok ? { ok: true, output } : { ok: false, output, error: `http_${response.status}` }; } diff --git a/test/providers.test.ts b/test/providers.test.ts index 06c37ef..54fde14 100644 --- a/test/providers.test.ts +++ b/test/providers.test.ts @@ -195,6 +195,7 @@ describe("openai compat provider", () => { tool_calls: [ { type: "function", function: { name: "list_dir", arguments: "{}" } }, { id: "", type: "function", function: { name: "read_file", arguments: "{}" } }, + { id: null, type: "function", function: { name: "list_dir", arguments: "{}" } }, ], }, finish_reason: "tool_calls", @@ -206,9 +207,9 @@ describe("openai compat provider", () => { const provider = new OpenAICompatProvider(openai_config({ fetch_fn })); const result = await provider.chat([{ role: "user", content: "go" }], [SAMPLE_TOOL]); const ids = (result.message.tool_calls ?? []).map((call) => call.id); - expect(ids).toHaveLength(2); - expect(ids.every((id) => id.length > 0)).toBe(true); - expect(new Set(ids).size).toBe(2); + expect(ids).toHaveLength(3); + expect(ids.every((id) => typeof id === "string" && id.length > 0)).toBe(true); + expect(new Set(ids).size).toBe(3); }); it("omits executable tool calls and appends a note when arguments do not parse", async () => { From 07d0d3be07003b274ccb97616de3b2e6b7cc9811 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 05:05:41 +0000 Subject: [PATCH 23/43] docs: webhook binds loopback by default; drop stale version from overview; wiki #163 - godot.md and the persona orchestrator README said the CLI webhook binds 0.0.0.0; it binds 127.0.0.1 unless LICH_GATEWAY_HOST is set, and a non-loopback bind requires LICH_GATEWAY_TOKEN. - docs/architecture/overview.md named 0.8.0 as the current package; point to CHANGELOG instead of a version that goes stale. - wiki: S-11 holes fixed by #163. Part of #133. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/architecture/overview.md | 6 +++--- docs/user-guide/godot.md | 4 ++-- examples/persona_orchestrator/README.md | 2 +- wiki/entities/lich-tools-and-guardrails.md | 5 +++-- wiki/log.md | 1 + 5 files changed, 10 insertions(+), 8 deletions(-) diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md index a674215..645224f 100644 --- a/docs/architecture/overview.md +++ b/docs/architecture/overview.md @@ -2,8 +2,8 @@ Lich is a small TypeScript AI agent harness: it drives a chat model in a think-act-observe loop, lets the model call tools, compresses history when the -context budget demands it, and persists transcripts. The published npm package -is 0.8.0, including editor MCP (`mcp_servers`, `lich mcp`). It runs on +context budget demands it, and persists transcripts. It includes editor MCP +(`mcp_servers`, `lich mcp`); see `CHANGELOG.md` for the current version. It runs on Node >= 20 and on Bun, is ESM with NodeNext resolution, and its runtime dependencies are `zod` (config validation), `ink`, and `react` (the TUI). Everything else is Node/Bun built-ins. @@ -104,7 +104,7 @@ Walkthrough of a single `Agent.run({ input })` call are connected (`initialize`, `notifications/initialized`, then `tools/list`) and registered as `mcp__`. An empty `tools_enabled` never connects. Disabled servers are skipped. A connect failure logs a warning - and the run continues. This client has shipped since 0.7.0 (current package 0.8.0). + and the run continues. This client has shipped since 0.7.0. 2. **Usage collector attached.** `run()` subscribes a `collect_usage` handler on `agent.events`; every `llm_end` event adds the call's token usage into a per-run `Usage` total. The subscription is removed in a `finally` block. diff --git a/docs/user-guide/godot.md b/docs/user-guide/godot.md index b2478f2..2e4fcc7 100644 --- a/docs/user-guide/godot.md +++ b/docs/user-guide/godot.md @@ -30,7 +30,7 @@ A game backend may instead call `run_agent`, which loads `config.plugins`. `crea ## Gateway contract -`lich gateway webhook` binds `0.0.0.0` on `LICH_GATEWAY_PORT` (default `8089`). `GET /health` is `200 {"status":"ok"}` — the process is up, not that a provider is healthy. Check it before the first round. +`lich gateway webhook` binds `127.0.0.1` on `LICH_GATEWAY_PORT` (default `8089`). To reach it from another machine, set `LICH_GATEWAY_HOST` (for example `0.0.0.0`); a non-loopback bind requires `LICH_GATEWAY_TOKEN`. `GET /health` is `200 {"status":"ok"}` — the process is up, not that a provider is healthy. Check it before the first round. `POST /message`. Only `text` is required. Omitted fields default to `platform` `"webhook"`, `chat_id` `"default"`, `user_id` `"anonymous"`. @@ -148,7 +148,7 @@ The plugin checks the line shape and returns `{ok: false, error}` instead of thr ## Security -Set `LICH_GATEWAY_TOKEN`. The webhook binds all interfaces, so an open port is an open chatbot with your provider keys and your tools. Mismatch or a missing header is `401`. +Set `LICH_GATEWAY_TOKEN`. The webhook binds loopback by default and refuses a non-loopback `LICH_GATEWAY_HOST` without a token, because an open port is an open chatbot with your provider keys and your tools. Mismatch or a missing header is `401`. Players' Godot clients do not talk to lich in production. Godot talks to your backend; the backend holds the token, sets `chat_id` / `user_id`, and rate-limits. Same split as embedding the library in that backend. diff --git a/examples/persona_orchestrator/README.md b/examples/persona_orchestrator/README.md index 24544f7..34bc95b 100644 --- a/examples/persona_orchestrator/README.md +++ b/examples/persona_orchestrator/README.md @@ -39,7 +39,7 @@ Tool shapes and the meteor gate: [`examples/game_bridge/README.md`](../game_brid ## HTTP -Loopback only (`127.0.0.1`), default port `8090` so it does not collide with the CLI webhook (`8089` on `0.0.0.0`). A non-empty `token` is **required**; requests must send matching `x-lich-token`. `POST /message` also requires `Host` to be loopback, `Content-Type: application/json`, and a body under ~1 MB. +Loopback only (`127.0.0.1`), default port `8090` so it does not collide with the CLI webhook (`8089`, also loopback by default). A non-empty `token` is **required**; requests must send matching `x-lich-token`. `POST /message` also requires `Host` to be loopback, `Content-Type: application/json`, and a body under ~1 MB. ```sh export LICH_GATEWAY_TOKEN=sekrit diff --git a/wiki/entities/lich-tools-and-guardrails.md b/wiki/entities/lich-tools-and-guardrails.md index 0377dd8..4d11620 100644 --- a/wiki/entities/lich-tools-and-guardrails.md +++ b/wiki/entities/lich-tools-and-guardrails.md @@ -4,7 +4,7 @@ created: 2026-09-23 updated: 2026-10-04 type: entity tags: [tools, security, runtime] -sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, "#161"] +sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, "#161", "#163"] confidence: high --- @@ -38,10 +38,11 @@ It also: **Fixed in #161:** a `tools_enabled` list (including `gateway.tools_enabled`) now filters plugin tools and `git_commit` (`src/agent/agent.ts:111@029b7e8`), and a throwing `before_tool_call` blocks the call. +**Fixed in #163 (S-11):** `http_request` reports non-2xx as `ok:false`; the path guard rejects only `..` and `../` prefixes (`src/tools/guard.ts:12-13@7001e31`); `terminal` uses `terminal_timeout_ms` as its default timeout. + ## Known holes (open on v0.9.0) - **Hooks have no timeout.** -- **S-11:** `http_request` always returns `ok:true`; `guard.ts:12` checks `startsWith("..")`; `terminal_timeout_ms` is dead code. - **`docs_read` resolves its root from `process.cwd()`** and memoizes it globally. These holes are why the embedded profile starts with no builtins at all; see [[embedded-safety-profile]]. diff --git a/wiki/log.md b/wiki/log.md index d137507..b2edb3c 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -38,3 +38,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-04 create | decision_lane example merged in #159: plugin-owned System One client, shadow/act modes, game_bridge checks still apply, bench pending | entities/decision-lane-example.md, index.md - 2026-10-04 update | Decision models and game_bridge pages link the decision_lane prototype | concepts/decision-models.md, entities/game-bridge-example.md - 2026-10-04 update | #161 merged: tools_enabled filters plugin tools and git_commit, throwing before_tool_call blocks, last-turn tool calls not run; holes lists and index updated, code pinned at 029b7e8 | entities/lich-plugins-and-hooks.md, entities/lich-tools-and-guardrails.md, entities/lich-agent-loop.md, index.md +- 2026-10-04 update | #163 merged: S-11 items (http_request status, .. guard, terminal_timeout_ms) moved out of open holes, pinned at 7001e31 | entities/lich-tools-and-guardrails.md From 16c4bdb4d9d8817a2da2008e2a3b25bc62a43874 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 05:07:35 +0000 Subject: [PATCH 24/43] refactor(providers): share HTTP/error helpers in providers/http.ts (#133) openai, anthropic and ollama carried byte-identical copies of the fetch skeleton (abort signal, request init, do_fetch, body read, success JSON, Retry-After, status mapping, http error) plus is_record / is_abort_like. Move them into src/providers/http.ts; ollama passes its wider overflow regex to to_http_error. failover.ts reuses is_abort_like. Anthropic's explicit 529 check is dropped as redundant (529 >= 500 already maps to rate_limit). No behavior change. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/architecture/extending.md | 10 ++- docs/architecture/providers.md | 3 +- src/providers/anthropic.ts | 141 +++-------------------------- src/providers/failover.ts | 14 +-- src/providers/http.ts | 159 +++++++++++++++++++++++++++++++++ src/providers/ollama.ts | 158 +++----------------------------- src/providers/openai.ts | 159 +++------------------------------ 7 files changed, 206 insertions(+), 438 deletions(-) create mode 100644 src/providers/http.ts diff --git a/docs/architecture/extending.md b/docs/architecture/extending.md index fb35a24..eb82bf6 100644 --- a/docs/architecture/extending.md +++ b/docs/architecture/extending.md @@ -99,13 +99,15 @@ adding ollama required: - Typed wire DTOs (request/response), never `any`. - Small pure mapping helpers: `to_*_messages` (our `Message` -> wire), `assistant_to_wire` / `tool_to_wire`, `to_*_tools`, `build_request_body`. - - The fetch skeleton shared by all clients: resolve auth, `build_endpoint`, + - The fetch skeleton: resolve auth and `build_endpoint` per client, then + the shared helpers from [`http.ts`](../../src/providers/http.ts): `do_fetch` (fetch throws -> `network`), `to_http_error` (non-OK -> classified kind + `Retry-After`), `read_success_json` (unparseable 2xx -> `bad_request`). - - An error mapper: a `status_to_error_kind` function implementing the - taxonomy table from [providers](./providers.md#error-taxonomy), plus a - body-sniffing regex for `overflow` on 400s. + - Error mapping: `to_http_error` applies the shared `status_to_error_kind` + (the taxonomy table from [providers](./providers.md#error-taxonomy)); + pass your own body-sniffing regex for `overflow` on 400s if the default + `OVERFLOW_BODY_PATTERN` does not fit. - A factory export (`create_ollama_provider`) if construction may grow. 2. **The `ProviderConfig` union.** Add the kind string to the `kind` union in diff --git a/docs/architecture/providers.md b/docs/architecture/providers.md index 24e52e3..6380558 100644 --- a/docs/architecture/providers.md +++ b/docs/architecture/providers.md @@ -132,7 +132,8 @@ Usage comes from `prompt_eval_count` / `eval_count`. ## Error taxonomy `ProviderErrorKind` and the mapping rules are identical across clients -(`status_to_error_kind` in each client), with one anthropic addition: +(`status_to_error_kind` in `src/providers/http.ts`, shared by all clients; +ollama passes its own overflow regex), with one anthropic addition: | Kind | Meaning | OpenAI mapping | Anthropic mapping | Ollama mapping | | --- | --- | --- | --- | --- | diff --git a/src/providers/anthropic.ts b/src/providers/anthropic.ts index c4b7cc7..5546cfd 100644 --- a/src/providers/anthropic.ts +++ b/src/providers/anthropic.ts @@ -1,4 +1,12 @@ -import { safe_json_parse, safe_stringify, truncate_text } from "../util/json.js"; +import { safe_stringify } from "../util/json.js"; +import { + build_abort_signal, + build_request_init, + do_fetch, + is_record, + read_success_json, + to_http_error, +} from "./http.js"; import { ProviderError } from "./types.js"; import type { AssistantMessage, @@ -8,7 +16,6 @@ import type { LLMProvider, Message, ProviderConfig, - ProviderErrorKind, ToolCall, ToolDefinition, Usage, @@ -17,10 +24,6 @@ import type { const DEFAULT_BASE_URL = "https://api.anthropic.com"; const ANTHROPIC_VERSION = "2023-06-01"; const DEFAULT_KEY_ENV = "ANTHROPIC_API_KEY"; -const MAX_ERROR_BODY_CHARS = 500; -const OVERFLOW_BODY_PATTERN = - /context.?length|maximum context|prompt(?: is)? too (?:long|large)|token.?limit|context window|too many tokens|exceed.{0,30}context limit/i; -const OVERLOADED_STATUS = 529; const UNPARSEABLE_ARGS_NOTE = "[unparseable tool arguments]"; const TRUNCATED_TOOL_CALLS_NOTE = "[truncated tool call omitted]"; const EMPTY_TEXT_PLACEHOLDER = "(empty)"; @@ -115,7 +118,7 @@ export class AnthropicProvider implements LLMProvider { this.config.fetch_fn ?? fetch, build_endpoint(this.config), build_request_init( - api_key, + build_headers(api_key), safe_stringify(build_request_body(this.config.model, messages, tools, options, this.config)), build_abort_signal(options, this.config.timeout_ms), ), @@ -124,7 +127,7 @@ export class AnthropicProvider implements LLMProvider { if (response.ok === false) { throw await to_http_error(response, this.config.name); } - return parse_chat_response(await read_success_json(response, this.config.name), this.config); + return parse_chat_response(await read_success_json(response, this.config.name), this.config); } } @@ -154,21 +157,6 @@ function build_headers(api_key: string): Record { }; } -function build_abort_signal(options: ChatOptions | undefined, timeout_ms: number | undefined): AbortSignal | undefined { - const signals: AbortSignal[] = []; - if (timeout_ms !== undefined && timeout_ms > 0) { - signals.push(AbortSignal.timeout(timeout_ms)); - } - if (options?.signal !== undefined) { - signals.push(options.signal); - } - const [only_signal] = signals; - if (only_signal !== undefined && signals.length === 1) { - return only_signal; - } - return AbortSignal.any(signals); -} - function build_request_body( model: string, messages: readonly Message[], @@ -297,92 +285,6 @@ function to_anthropic_tools(tools: readonly ToolDefinition[]): AnthropicToolDto[ })); } -function build_request_init(api_key: string, body: string, signal: AbortSignal | undefined): RequestInit { - return { - method: "POST", - headers: build_headers(api_key), - body, - ...(signal !== undefined ? { signal } : {}), - }; -} - -async function do_fetch(fetch_fn: typeof fetch, url: string, init: RequestInit, provider_name: string): Promise { - try { - return await fetch_fn(url, init); - } catch (error) { - const label = is_abort_like(error) === true ? "request aborted or timed out" : "fetch failed"; - throw new ProviderError({ - kind: "network", - provider_name, - message: `${label}: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_response_text(response: Response, provider_name: string): Promise { - try { - return await response.text(); - } catch (error) { - throw new ProviderError({ - kind: "network", - provider_name, - message: `failed to read response body: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_success_json(response: Response, provider_name: string): Promise { - const text = await read_response_text(response, provider_name); - const dto = safe_json_parse(text); - if (dto === undefined) { - throw new ProviderError({ - kind: "bad_request", - provider_name, - message: `unparseable success response: ${truncate_text(text, MAX_ERROR_BODY_CHARS)}`, - }); - } - return dto; -} - -function parse_retry_after_ms(response: Response): number | undefined { - const raw = response.headers.get("retry-after"); - if (raw === null) { - return undefined; - } - const seconds = Number(raw); - if (Number.isFinite(seconds) === false || seconds < 0) { - return undefined; - } - return Math.round(seconds * 1000); -} - -function status_to_error_kind(status: number, body_text: string): ProviderErrorKind { - if (status === 401 || status === 403) { - return "auth"; - } - if (status === 429 || status === OVERLOADED_STATUS || status >= 500) { - return "rate_limit"; - } - if (status === 413 || (status === 400 && OVERFLOW_BODY_PATTERN.test(body_text) === true)) { - return "overflow"; - } - return "bad_request"; -} - -async function to_http_error(response: Response, provider_name: string): Promise { - const body_text = truncate_text(await read_response_text(response, provider_name), MAX_ERROR_BODY_CHARS); - const retry_after_ms = parse_retry_after_ms(response); - return new ProviderError({ - kind: status_to_error_kind(response.status, body_text), - provider_name, - message: `${provider_name} http ${response.status}: ${body_text}`, - status: response.status, - ...(retry_after_ms !== undefined ? { retry_after_ms } : {}), - }); -} - function parse_chat_response(dto: AnthropicChatResponseDto, config: ProviderConfig): ChatResult { const finish_reason = map_stop_reason(dto.stop_reason); return { @@ -478,25 +380,4 @@ function map_stop_reason(raw: string | null | undefined): FinishReason { return "length"; } return "unknown"; -} - -function is_record(value: unknown): value is Record { - return typeof value === "object" && value !== null; -} - -function describe_error(error: unknown): string { - return error instanceof Error ? error.message : String(error); -} - -function error_name(error: unknown): string | undefined { - if (typeof error === "object" && error !== null && "name" in error) { - const name = (error as { name?: unknown }).name; - return typeof name === "string" ? name : undefined; - } - return undefined; -} - -function is_abort_like(error: unknown): boolean { - const name = error_name(error); - return name === "AbortError" || name === "TimeoutError"; } \ No newline at end of file diff --git a/src/providers/failover.ts b/src/providers/failover.ts index df0a985..05fffa0 100644 --- a/src/providers/failover.ts +++ b/src/providers/failover.ts @@ -1,4 +1,5 @@ import { sleep } from "../util/sleep.js"; +import { is_abort_like } from "./http.js"; import { ProviderError } from "./types.js"; import type { ProviderErrorKind } from "./types.js"; @@ -152,17 +153,4 @@ function make_abort_error(cause: unknown): Error { function is_type_error(error: unknown): boolean { return error instanceof TypeError; -} - -function error_name(error: unknown): string | undefined { - if (typeof error === "object" && error !== null && "name" in error) { - const name = (error as { name?: unknown }).name; - return typeof name === "string" ? name : undefined; - } - return undefined; -} - -function is_abort_like(error: unknown): boolean { - const name = error_name(error); - return name === "AbortError" || name === "TimeoutError"; } \ No newline at end of file diff --git a/src/providers/http.ts b/src/providers/http.ts new file mode 100644 index 0000000..0b85de5 --- /dev/null +++ b/src/providers/http.ts @@ -0,0 +1,159 @@ +import { safe_json_parse, truncate_text } from "../util/json.js"; +import { ProviderError } from "./types.js"; +import type { ChatOptions, ProviderErrorKind } from "./types.js"; + +/** + * HTTP/error helpers shared by the hand-written provider clients + * (openai, anthropic, ollama). + */ +export const MAX_ERROR_BODY_CHARS = 500; +export const OVERFLOW_BODY_PATTERN = + /context.?length|maximum context|prompt(?: is)? too (?:long|large)|token.?limit|context window|too many tokens|exceed.{0,30}context limit/i; + +export function first_non_empty(values: ReadonlyArray): string | undefined { + for (const value of values) { + if (value !== undefined && value.length > 0) { + return value; + } + } + return undefined; +} + +export function build_bearer_headers(api_key: string | undefined): Record { + const headers: Record = { "content-type": "application/json" }; + if (api_key !== undefined) { + headers.authorization = `Bearer ${api_key}`; + } + return headers; +} + +export function build_abort_signal(options: ChatOptions | undefined, timeout_ms: number | undefined): AbortSignal | undefined { + const signals: AbortSignal[] = []; + if (timeout_ms !== undefined && timeout_ms > 0) { + signals.push(AbortSignal.timeout(timeout_ms)); + } + if (options?.signal !== undefined) { + signals.push(options.signal); + } + const [only_signal] = signals; + if (only_signal !== undefined && signals.length === 1) { + return only_signal; + } + return AbortSignal.any(signals); +} + +export function build_request_init( + headers: Record, + body: string, + signal: AbortSignal | undefined, +): RequestInit { + return { + method: "POST", + headers, + body, + ...(signal !== undefined ? { signal } : {}), + }; +} + +export async function do_fetch(fetch_fn: typeof fetch, url: string, init: RequestInit, provider_name: string): Promise { + try { + return await fetch_fn(url, init); + } catch (error) { + const label = is_abort_like(error) === true ? "request aborted or timed out" : "fetch failed"; + throw new ProviderError({ + kind: "network", + provider_name, + message: `${label}: ${describe_error(error)}`, + cause: error, + }); + } +} + +async function read_response_text(response: Response, provider_name: string): Promise { + try { + return await response.text(); + } catch (error) { + throw new ProviderError({ + kind: "network", + provider_name, + message: `failed to read response body: ${describe_error(error)}`, + cause: error, + }); + } +} + +export async function read_success_json(response: Response, provider_name: string): Promise { + const text = await read_response_text(response, provider_name); + const dto = safe_json_parse(text); + if (dto === undefined) { + throw new ProviderError({ + kind: "bad_request", + provider_name, + message: `unparseable success response: ${truncate_text(text, MAX_ERROR_BODY_CHARS)}`, + }); + } + return dto; +} + +function parse_retry_after_ms(response: Response): number | undefined { + const raw = response.headers.get("retry-after"); + if (raw === null) { + return undefined; + } + const seconds = Number(raw); + if (Number.isFinite(seconds) === false || seconds < 0) { + return undefined; + } + return Math.round(seconds * 1000); +} + +/** 5xx (including Anthropic's 529 overloaded) maps to rate_limit so it is retried. */ +function status_to_error_kind(status: number, body_text: string, overflow_pattern: RegExp): ProviderErrorKind { + if (status === 401 || status === 403) { + return "auth"; + } + if (status === 429 || status >= 500) { + return "rate_limit"; + } + if (status === 413 || (status === 400 && overflow_pattern.test(body_text) === true)) { + return "overflow"; + } + return "bad_request"; +} + +export async function to_http_error( + response: Response, + provider_name: string, + overflow_pattern: RegExp = OVERFLOW_BODY_PATTERN, +): Promise { + const body_text = truncate_text(await read_response_text(response, provider_name), MAX_ERROR_BODY_CHARS); + const retry_after_ms = parse_retry_after_ms(response); + return new ProviderError({ + kind: status_to_error_kind(response.status, body_text, overflow_pattern), + provider_name, + message: `${provider_name} http ${response.status}: ${body_text}`, + status: response.status, + ...(retry_after_ms !== undefined ? { retry_after_ms } : {}), + }); +} + +export function is_record(value: unknown): value is Record { + return typeof value === "object" && value !== null; +} + +function describe_error(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +function error_name(error: unknown): string | undefined { + if (typeof error === "object" && error !== null && "name" in error) { + const name = (error as { name?: unknown }).name; + return typeof name === "string" ? name : undefined; + } + return undefined; +} + +export function is_abort_like(error: unknown): boolean { + const name = error_name(error); + return name === "AbortError" || name === "TimeoutError"; +} diff --git a/src/providers/ollama.ts b/src/providers/ollama.ts index ea09a4b..673412c 100644 --- a/src/providers/ollama.ts +++ b/src/providers/ollama.ts @@ -1,4 +1,15 @@ import { safe_json_parse, safe_stringify, truncate_text } from "../util/json.js"; +import { + build_abort_signal, + build_bearer_headers, + build_request_init, + do_fetch, + first_non_empty, + is_record, + MAX_ERROR_BODY_CHARS, + read_success_json, + to_http_error, +} from "./http.js"; import { ProviderError } from "./types.js"; import type { AssistantMessage, @@ -8,7 +19,6 @@ import type { LLMProvider, Message, ProviderConfig, - ProviderErrorKind, ToolCall, ToolDefinition, ToolMessage, @@ -16,7 +26,6 @@ import type { } from "./types.js"; const DEFAULT_BASE_URL = "http://localhost:11434"; -const MAX_ERROR_BODY_CHARS = 500; const OVERFLOW_BODY_PATTERN = /context.?length|maximum context|prompt(?: is)? too (?:long|large)|token.?limit|context window|too many tokens|too long|exceed.{0,30}context limit/i; const UNPARSEABLE_ARGS_NOTE = "[unparseable tool arguments]"; @@ -117,16 +126,16 @@ export class OllamaProvider implements LLMProvider { this.config.fetch_fn ?? fetch, build_endpoint(this.config), build_request_init( - resolve_api_key(this.config), + build_bearer_headers(resolve_api_key(this.config)), safe_stringify(build_request_body(this.config, messages, tools, options)), build_abort_signal(options, this.config.timeout_ms), ), this.config.name, ); if (response.ok === false) { - throw await to_http_error(response, this.config.name); + throw await to_http_error(response, this.config.name, OVERFLOW_BODY_PATTERN); } - return to_chat_response(await read_success_json(response, this.config.name), this.config); + return to_chat_response(await read_success_json(response, this.config.name), this.config); } } @@ -143,43 +152,11 @@ function resolve_api_key(config: ProviderConfig): string | undefined { return first_non_empty([config.api_key, from_named_env]); } -function first_non_empty(values: ReadonlyArray): string | undefined { - for (const value of values) { - if (value !== undefined && value.length > 0) { - return value; - } - } - return undefined; -} - function build_endpoint(config: ProviderConfig): string { const base = config.base_url ?? DEFAULT_BASE_URL; return base.endsWith("/") === true ? `${base}api/chat` : `${base}/api/chat`; } -function build_headers(api_key: string | undefined): Record { - const headers: Record = { "content-type": "application/json" }; - if (api_key !== undefined) { - headers.authorization = `Bearer ${api_key}`; - } - return headers; -} - -function build_abort_signal(options: ChatOptions | undefined, timeout_ms: number | undefined): AbortSignal | undefined { - const signals: AbortSignal[] = []; - if (timeout_ms !== undefined && timeout_ms > 0) { - signals.push(AbortSignal.timeout(timeout_ms)); - } - if (options?.signal !== undefined) { - signals.push(options.signal); - } - const [only_signal] = signals; - if (only_signal !== undefined && signals.length === 1) { - return only_signal; - } - return AbortSignal.any(signals); -} - function to_ollama_messages(messages: readonly Message[]): OllamaMessageDto[] { const wire: OllamaMessageDto[] = []; for (const message of messages) { @@ -267,92 +244,6 @@ function build_wire_options( return has_any === true ? wire_options : undefined; } -function build_request_init(api_key: string | undefined, body: string, signal: AbortSignal | undefined): RequestInit { - return { - method: "POST", - headers: build_headers(api_key), - body, - ...(signal !== undefined ? { signal } : {}), - }; -} - -async function do_fetch(fetch_fn: typeof fetch, url: string, init: RequestInit, provider_name: string): Promise { - try { - return await fetch_fn(url, init); - } catch (error) { - const label = is_abort_like(error) === true ? "request aborted or timed out" : "fetch failed"; - throw new ProviderError({ - kind: "network", - provider_name, - message: `${label}: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_response_text(response: Response, provider_name: string): Promise { - try { - return await response.text(); - } catch (error) { - throw new ProviderError({ - kind: "network", - provider_name, - message: `failed to read response body: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_success_json(response: Response, provider_name: string): Promise { - const text = await read_response_text(response, provider_name); - const dto = safe_json_parse(text); - if (dto === undefined) { - throw new ProviderError({ - kind: "bad_request", - provider_name, - message: `unparseable success response: ${truncate_text(text, MAX_ERROR_BODY_CHARS)}`, - }); - } - return dto; -} - -function parse_retry_after_ms(response: Response): number | undefined { - const raw = response.headers.get("retry-after"); - if (raw === null) { - return undefined; - } - const seconds = Number(raw); - if (Number.isFinite(seconds) === false || seconds < 0) { - return undefined; - } - return Math.round(seconds * 1000); -} - -function status_to_error_kind(status: number, body_text: string): ProviderErrorKind { - if (status === 401 || status === 403) { - return "auth"; - } - if (status === 429 || status >= 500) { - return "rate_limit"; - } - if (status === 413 || (status === 400 && OVERFLOW_BODY_PATTERN.test(body_text) === true)) { - return "overflow"; - } - return "bad_request"; -} - -async function to_http_error(response: Response, provider_name: string): Promise { - const body_text = truncate_text(await read_response_text(response, provider_name), MAX_ERROR_BODY_CHARS); - const retry_after_ms = parse_retry_after_ms(response); - return new ProviderError({ - kind: status_to_error_kind(response.status, body_text), - provider_name, - message: `${provider_name} http ${response.status}: ${body_text}`, - status: response.status, - ...(retry_after_ms !== undefined ? { retry_after_ms } : {}), - }); -} - function to_chat_response(dto: OllamaChatResponseDto, config: ProviderConfig): ChatResult { if (dto.error !== undefined && dto.error.length > 0) { throw new ProviderError({ @@ -459,25 +350,4 @@ function map_done_reason(done_reason: string | undefined, has_tool_calls: boolea return "length"; } return "unknown"; -} - -function is_record(value: unknown): value is Record { - return typeof value === "object" && value !== null; -} - -function describe_error(error: unknown): string { - return error instanceof Error ? error.message : String(error); -} - -function error_name(error: unknown): string | undefined { - if (typeof error === "object" && error !== null && "name" in error) { - const name = (error as { name?: unknown }).name; - return typeof name === "string" ? name : undefined; - } - return undefined; -} - -function is_abort_like(error: unknown): boolean { - const name = error_name(error); - return name === "AbortError" || name === "TimeoutError"; } \ No newline at end of file diff --git a/src/providers/openai.ts b/src/providers/openai.ts index 4a32884..afe37cf 100644 --- a/src/providers/openai.ts +++ b/src/providers/openai.ts @@ -1,4 +1,14 @@ -import { safe_json_parse, safe_stringify, truncate_text } from "../util/json.js"; +import { safe_json_parse, safe_stringify } from "../util/json.js"; +import { + build_abort_signal, + build_bearer_headers, + build_request_init, + do_fetch, + first_non_empty, + is_record, + read_success_json, + to_http_error, +} from "./http.js"; import { ProviderError } from "./types.js"; import type { AssistantMessage, @@ -8,7 +18,6 @@ import type { LLMProvider, Message, ProviderConfig, - ProviderErrorKind, ToolCall, ToolDefinition, ToolMessage, @@ -26,9 +35,6 @@ function next_tool_call_id(): string { const DEFAULT_BASE_URL = "https://api.openai.com/v1"; const WELL_KNOWN_HOST = "api.openai.com"; const WELL_KNOWN_KEY_ENV = "OPENAI_API_KEY"; -const MAX_ERROR_BODY_CHARS = 500; -const OVERFLOW_BODY_PATTERN = - /context.?length|maximum context|prompt(?: is)? too (?:long|large)|token.?limit|context window|too many tokens|exceed.{0,30}context limit/i; const UNPARSEABLE_ARGS_NOTE = "[unparseable tool arguments]"; const TRUNCATED_TOOL_CALLS_NOTE = "[truncated tool call omitted]"; const REASONING_MODEL_PATTERN = /^(o[1-9]|o[1-9]-|gpt-5)/i; @@ -124,7 +130,7 @@ export class OpenAICompatProvider implements LLMProvider { this.config.fetch_fn ?? fetch, build_endpoint(this.config), build_request_init( - api_key, + build_bearer_headers(api_key), safe_stringify(build_request_body(this.config.model, messages, tools, options)), build_abort_signal(options, this.config.timeout_ms), ), @@ -133,7 +139,7 @@ export class OpenAICompatProvider implements LLMProvider { if (response.ok === false) { throw await to_http_error(response, this.config.name); } - return parse_chat_response(await read_success_json(response, this.config.name), this.config); + return parse_chat_response(await read_success_json(response, this.config.name), this.config); } } @@ -159,43 +165,11 @@ function resolve_api_key(config: ProviderConfig): string | undefined { return first_non_empty([config.api_key, from_named_env, from_well_known_env]); } -function first_non_empty(values: ReadonlyArray): string | undefined { - for (const value of values) { - if (value !== undefined && value.length > 0) { - return value; - } - } - return undefined; -} - function build_endpoint(config: ProviderConfig): string { const base = config.base_url ?? DEFAULT_BASE_URL; return base.endsWith("/") === true ? `${base}chat/completions` : `${base}/chat/completions`; } -function build_headers(api_key: string | undefined): Record { - const headers: Record = { "content-type": "application/json" }; - if (api_key !== undefined) { - headers.authorization = `Bearer ${api_key}`; - } - return headers; -} - -function build_abort_signal(options: ChatOptions | undefined, timeout_ms: number | undefined): AbortSignal | undefined { - const signals: AbortSignal[] = []; - if (timeout_ms !== undefined && timeout_ms > 0) { - signals.push(AbortSignal.timeout(timeout_ms)); - } - if (options?.signal !== undefined) { - signals.push(options.signal); - } - const [only_signal] = signals; - if (only_signal !== undefined && signals.length === 1) { - return only_signal; - } - return AbortSignal.any(signals); -} - function to_openai_messages(messages: readonly Message[]): OpenAiMessageDto[] { const wire: OpenAiMessageDto[] = []; for (const message of messages) { @@ -263,92 +237,6 @@ function is_reasoning_model(model: string): boolean { return REASONING_MODEL_PATTERN.test(model) === true; } -function build_request_init(api_key: string | undefined, body: string, signal: AbortSignal | undefined): RequestInit { - return { - method: "POST", - headers: build_headers(api_key), - body, - ...(signal !== undefined ? { signal } : {}), - }; -} - -async function do_fetch(fetch_fn: typeof fetch, url: string, init: RequestInit, provider_name: string): Promise { - try { - return await fetch_fn(url, init); - } catch (error) { - const label = is_abort_like(error) === true ? "request aborted or timed out" : "fetch failed"; - throw new ProviderError({ - kind: "network", - provider_name, - message: `${label}: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_response_text(response: Response, provider_name: string): Promise { - try { - return await response.text(); - } catch (error) { - throw new ProviderError({ - kind: "network", - provider_name, - message: `failed to read response body: ${describe_error(error)}`, - cause: error, - }); - } -} - -async function read_success_json(response: Response, provider_name: string): Promise { - const text = await read_response_text(response, provider_name); - const dto = safe_json_parse(text); - if (dto === undefined) { - throw new ProviderError({ - kind: "bad_request", - provider_name, - message: `unparseable success response: ${truncate_text(text, MAX_ERROR_BODY_CHARS)}`, - }); - } - return dto; -} - -function parse_retry_after_ms(response: Response): number | undefined { - const raw = response.headers.get("retry-after"); - if (raw === null) { - return undefined; - } - const seconds = Number(raw); - if (Number.isFinite(seconds) === false || seconds < 0) { - return undefined; - } - return Math.round(seconds * 1000); -} - -function status_to_error_kind(status: number, body_text: string): ProviderErrorKind { - if (status === 401 || status === 403) { - return "auth"; - } - if (status === 429 || status >= 500) { - return "rate_limit"; - } - if (status === 413 || (status === 400 && OVERFLOW_BODY_PATTERN.test(body_text) === true)) { - return "overflow"; - } - return "bad_request"; -} - -async function to_http_error(response: Response, provider_name: string): Promise { - const body_text = truncate_text(await read_response_text(response, provider_name), MAX_ERROR_BODY_CHARS); - const retry_after_ms = parse_retry_after_ms(response); - return new ProviderError({ - kind: status_to_error_kind(response.status, body_text), - provider_name, - message: `${provider_name} http ${response.status}: ${body_text}`, - status: response.status, - ...(retry_after_ms !== undefined ? { retry_after_ms } : {}), - }); -} - function parse_chat_response(dto: OpenAiChatResponseDto, config: ProviderConfig): ChatResult { const choice = dto.choices?.[0]; if (choice === undefined || choice.message === undefined) { @@ -424,25 +312,4 @@ function map_finish_reason(raw: string | null | undefined): FinishReason { return "length"; } return "unknown"; -} - -function is_record(value: unknown): value is Record { - return typeof value === "object" && value !== null; -} - -function describe_error(error: unknown): string { - return error instanceof Error ? error.message : String(error); -} - -function error_name(error: unknown): string | undefined { - if (typeof error === "object" && error !== null && "name" in error) { - const name = (error as { name?: unknown }).name; - return typeof name === "string" ? name : undefined; - } - return undefined; -} - -function is_abort_like(error: unknown): boolean { - const name = error_name(error); - return name === "AbortError" || name === "TimeoutError"; } \ No newline at end of file From fe394bd5f0733ceaefc7a5d41a5db78c7873e9d6 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 05:31:39 +0000 Subject: [PATCH 25/43] fix: Ctrl+C/Esc cancel, atomic config update, test tmp cleanup (#133) - CLI one-shot and chat pass an abort signal to agent.run. Ctrl+C aborts the one-shot run (exit 1) or the running chat turn (session kept); at the chat prompt it ends chat. A second Ctrl+C while cancelling exits 130. - TUI: Esc cancels a running turn; an aborted run shows "run cancelled". - write_lich_config update writes a temp file and renames it into place, keeping the existing mode, so readers never see a partial config. - gatekeeper, tools, run_tests, skills_search and kanban_audit tests remove the dirs they create under test/.tmp. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 7 ++ docs/user-guide/cli.md | 5 +- docs/user-guide/tui.md | 4 +- src/cli.ts | 56 +++++++++++++-- src/cli_config.ts | 36 +++++++--- src/tui/app.tsx | 11 ++- src/tui/command_bar.tsx | 10 ++- src/tui/state.ts | 5 +- test/cli_chat.test.ts | 1 + test/cli_config.test.ts | 30 +++++++- test/cli_plugins.test.ts | 2 + test/cli_sigint.test.ts | 128 +++++++++++++++++++++++++++++++++++ test/gatekeeper.test.ts | 28 +++++--- test/kanban_audit.test.ts | 8 ++- test/run_tests.test.ts | 5 +- test/skills_search.test.ts | 5 +- test/tools.test.ts | 8 +++ test/tui.test.ts | 10 +++ test/tui_command_bar.test.ts | 69 +++++++++++++++++++ 19 files changed, 389 insertions(+), 39 deletions(-) create mode 100644 test/cli_sigint.test.ts create mode 100644 test/tui_command_bar.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 891fcc0..a61a618 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,13 @@ ## Unreleased +- Ctrl+C in a one-shot run or in `lich chat` cancels the run or turn + through its abort signal; chat keeps the session and returns to the prompt. + A second Ctrl+C quits at once (exit `130`). In the TUI, Esc cancels a running + turn (#133). +- Updating `.lich/config.json` (`lich mcp ...`) writes a temp file and renames + it into place, keeping the file's mode, so a crash or concurrent reader + never sees a half-written config (#133). - `http_request` reports a non-2xx status as `ok: false` (`error: http_`) and still returns the status and body in `output`. - `terminal` uses config `terminal_timeout_ms` as its default command timeout diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index eba74a3..8b6e476 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -26,8 +26,8 @@ lich --version # package.json version (published package and this tree: - **Bare `lich`** opens the same TUI as `lich tui`. It does not print usage. On a TTY, if neither `.lich/config.json` nor `~/.config/lich/config.json` exists, a setup wizard runs first (name, provider, optional gateway env-var names, optional plugins) and writes `.lich/config.json` once. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. An existing config in that chain skips the wizard and is not replaced. Non-TTY stdin skips the wizard and prints guidance instead of hanging. `lich --help` still prints usage. - **`lich init`** writes that starter file without prompts, using the same writer as the wizard. Existing flags such as `--model` are written into the file and win over `LICH_MODEL`. It never overwrites an existing `.lich/config.json`. `.lich/` is gitignored. -- **One-shot** joins all positional words into a single task, runs the agent loop, prints the final answer to stdout, and exits. Progress (turn numbers, tool results) goes to stderr. -- **Chat** is a readline REPL over one long-lived agent: each line is a turn, memory persists across lines, and an empty line, `/exit`, or `/quit` ends the session. After each turn it prints a `[turns N | tokens M]` footer. +- **One-shot** joins all positional words into a single task, runs the agent loop, prints the final answer to stdout, and exits. Progress (turn numbers, tool results) goes to stderr. Ctrl+C cancels the run (exit `1`); a second Ctrl+C quits at once (exit `130`). +- **Chat** is a readline REPL over one long-lived agent: each line is a turn, memory persists across lines, and an empty line, `/exit`, or `/quit` ends the session. After each turn it prints a `[turns N | tokens M]` footer. Ctrl+C cancels the running turn and keeps the session; at the prompt it ends chat. A second Ctrl+C while a turn is still cancelling quits at once (exit `130`). - **TUI** launches the ink interface. See the [TUI guide](tui.md). - **Serve** starts a loopback-only WebSocket JSON-RPC server with the same Agent/config resolution as TUI/chat. On listen it prints one JSON line `{"port":…,"token":…}` to stdout for clients (e.g. ossuary) to parse. `--host` defaults to `127.0.0.1` and must be loopback; `--port` defaults to `0` (ephemeral). See the [serve architecture note](../architecture/serve.md). - **Ossuary** opens the Electron desktop shell when `apps/ossuary` is present (git clone). It does not require a model flag at the CLI entry — the window spawns `lich serve` using `--work-dir` / `LICH_WORK_DIR` (default cwd). Missing `apps/ossuary` fails with a clone hint. See the [Ossuary guide](ossuary.md). @@ -204,6 +204,7 @@ jq -r 'select(.kind=="message") | "\(.message.role): \(.message.content // "(too | --- | --- | | `0` | Success: final answer produced (also `--help`, `--version`, `config`, `init`, a successful `mcp` action, `update` when nothing newer is installed or the install succeeds, and a TUI that exits cleanly). | | `1` | Any failure: unknown flag, missing model, unreadable config, provider error after failover, aborted run, budget exhaustion, non-TTY bare `lich`, or a cancelled setup wizard. | +| `130` | A second Ctrl+C while a one-shot run or chat turn was still cancelling. | ## Log levels diff --git a/docs/user-guide/tui.md b/docs/user-guide/tui.md index 823d5cc..81f2dbb 100644 --- a/docs/user-guide/tui.md +++ b/docs/user-guide/tui.md @@ -9,7 +9,7 @@ lich # front door: TUI, plus a first-run setup wizard when no config exi lich tui # same TUI, no wizard. From a clone: bun src/cli.ts tui ``` -The TUI needs a TTY and a resolvable provider (same resolution as every mode). On startup it prints a dim header from the active theme welcome string, e.g. `⚱ lich v0.8.0 — the agent that will not stay dead · qwen3:8b (ollama)`. `{version}` is `LICH_VERSION` from `package.json`. That banner is the only tagline placement. Quit with `/exit`, `/quit`, `/q`, or Ctrl+C. +The TUI needs a TTY and a resolvable provider (same resolution as every mode). On startup it prints a dim header from the active theme welcome string, e.g. `⚱ lich v0.8.0 — the agent that will not stay dead · qwen3:8b (ollama)`. `{version}` is `LICH_VERSION` from `package.json`. That banner is the only tagline placement. Quit with `/exit`, `/quit`, `/q`, or Ctrl+C. Esc cancels a running turn. ## Anatomy @@ -25,7 +25,7 @@ model qwen3:8b · turns 2 · tokens 1,204 · [dormant] · /path/.lich/sessions/. - **Header** — theme welcome string (version, model, provider kind). The tagline appears only here. - **Transcript** — user lines (`mortal ›` by default), replies (`lich ›`, or the theme `response_label`), tool rows (`⏺ name(args)` with a result line), and meta notices (`· context compressed — memories distilled ...`, `· error: ...`). The view keeps the newest 50 blocks; older lines scroll out of the transcript (session JSONLs still hold everything — see [limitations](#known-limitations)). -- **Input row** — `› ` when idle, `… ` while the agent works; Enter submits, Backspace edits, pasted newlines collapse to spaces. +- **Input row** — `› ` when idle, `… ` while the agent works; Enter submits, Backspace edits, pasted newlines collapse to spaces, and Esc cancels a running turn (a `· run cancelled` notice follows). - **Status bar** — see below. ## Slash commands diff --git a/src/cli.ts b/src/cli.ts index bd75e50..3ea4cde 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -436,10 +436,13 @@ export async function run_one_shot(config: unknown, input: string): Promise { +/** Ctrl+C aborts the run; a second Ctrl+C exits at once. Returns the remover. */ +function abort_on_sigint(controller: AbortController): () => void { + const on_sigint = (): void => { + if (controller.signal.aborted === true) { + process.exit(130); + } + process.stderr.write("[lich] cancelling (Ctrl+C again to quit)\n"); + controller.abort(); + }; + process.on("SIGINT", on_sigint); + return () => { + process.off("SIGINT", on_sigint); + }; +} + +async function run_chat_turn( + agent: Agent, + input: string, + history: readonly Message[], + signal: AbortSignal, +): Promise { const stop_progress = attach_progress(agent.events); try { - const result = await agent.run({ input, history }); + const result = await agent.run({ input, history, signal }); const final = result.outcome.final; if (final !== undefined && final.content.length > 0) { process.stdout.write(`${final.content}\n`); } + if (result.outcome.stopped_reason === "aborted") { + process.stderr.write("[lich] turn cancelled\n"); + } process.stdout.write(`[turns ${result.outcome.turns_used} | tokens ${result.usage_total.total_tokens}]\n`); return result.messages; } catch (error) { @@ -487,6 +513,22 @@ export async function run_chat(config: unknown): Promise { const theme = load_theme(agent.config.theme); const rl = createInterface({ input: process.stdin, output: process.stdout }); let history: Message[] = []; + // Ctrl+C cancels the running turn (twice quits); at the prompt it ends chat. + // readline reports it on a TTY, the process signal covers piped stdin. + let turn: AbortController | undefined; + const on_sigint = (): void => { + if (turn === undefined) { + rl.close(); + return; + } + if (turn.signal.aborted === true) { + process.exit(130); + } + process.stderr.write("[lich] cancelling turn (Ctrl+C again to quit)\n"); + turn.abort(); + }; + rl.on("SIGINT", on_sigint); + process.on("SIGINT", on_sigint); try { process.stdout.write("> "); for await (const line of rl) { @@ -499,11 +541,17 @@ export async function run_chat(config: unknown): Promise { process.stdout.write(`${theme.glyph} ${theme.goodbye}\n`); return 0; } - history = await run_chat_turn(agent, command, history); + turn = new AbortController(); + try { + history = await run_chat_turn(agent, command, history, turn.signal); + } finally { + turn = undefined; + } process.stdout.write("> "); } return 0; } finally { + process.off("SIGINT", on_sigint); rl.close(); agent.close(); } diff --git a/src/cli_config.ts b/src/cli_config.ts index b0dfbbb..ab9b0a1 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -2,7 +2,7 @@ * Config-file discovery for the CLI: an ordered search chain, safe loading, * and a starter template so `lich tui` works with zero environment setup. */ -import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { chmodSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs"; import { homedir } from "node:os"; import path from "node:path"; import { safe_json_parse } from "./util/json.js"; @@ -137,19 +137,39 @@ export function write_lich_config( if (existsSync(file) === true && update !== true) { return already_exists(file); } + const text = `${JSON.stringify(config, null, 2)}\n`; + if (update === true) { + replace_file(file, text); + return { path: file, written: true, message: `updated ${file}` }; + } try { - writeFileSync(file, `${JSON.stringify(config, null, 2)}\n`, { - encoding: "utf8", - flag: update === true ? "w" : "wx", - }); + writeFileSync(file, text, { encoding: "utf8", flag: "wx" }); } catch (error) { - if (update !== true && is_eexist(error) === true) { + if (is_eexist(error) === true) { return already_exists(file); } throw error; } - const verb = update === true ? "updated" : "wrote"; - return { path: file, written: true, message: `${verb} ${file}` }; + return { path: file, written: true, message: `wrote ${file}` }; +} + +/** + * Write a sibling temp file, then rename it over `file`, so readers never see + * a partial config. The temp file takes the existing file's mode. + */ +function replace_file(file: string, text: string): void { + const temp = `${file}.${process.pid}.tmp`; + const mode = statSync(file, { throwIfNoEntry: false })?.mode; + try { + writeFileSync(temp, text, "utf8"); + if (mode !== undefined) { + chmodSync(temp, mode); + } + renameSync(temp, file); + } catch (error) { + rmSync(temp, { force: true }); + throw error; + } } function already_exists(file: string): LichConfigWriteResult { diff --git a/src/tui/app.tsx b/src/tui/app.tsx index 5bc21d0..4d3040d 100644 --- a/src/tui/app.tsx +++ b/src/tui/app.tsx @@ -55,6 +55,7 @@ type SetHistory = (messages: readonly Message[]) => void; interface AgentRunControls { readonly start_message_run: (text: string) => void; + readonly cancel_run: () => void; readonly set_history: SetHistory; } @@ -167,9 +168,13 @@ function use_agent_run( [agent, add_blocks, finish_run, session, set_blocks, set_state, theme], ); + const cancel_run = useCallback((): void => { + controller_ref.current?.abort(); + }, []); + useEffect(() => () => controller_ref.current?.abort(), []); - return { start_message_run, set_history }; + return { start_message_run, cancel_run, set_history }; } /** Slash-command dispatch: pure client-side actions, never hits the agent. */ @@ -289,7 +294,7 @@ export function TuiApp({ agent, theme, session, initial_history, resumed_id }: T set_blocks((current) => [...current, ...added].slice(-HISTORY_CAP)); }, []); - const { start_message_run, set_history } = use_agent_run( + const { start_message_run, cancel_run, set_history } = use_agent_run( agent, theme, add_blocks, @@ -334,7 +339,7 @@ export function TuiApp({ agent, theme, session, initial_history, resumed_id }: T {resume_line === undefined ? banner : `${banner}\n${resume_line}`} - + ); } diff --git a/src/tui/command_bar.tsx b/src/tui/command_bar.tsx index a125d73..b590f6f 100644 --- a/src/tui/command_bar.tsx +++ b/src/tui/command_bar.tsx @@ -1,7 +1,8 @@ /** * Input row: printable characters accumulate in a buffer, Enter submits, * Backspace/Delete edits, Up/Down walk a 20-entry recall ring, and pasted - * newlines collapse to spaces. Ctrl+C is left to ink's default handling. + * newlines collapse to spaces. Esc cancels a running turn. Ctrl+C is left to + * ink's default handling. */ import { useState } from "react"; import { Box, Text, useInput } from "ink"; @@ -11,6 +12,7 @@ const INPUT_HISTORY_CAP = 20; interface CommandBarProps { readonly busy: boolean; readonly on_submit: (text: string) => void; + readonly on_cancel?: () => void; } /** Push onto a capped ring (newest first) without mutating the source. */ @@ -18,7 +20,7 @@ function push_history(ring: readonly string[], entry: string): readonly string[] return [entry, ...ring.filter((item) => item !== entry)].slice(0, INPUT_HISTORY_CAP); } -export function CommandBar({ busy, on_submit }: CommandBarProps): React.JSX.Element { +export function CommandBar({ busy, on_submit, on_cancel }: CommandBarProps): React.JSX.Element { const [buffer, set_buffer] = useState(""); const [recall_ring, set_recall_ring] = useState([]); const [recall_index, set_recall_index] = useState(undefined); @@ -72,6 +74,10 @@ export function CommandBar({ busy, on_submit }: CommandBarProps): React.JSX.Elem set_buffer((current) => current.slice(0, -1)); return; } + if (key.escape === true && busy === true) { + on_cancel?.(); + return; + } if (key.ctrl === true || key.escape === true || key.tab === true || key.meta === true) { return; } diff --git a/src/tui/state.ts b/src/tui/state.ts index 2478d62..e8fb123 100644 --- a/src/tui/state.ts +++ b/src/tui/state.ts @@ -235,6 +235,9 @@ export function run_notice_blocks(result: AgentRunResult, theme: ThemeSpec): His if (result.outcome.stopped_reason === "budget") { blocks.push({ role: "error", lines: [`\u00b7 ${fill_template(theme.notices.budget_exhausted, {})}`] }); } + if (result.outcome.stopped_reason === "aborted") { + blocks.push({ role: "error", lines: ["\u00b7 run cancelled"] }); + } const final_block = result.outcome.final === undefined ? undefined : assistant_result_block(result.outcome.final, theme); if (final_block !== undefined) { blocks.push(final_block); @@ -346,5 +349,5 @@ export const SLASH_COMMAND_NAMES: readonly string[] = [ export const HELP_LINES: readonly string[] = [ "commands: /help /model /usage /clear /sessions /resume /exit (aliases: /quit /q)", - "enter submits \u00b7 backspace deletes \u00b7 up/down recalls history \u00b7 pasted newlines become spaces", + "enter submits \u00b7 backspace deletes \u00b7 up/down recalls history \u00b7 pasted newlines become spaces \u00b7 esc cancels a running turn", ]; \ No newline at end of file diff --git a/test/cli_chat.test.ts b/test/cli_chat.test.ts index afb4d82..b8ed1b0 100644 --- a/test/cli_chat.test.ts +++ b/test/cli_chat.test.ts @@ -19,6 +19,7 @@ vi.mock("node:readline", () => ({ } }, close: () => undefined, + on: () => undefined, }), })); diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 8e9b796..71d6353 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -2,11 +2,19 @@ * Unit tests for CLI config-file discovery: search chain, explicit paths, * safe JSON loading, and the starter template printed by `lich config`. */ -import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { chmodSync, existsSync, linkSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs"; import { homedir, tmpdir } from "node:os"; import path from "node:path"; import { afterEach, describe, expect, it } from "vitest"; -import { config_search_paths, config_template, ensure_lich_config_dir, find_config_file, load_config } from "../src/cli_config.js"; +import { + config_search_paths, + config_template, + ensure_lich_config_dir, + find_config_file, + load_config, + project_config_path, + write_lich_config, +} from "../src/cli_config.js"; const TEST_TMP_ROOT = path.resolve("test/.tmp/cli-config"); const created: string[] = []; @@ -177,6 +185,24 @@ describe("ensure_lich_config_dir", () => { }); }); +describe("write_lich_config update", () => { + it("replaces the file by rename instead of rewriting it in place, keeping its mode", () => { + const dir = make_temp_dir("atomic"); + write_lich_config(dir, { version: 1 }); + const file = project_config_path(dir); + chmodSync(file, 0o600); + // A hard link shares the old inode: an in-place write would change it too. + const old_inode = path.join(dir, "old-inode.json"); + linkSync(file, old_inode); + const result = write_lich_config(dir, { version: 2 }, true); + expect(result.written).toBe(true); + expect(JSON.parse(readFileSync(file, "utf8"))).toEqual({ version: 2 }); + expect(JSON.parse(readFileSync(old_inode, "utf8"))).toEqual({ version: 1 }); + expect(statSync(file).mode & 0o777).toBe(0o600); + expect(readdirSync(path.dirname(file))).toEqual(["config.json"]); + }); +}); + describe("integration shape", () => { it("uses test scratch space and never the os temp root", () => { expect(TEST_TMP_ROOT.startsWith(path.resolve("test/.tmp"))).toBe(true); diff --git a/test/cli_plugins.test.ts b/test/cli_plugins.test.ts index 8553063..62858d9 100644 --- a/test/cli_plugins.test.ts +++ b/test/cli_plugins.test.ts @@ -21,6 +21,7 @@ vi.mock("node:readline", () => { const createInterface = (): { question: (prompt: string, callback: (line: string) => void) => void; once: () => void; + on: () => void; removeListener: () => void; close: () => void; [Symbol.asyncIterator]: () => AsyncIterator; @@ -29,6 +30,7 @@ vi.mock("node:readline", () => { callback(""); }, once: () => undefined, + on: () => undefined, removeListener: () => undefined, close: () => undefined, async *[Symbol.asyncIterator]() { diff --git a/test/cli_sigint.test.ts b/test/cli_sigint.test.ts new file mode 100644 index 0000000..2e6cc40 --- /dev/null +++ b/test/cli_sigint.test.ts @@ -0,0 +1,128 @@ +/** + * Ctrl+C in the CLI (#133): one-shot aborts the run; chat cancels the running + * turn and keeps the session, and at the prompt ends chat. + */ +import { EventEmitter } from "node:events"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import type { Message } from "../src/providers/types.js"; + +const sigint_state = vi.hoisted(() => ({ + lines: [] as string[], + rl: undefined as (EventEmitter & { closed: boolean }) | undefined, + signals: [] as AbortSignal[], + runs: [] as string[], + finished: 0, +})); + +vi.mock("node:readline", async () => { + const { EventEmitter: Emitter } = await import("node:events"); + return { + createInterface: () => { + const rl = Object.assign(new Emitter(), { closed: false }); + sigint_state.rl = rl; + const closed = new Promise((resolve) => rl.once("close", resolve)); + return Object.assign(rl, { + async *[Symbol.asyncIterator]() { + for (const line of sigint_state.lines) { + yield line; + } + await closed; + }, + close: () => { + if (rl.closed === false) { + rl.closed = true; + rl.emit("close"); + } + }, + }); + }, + }; +}); + +vi.mock("../src/agent/agent.js", () => ({ + create_agent_with_plugins: async () => ({ + config: { theme: "lich" }, + events: { on: () => () => undefined }, + close: () => undefined, + run: async (options: { input: string; history?: readonly Message[]; signal?: AbortSignal }) => { + const signal = options.signal; + if (signal === undefined) { + throw new Error("no signal"); + } + sigint_state.signals.push(signal); + sigint_state.runs.push(options.input); + await new Promise((resolve) => signal.addEventListener("abort", resolve, { once: true })); + sigint_state.finished += 1; + const messages: Message[] = [...(options.history ?? []), { role: "user", content: options.input }]; + return { + outcome: { final: undefined, turns_used: 1, stopped_reason: "aborted", messages }, + messages, + usage_total: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, + }; + }, + }), +})); + +vi.mock("../src/util/theme.js", () => ({ + load_theme: () => ({ glyph: "*", goodbye: "bye", notices: { budget_exhausted: "" } }), + notice_flavor: () => "", +})); + +import { run_chat, run_one_shot } from "../src/cli.js"; + +const CONFIG = { providers: [{ kind: "openai_compat", name: "m", model: "m" }] }; + +afterEach(() => { + vi.restoreAllMocks(); + sigint_state.lines = []; + sigint_state.rl = undefined; + sigint_state.signals = []; + sigint_state.runs = []; + sigint_state.finished = 0; +}); + +async function until(check: () => boolean): Promise { + for (let i = 0; i < 200 && check() === false; i += 1) { + await new Promise((resolve) => setTimeout(resolve, 5)); + } + expect(check()).toBe(true); +} + +/** Call only the SIGINT listeners the code under test added, never the runner's own. */ +function send_sigint(baseline: readonly unknown[]): void { + for (const listener of process.listeners("SIGINT")) { + if (baseline.includes(listener) === false) { + listener("SIGINT"); + } + } +} + +describe("CLI Ctrl+C", () => { + it("one-shot: SIGINT aborts the run, exits 1 and removes its listener", async () => { + vi.spyOn(process.stderr, "write").mockImplementation(() => true); + const baseline = process.listeners("SIGINT"); + const pending = run_one_shot(CONFIG, "slow"); + await until(() => sigint_state.signals.length === 1); + send_sigint(baseline); + expect(await pending).toBe(1); + expect(sigint_state.signals[0]?.aborted).toBe(true); + expect(process.listeners("SIGINT")).toEqual(baseline); + }); + + it("chat: Ctrl+C cancels the running turn, then ends chat at the prompt", async () => { + vi.spyOn(process.stderr, "write").mockImplementation(() => true); + vi.spyOn(process.stdout, "write").mockImplementation(() => true); + const baseline = process.listeners("SIGINT"); + sigint_state.lines = ["slow"]; + const pending = run_chat(CONFIG); + await until(() => sigint_state.signals.length === 1); + sigint_state.rl?.emit("SIGINT"); + await until(() => sigint_state.finished === 1); + expect(sigint_state.signals[0]?.aborted).toBe(true); + expect(sigint_state.rl?.closed).toBe(false); + send_sigint(baseline); + expect(await pending).toBe(0); + expect(sigint_state.runs).toEqual(["slow"]); + expect(process.listeners("SIGINT")).toEqual(baseline); + }); +}); diff --git a/test/gatekeeper.test.ts b/test/gatekeeper.test.ts index cc4a70f..a2919e0 100644 --- a/test/gatekeeper.test.ts +++ b/test/gatekeeper.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { execFile } from "node:child_process"; -import { access, chmod, mkdir, mkdtemp, readFile, writeFile } from "node:fs/promises"; +import { access, chmod, mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import path from "node:path"; import { gatekeeper_plugin } from "../src/plugins/builtin/gatekeeper.plugin.js"; import type { AfterToolCallInfo, BeforeToolCallInfo, HookContext, Plugin } from "../src/plugins/types.js"; @@ -11,6 +11,15 @@ import { ToolRegistry } from "../src/tools/registry.js"; import { TMP_BASE } from "./helpers/tmp_base.js"; let tmp_root: string; +const fixture_dirs: string[] = []; + +/** A fresh dir under test/.tmp, removed after the test. */ +async function fixture_dir(prefix: string): Promise { + await mkdir(TMP_BASE, { recursive: true }); + const dir = await mkdtemp(path.join(TMP_BASE, prefix)); + fixture_dirs.push(dir); + return dir; +} /** Run git in `dir`; resolve exit code + combined output. */ function git(dir: string, args: readonly string[]): Promise<{ code: number; out: string }> { @@ -49,8 +58,7 @@ async function wait_for_text(file: string): Promise { /** Seed a fresh repo with a root commit (A3) and return its dir. */ async function seeded_repo(): Promise { - await mkdir(TMP_BASE, { recursive: true }); - const dir = await mkdtemp(path.join(TMP_BASE, "gatekeeper-")); + const dir = await fixture_dir("gatekeeper-"); await git(dir, ["init"]); await git(dir, ["-c", "user.name=t", "-c", "user.email=t@t", "commit", "--allow-empty", "-m", "seed"]); return dir; @@ -199,7 +207,7 @@ describe("gatekeeper plugin", () => { }); it("keeps probe plugins from reading the gatekeeper's state (sub-map isolation)", async () => { - const probe_dir = await mkdtemp(path.join(TMP_BASE, "probe-")); + const probe_dir = await fixture_dir("probe-"); const probe_source = ` export default { name: "probe", @@ -225,7 +233,7 @@ describe("gatekeeper plugin", () => { }); it("registers before config plugins so first-wins shadows a colliding tool", async () => { - const shadow_dir = await mkdtemp(path.join(TMP_BASE, "shadow-")); + const shadow_dir = await fixture_dir("shadow-"); const shadow_source = ` export default { name: "shadow", @@ -289,7 +297,7 @@ describe("gatekeeper plugin", () => { }); it("refuses to commit when HEAD is unreachable (unborn branch, A3)", async () => { - const fresh = await mkdtemp(path.join(TMP_BASE, "unborn-")); + const fresh = await fixture_dir("unborn-"); await git(fresh, ["init"]); const registry = new ToolRegistry(); const gatekeeper_loaded = { plugin: gatekeeper_plugin(true), entry: "builtin:gatekeeper" }; @@ -370,7 +378,7 @@ describe("gatekeeper plugin", () => { }); it("loader hard-rejects builtin-colliding plugin names (A5)", async () => { - const collide_dir = await mkdtemp(path.join(TMP_BASE, "collide-")); + const collide_dir = await fixture_dir("collide-"); const collide_source = ` export default { name: "gatekeeper", @@ -390,6 +398,8 @@ describe("gatekeeper plugin", () => { }); }); -afterEach(() => { - // fixture repos live under test/.tmp (gitignored); no manual cleanup needed +afterEach(async () => { + for (const dir of fixture_dirs.splice(0)) { + await rm(dir, { recursive: true, force: true }); + } }); \ No newline at end of file diff --git a/test/kanban_audit.test.ts b/test/kanban_audit.test.ts index 1de72b4..aa06c37 100644 --- a/test/kanban_audit.test.ts +++ b/test/kanban_audit.test.ts @@ -6,9 +6,9 @@ * "audit complete". Also covers the item-list page-cap guard. */ import { spawnSync } from "node:child_process"; -import { chmodSync, mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import { chmodSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; import path from "node:path"; -import { beforeEach, describe, expect, it } from "vitest"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { TMP_BASE } from "./helpers/tmp_base.js"; const REPO = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); @@ -93,6 +93,10 @@ beforeEach(() => { write_issues(dir, OPEN_ISSUES); }); +afterEach(() => { + rmSync(dir, { recursive: true, force: true }); +}); + describe("kanban.sh audit", () => { it("does not flag a mention-only PR (bare #N in title/body)", () => { write_prs(dir, [{ number: 124, title: "docs: project wiki", body: "Part of #113", closingIssuesReferences: [] }]); diff --git a/test/run_tests.test.ts b/test/run_tests.test.ts index 8fee11e..b2e9134 100644 --- a/test/run_tests.test.ts +++ b/test/run_tests.test.ts @@ -1,7 +1,7 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { spawnSync } from "node:child_process"; import { existsSync } from "node:fs"; -import { mkdtemp } from "node:fs/promises"; +import { mkdtemp, rm } from "node:fs/promises"; import path from "node:path"; import { ToolExecutor } from "../src/tools/executor.js"; import { register_builtin_tools } from "../src/tools/builtin/index.js"; @@ -46,8 +46,9 @@ beforeEach(async () => { executor = new ToolExecutor(registry); }); -afterEach(() => { +afterEach(async () => { reset_test_command_runner(); + await rm(tmp_root, { recursive: true, force: true }); }); describe("run_tests", () => { diff --git a/test/skills_search.test.ts b/test/skills_search.test.ts index 874af0d..0d75b19 100644 --- a/test/skills_search.test.ts +++ b/test/skills_search.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { existsSync } from "node:fs"; -import { mkdir, mkdtemp, writeFile } from "node:fs/promises"; +import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; import { fileURLToPath } from "node:url"; import path from "node:path"; import { ToolExecutor } from "../src/tools/executor.js"; @@ -26,9 +26,10 @@ beforeEach(async () => { reset_docs_cache(); }); -afterEach(() => { +afterEach(async () => { reset_docs_search_cache(); reset_docs_cache(); + await rm(tmp_root, { recursive: true, force: true }); }); async function write_skill(name: string, content: string): Promise { diff --git a/test/tools.test.ts b/test/tools.test.ts index 13df948..1618561 100644 --- a/test/tools.test.ts +++ b/test/tools.test.ts @@ -98,8 +98,14 @@ beforeAll(async () => { tmp_root = await mkdtemp(path.join(TMP_BASE, "tools-")); }); +/** Dirs created outside tmp_root by the symlink tests. */ +const outside_dirs: string[] = []; + afterAll(async () => { await iter_rm(tmp_root); + for (const dir of outside_dirs) { + await iter_rm(dir); + } }); describe("guard", () => { @@ -402,6 +408,7 @@ describe("terminal", () => { describe("symlink confinement", () => { it("rejects file-tool paths that symlink outside work_dir", async () => { const outside = await mkdtemp(path.join(TMP_BASE, "outside-")); + outside_dirs.push(outside); await writeFile(path.join(outside, "secret.txt"), "leak\n", "utf8"); symlinkSync(outside, path.join(tmp_root, "escape_link")); const registry = new ToolRegistry(); @@ -420,6 +427,7 @@ describe("symlink confinement", () => { it("rejects writing through a symlink leaf that points outside", async () => { const outside = await mkdtemp(path.join(TMP_BASE, "outside-leaf-")); + outside_dirs.push(outside); const outside_file = path.join(outside, "target.txt"); await writeFile(outside_file, "before\n", "utf8"); symlinkSync(outside_file, path.join(tmp_root, "leaf_link.txt")); diff --git a/test/tui.test.ts b/test/tui.test.ts index 6dc0323..2c8fac4 100644 --- a/test/tui.test.ts +++ b/test/tui.test.ts @@ -418,6 +418,16 @@ describe("notice blocks", () => { expect(blocks[1]?.lines[0]).toBe(`${LICH_THEME.response_label} › partial`); }); + it("shows a cancelled notice for an aborted run", () => { + const blocks = run_notice_blocks({ + outcome: { messages: [], final: undefined, result: undefined, turns_used: 1, stopped_reason: "aborted" }, + messages: [], + usage_total: usage(0), + session_path: undefined, + }, LICH_THEME); + expect(blocks).toEqual([{ role: "error", lines: ["\u00b7 run cancelled"] }]); + }); + it("omits the final block when content is empty", () => { const blocks = run_notice_blocks({ outcome: { messages: [], final: { role: "assistant", content: "" }, result: undefined, turns_used: 1, stopped_reason: "final" }, diff --git a/test/tui_command_bar.test.ts b/test/tui_command_bar.test.ts new file mode 100644 index 0000000..6096c9f --- /dev/null +++ b/test/tui_command_bar.test.ts @@ -0,0 +1,69 @@ +/** + * CommandBar keys: Esc cancels a running turn (#133) and does nothing while idle. + */ +import { EventEmitter } from "node:events"; +import { createElement } from "react"; +import { render } from "ink"; +import { describe, expect, it, vi } from "vitest"; +import { CommandBar } from "../src/tui/command_bar.js"; + +/** Minimal TTY streams ink accepts; `press` feeds raw key bytes. */ +function fake_tty() { + const stdin = Object.assign(new EventEmitter(), { + isTTY: true, + setRawMode: () => undefined, + setEncoding: () => undefined, + ref: () => undefined, + unref: () => undefined, + pending: [] as string[], + read(): string | null { + return stdin.pending.shift() ?? null; + }, + }); + const stdout = Object.assign(new EventEmitter(), { columns: 80, rows: 24, isTTY: true, write: () => true }); + const press = async (bytes: string): Promise => { + stdin.pending.push(bytes); + stdin.emit("readable"); + await new Promise((resolve) => setTimeout(resolve, 50)); + }; + return { stdin, stdout, press }; +} + +async function render_bar(busy: boolean) { + const on_cancel = vi.fn(); + const on_submit = vi.fn(); + const tty = fake_tty(); + const instance = render(createElement(CommandBar, { busy, on_submit, on_cancel }), { + stdin: tty.stdin as unknown as NodeJS.ReadStream, + stdout: tty.stdout as unknown as NodeJS.WriteStream, + exitOnCtrlC: false, + patchConsole: false, + }); + await new Promise((resolve) => setTimeout(resolve, 20)); + return { on_cancel, on_submit, press: tty.press, unmount: () => instance.unmount() }; +} + +describe("CommandBar", () => { + it("calls on_cancel on Esc while busy", async () => { + const bar = await render_bar(true); + try { + await bar.press("\u001b"); + expect(bar.on_cancel).toHaveBeenCalledTimes(1); + } finally { + bar.unmount(); + } + }); + + it("ignores Esc while idle", async () => { + const bar = await render_bar(false); + try { + await bar.press("\u001b"); + await bar.press("hi"); + await bar.press("\r"); + expect(bar.on_cancel).not.toHaveBeenCalled(); + expect(bar.on_submit).toHaveBeenCalledWith("hi"); + } finally { + bar.unmount(); + } + }); +}); From 84f62a4ff0484ef29298c0aaeee8c77d56c7a931 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 05:42:04 +0000 Subject: [PATCH 26/43] fix: on cancel, drop the unanswered user line and do not reprint the old reply On abort the loop's `final` is the last assistant so far (often the previous turn's reply) and `messages` ends with the cancelled user line. Chat and the TUI now show only the cancel notice and keep history_after_abort(messages), which drops trailing user lines. The config temp file is created with the existing mode. Tests cover a second Ctrl+C (exit 130) and the chat turn after a cancel. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/agent/loop.ts | 8 ++++++ src/cli.ts | 10 ++++--- src/cli_config.ts | 3 +- src/tui/app.tsx | 5 ++-- src/tui/state.ts | 3 +- test/cli_sigint.test.ts | 63 ++++++++++++++++++++++++++++++++--------- test/tui.test.ts | 4 +-- 7 files changed, 72 insertions(+), 24 deletions(-) diff --git a/src/agent/loop.ts b/src/agent/loop.ts index 5be46c7..f823a1c 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -345,6 +345,14 @@ export function history_after_run_error(error: unknown): Message[] | undefined { return progressed ? kept : undefined; } +/** + * Messages to keep after an aborted run: completed turns stay, the trailing + * user line that never got an answer is dropped. + */ +export function history_after_abort(messages: readonly Message[]): Message[] { + return drop_trailing_users(messages); +} + function drop_trailing_users(messages: readonly Message[]): Message[] { const kept = [...messages]; while (kept.at(-1)?.role === "user") { diff --git a/src/cli.ts b/src/cli.ts index 3ea4cde..022803e 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -8,7 +8,7 @@ import { createInterface } from "node:readline"; import path from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; import { create_agent_with_plugins, type Agent, type AgentRunResult } from "./agent/agent.js"; -import { history_after_run_error } from "./agent/loop.js"; +import { history_after_abort, history_after_run_error } from "./agent/loop.js"; import { parse_agent_config, type AgentConfig } from "./agent/config.js"; import type { EnvelopedAgentEmitter } from "./agent/events.js"; import { LICH_VERSION } from "./index.js"; @@ -485,13 +485,15 @@ async function run_chat_turn( const stop_progress = attach_progress(agent.events); try { const result = await agent.run({ input, history, signal }); + if (result.outcome.stopped_reason === "aborted") { + // `final` on abort is the last assistant so far, often the previous turn's reply. + process.stderr.write("[lich] turn cancelled\n"); + return history_after_abort(result.messages); + } const final = result.outcome.final; if (final !== undefined && final.content.length > 0) { process.stdout.write(`${final.content}\n`); } - if (result.outcome.stopped_reason === "aborted") { - process.stderr.write("[lich] turn cancelled\n"); - } process.stdout.write(`[turns ${result.outcome.turns_used} | tokens ${result.usage_total.total_tokens}]\n`); return result.messages; } catch (error) { diff --git a/src/cli_config.ts b/src/cli_config.ts index ab9b0a1..df90922 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -161,7 +161,8 @@ function replace_file(file: string, text: string): void { const temp = `${file}.${process.pid}.tmp`; const mode = statSync(file, { throwIfNoEntry: false })?.mode; try { - writeFileSync(temp, text, "utf8"); + // Create it with the old mode so a 0600 config is never briefly wider. + writeFileSync(temp, text, { encoding: "utf8", mode: mode === undefined ? undefined : mode & 0o777 }); if (mode !== undefined) { chmodSync(temp, mode); } diff --git a/src/tui/app.tsx b/src/tui/app.tsx index 4d3040d..431da33 100644 --- a/src/tui/app.tsx +++ b/src/tui/app.tsx @@ -6,7 +6,7 @@ import { useCallback, useEffect, useRef, useState, type Dispatch, type SetStateAction } from "react"; import { Box, Text } from "ink"; import type { Agent, AgentRunResult } from "../agent/agent.js"; -import { history_after_run_error } from "../agent/loop.js"; +import { history_after_abort, history_after_run_error } from "../agent/loop.js"; import type { AgentEvent } from "../agent/events.js"; import type { AgentConfig } from "../agent/config.js"; import type { Message } from "../providers/types.js"; @@ -135,7 +135,8 @@ function use_agent_run( }, []); const finish_run = useCallback((result: AgentRunResult): void => { - history_ref.current = result.messages; + history_ref.current = + result.outcome.stopped_reason === "aborted" ? history_after_abort(result.messages) : result.messages; set_state((current) => apply_run_result(current, result)); add_blocks(run_notice_blocks(result, theme)); }, [add_blocks, set_state, theme]); diff --git a/src/tui/state.ts b/src/tui/state.ts index e8fb123..e76712f 100644 --- a/src/tui/state.ts +++ b/src/tui/state.ts @@ -236,7 +236,8 @@ export function run_notice_blocks(result: AgentRunResult, theme: ThemeSpec): His blocks.push({ role: "error", lines: [`\u00b7 ${fill_template(theme.notices.budget_exhausted, {})}`] }); } if (result.outcome.stopped_reason === "aborted") { - blocks.push({ role: "error", lines: ["\u00b7 run cancelled"] }); + // `final` on abort is the last assistant so far, already on screen. + return [...blocks, { role: "error", lines: ["\u00b7 run cancelled"] }]; } const final_block = result.outcome.final === undefined ? undefined : assistant_result_block(result.outcome.final, theme); if (final_block !== undefined) { diff --git a/test/cli_sigint.test.ts b/test/cli_sigint.test.ts index 2e6cc40..46957c4 100644 --- a/test/cli_sigint.test.ts +++ b/test/cli_sigint.test.ts @@ -10,8 +10,9 @@ const sigint_state = vi.hoisted(() => ({ lines: [] as string[], rl: undefined as (EventEmitter & { closed: boolean }) | undefined, signals: [] as AbortSignal[], - runs: [] as string[], + runs: [] as Array<{ input: string; history: readonly Message[] }>, finished: 0, + release: undefined as (() => void) | undefined, })); vi.mock("node:readline", async () => { @@ -44,21 +45,32 @@ vi.mock("../src/agent/agent.js", () => ({ config: { theme: "lich" }, events: { on: () => () => undefined }, close: () => undefined, + // "slow" waits for the abort, "stuck" also ignores it until released; anything else replies at once. run: async (options: { input: string; history?: readonly Message[]; signal?: AbortSignal }) => { const signal = options.signal; if (signal === undefined) { throw new Error("no signal"); } + const history = options.history ?? []; sigint_state.signals.push(signal); - sigint_state.runs.push(options.input); + sigint_state.runs.push({ input: options.input, history }); + const messages: Message[] = [...history, { role: "user", content: options.input }]; + const usage_total = { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }; + if (options.input !== "slow" && options.input !== "stuck") { + messages.push({ role: "assistant", content: `echo:${options.input}` }); + const final = messages[messages.length - 1]; + return { outcome: { final, turns_used: 1, stopped_reason: "final", messages }, messages, usage_total }; + } await new Promise((resolve) => signal.addEventListener("abort", resolve, { once: true })); + if (options.input === "stuck") { + await new Promise((resolve) => { + sigint_state.release = resolve; + }); + } sigint_state.finished += 1; - const messages: Message[] = [...(options.history ?? []), { role: "user", content: options.input }]; - return { - outcome: { final: undefined, turns_used: 1, stopped_reason: "aborted", messages }, - messages, - usage_total: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }, - }; + // Like the loop: on abort `final` is the last assistant so far, from an earlier turn. + const final = [...history].reverse().find((message) => message.role === "assistant"); + return { outcome: { final, turns_used: 0, stopped_reason: "aborted", messages }, messages, usage_total }; }, }), })); @@ -79,6 +91,7 @@ afterEach(() => { sigint_state.signals = []; sigint_state.runs = []; sigint_state.finished = 0; + sigint_state.release = undefined; }); async function until(check: () => boolean): Promise { @@ -109,20 +122,42 @@ describe("CLI Ctrl+C", () => { expect(process.listeners("SIGINT")).toEqual(baseline); }); + it("one-shot: a second SIGINT while the run is still cancelling exits 130", async () => { + vi.spyOn(process.stderr, "write").mockImplementation(() => true); + const exit = vi.spyOn(process, "exit").mockImplementation((() => undefined) as never); + const baseline = process.listeners("SIGINT"); + const pending = run_one_shot(CONFIG, "stuck"); + await until(() => sigint_state.signals.length === 1); + send_sigint(baseline); + await until(() => sigint_state.release !== undefined); + expect(exit).not.toHaveBeenCalled(); + send_sigint(baseline); + expect(exit).toHaveBeenCalledWith(130); + sigint_state.release?.(); + expect(await pending).toBe(1); + }); + it("chat: Ctrl+C cancels the running turn, then ends chat at the prompt", async () => { vi.spyOn(process.stderr, "write").mockImplementation(() => true); - vi.spyOn(process.stdout, "write").mockImplementation(() => true); + const stdout: string[] = []; + vi.spyOn(process.stdout, "write").mockImplementation((chunk: string | Uint8Array) => { + stdout.push(String(chunk)); + return true; + }); const baseline = process.listeners("SIGINT"); - sigint_state.lines = ["slow"]; + sigint_state.lines = ["first", "slow", "after"]; const pending = run_chat(CONFIG); - await until(() => sigint_state.signals.length === 1); + await until(() => sigint_state.signals.length === 2); sigint_state.rl?.emit("SIGINT"); - await until(() => sigint_state.finished === 1); - expect(sigint_state.signals[0]?.aborted).toBe(true); + await until(() => sigint_state.runs.length === 3); + expect(sigint_state.signals[1]?.aborted).toBe(true); expect(sigint_state.rl?.closed).toBe(false); + // The cancelled line is dropped and the previous reply is not printed again. + expect(sigint_state.runs[2]?.history.map((message) => message.content)).toEqual(["first", "echo:first"]); + expect(stdout.filter((chunk) => chunk === "echo:first\n")).toHaveLength(1); send_sigint(baseline); expect(await pending).toBe(0); - expect(sigint_state.runs).toEqual(["slow"]); + expect(sigint_state.runs.map((run) => run.input)).toEqual(["first", "slow", "after"]); expect(process.listeners("SIGINT")).toEqual(baseline); }); }); diff --git a/test/tui.test.ts b/test/tui.test.ts index 2c8fac4..68d4c87 100644 --- a/test/tui.test.ts +++ b/test/tui.test.ts @@ -418,9 +418,9 @@ describe("notice blocks", () => { expect(blocks[1]?.lines[0]).toBe(`${LICH_THEME.response_label} › partial`); }); - it("shows a cancelled notice for an aborted run", () => { + it("shows only a cancelled notice for an aborted run, not the earlier reply", () => { const blocks = run_notice_blocks({ - outcome: { messages: [], final: undefined, result: undefined, turns_used: 1, stopped_reason: "aborted" }, + outcome: { messages: [], final: { role: "assistant", content: "earlier" }, result: undefined, turns_used: 0, stopped_reason: "aborted" }, messages: [], usage_total: usage(0), session_path: undefined, From 3cfddb1b8cc4f2bb29a4bd4947e4baf1e8e30a2c Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 05:49:58 +0000 Subject: [PATCH 27/43] fix: keep this turn's reply when a cancel lands during tools reply_after_abort(outcome) returns `final` only when it comes after the last user line, i.e. this run wrote it. Chat prints it and the TUI shows it before the cancel notice; an earlier turn's reply is still skipped. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/agent/loop.ts | 21 +++++++++++++++++++++ src/cli.ts | 8 ++++++-- src/tui/state.ts | 7 +++++-- test/cli_sigint.test.ts | 31 +++++++++++++++++++++++++++---- test/tui.test.ts | 11 +++++++++++ 5 files changed, 70 insertions(+), 8 deletions(-) diff --git a/src/agent/loop.ts b/src/agent/loop.ts index f823a1c..e61214b 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -353,6 +353,27 @@ export function history_after_abort(messages: readonly Message[]): Message[] { return drop_trailing_users(messages); } +/** + * `final` of an aborted run when this run produced it (it comes after the last + * user line); undefined when it is an earlier turn's reply. + */ +export function reply_after_abort(outcome: LoopOutcome): AssistantMessage | undefined { + const final = outcome.final; + if (final === undefined) { + return undefined; + } + for (let index = outcome.messages.length - 1; index >= 0; index -= 1) { + const message = outcome.messages[index]; + if (message === final) { + return final; + } + if (message?.role === "user") { + return undefined; + } + } + return undefined; +} + function drop_trailing_users(messages: readonly Message[]): Message[] { const kept = [...messages]; while (kept.at(-1)?.role === "user") { diff --git a/src/cli.ts b/src/cli.ts index 022803e..cb21642 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -8,7 +8,7 @@ import { createInterface } from "node:readline"; import path from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; import { create_agent_with_plugins, type Agent, type AgentRunResult } from "./agent/agent.js"; -import { history_after_abort, history_after_run_error } from "./agent/loop.js"; +import { history_after_abort, history_after_run_error, reply_after_abort } from "./agent/loop.js"; import { parse_agent_config, type AgentConfig } from "./agent/config.js"; import type { EnvelopedAgentEmitter } from "./agent/events.js"; import { LICH_VERSION } from "./index.js"; @@ -486,7 +486,11 @@ async function run_chat_turn( try { const result = await agent.run({ input, history, signal }); if (result.outcome.stopped_reason === "aborted") { - // `final` on abort is the last assistant so far, often the previous turn's reply. + // `final` on abort is the last assistant so far; print it only if this turn wrote it. + const reply = reply_after_abort(result.outcome); + if (reply !== undefined && reply.content.length > 0) { + process.stdout.write(`${reply.content}\n`); + } process.stderr.write("[lich] turn cancelled\n"); return history_after_abort(result.messages); } diff --git a/src/tui/state.ts b/src/tui/state.ts index e76712f..e95c999 100644 --- a/src/tui/state.ts +++ b/src/tui/state.ts @@ -5,6 +5,7 @@ */ import type { AgentEventBody } from "../agent/events.js"; import type { AgentRunResult } from "../agent/agent.js"; +import { reply_after_abort } from "../agent/loop.js"; import type { AssistantMessage, Message, ToolCall, Usage } from "../providers/types.js"; import { safe_json_parse, truncate_text } from "../util/json.js"; import type { ThemeSpec } from "../util/lore.js"; @@ -236,8 +237,10 @@ export function run_notice_blocks(result: AgentRunResult, theme: ThemeSpec): His blocks.push({ role: "error", lines: [`\u00b7 ${fill_template(theme.notices.budget_exhausted, {})}`] }); } if (result.outcome.stopped_reason === "aborted") { - // `final` on abort is the last assistant so far, already on screen. - return [...blocks, { role: "error", lines: ["\u00b7 run cancelled"] }]; + // `final` on abort is the last assistant so far; show it only if this run wrote it. + const reply = reply_after_abort(result.outcome); + const reply_block = reply === undefined ? undefined : assistant_result_block(reply, theme); + return [...blocks, ...(reply_block === undefined ? [] : [reply_block]), { role: "error", lines: ["\u00b7 run cancelled"] }]; } const final_block = result.outcome.final === undefined ? undefined : assistant_result_block(result.outcome.final, theme); if (final_block !== undefined) { diff --git a/test/cli_sigint.test.ts b/test/cli_sigint.test.ts index 46957c4..89dfb50 100644 --- a/test/cli_sigint.test.ts +++ b/test/cli_sigint.test.ts @@ -45,7 +45,8 @@ vi.mock("../src/agent/agent.js", () => ({ config: { theme: "lich" }, events: { on: () => () => undefined }, close: () => undefined, - // "slow" waits for the abort, "stuck" also ignores it until released; anything else replies at once. + // "slow" waits for the abort, "stuck" also ignores it until released, "midtool" has + // already replied this turn when the abort lands; anything else replies at once. run: async (options: { input: string; history?: readonly Message[]; signal?: AbortSignal }) => { const signal = options.signal; if (signal === undefined) { @@ -56,7 +57,7 @@ vi.mock("../src/agent/agent.js", () => ({ sigint_state.runs.push({ input: options.input, history }); const messages: Message[] = [...history, { role: "user", content: options.input }]; const usage_total = { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }; - if (options.input !== "slow" && options.input !== "stuck") { + if (options.input !== "slow" && options.input !== "stuck" && options.input !== "midtool") { messages.push({ role: "assistant", content: `echo:${options.input}` }); const final = messages[messages.length - 1]; return { outcome: { final, turns_used: 1, stopped_reason: "final", messages }, messages, usage_total }; @@ -68,8 +69,11 @@ vi.mock("../src/agent/agent.js", () => ({ }); } sigint_state.finished += 1; - // Like the loop: on abort `final` is the last assistant so far, from an earlier turn. - const final = [...history].reverse().find((message) => message.role === "assistant"); + if (options.input === "midtool") { + messages.push({ role: "assistant", content: "partial reply" }); + } + // Like the loop: on abort `final` is the last assistant so far. + const final = [...messages].reverse().find((message) => message.role === "assistant"); return { outcome: { final, turns_used: 0, stopped_reason: "aborted", messages }, messages, usage_total }; }, }), @@ -137,6 +141,25 @@ describe("CLI Ctrl+C", () => { expect(await pending).toBe(1); }); + it("chat: a reply this turn wrote before the cancel is printed and kept", async () => { + vi.spyOn(process.stderr, "write").mockImplementation(() => true); + const stdout: string[] = []; + vi.spyOn(process.stdout, "write").mockImplementation((chunk: string | Uint8Array) => { + stdout.push(String(chunk)); + return true; + }); + const baseline = process.listeners("SIGINT"); + sigint_state.lines = ["midtool", "after"]; + const pending = run_chat(CONFIG); + await until(() => sigint_state.signals.length === 1); + sigint_state.rl?.emit("SIGINT"); + await until(() => sigint_state.runs.length === 2); + expect(stdout).toContain("partial reply\n"); + expect(sigint_state.runs[1]?.history.map((message) => message.content)).toEqual(["midtool", "partial reply"]); + send_sigint(baseline); + expect(await pending).toBe(0); + }); + it("chat: Ctrl+C cancels the running turn, then ends chat at the prompt", async () => { vi.spyOn(process.stderr, "write").mockImplementation(() => true); const stdout: string[] = []; diff --git a/test/tui.test.ts b/test/tui.test.ts index 68d4c87..f7f182e 100644 --- a/test/tui.test.ts +++ b/test/tui.test.ts @@ -428,6 +428,17 @@ describe("notice blocks", () => { expect(blocks).toEqual([{ role: "error", lines: ["\u00b7 run cancelled"] }]); }); + it("shows a reply the aborted run wrote before the cancel notice", () => { + const reply = { role: "assistant" as const, content: "partial" }; + const blocks = run_notice_blocks({ + outcome: { messages: [{ role: "user", content: "go" }, reply], final: reply, result: undefined, turns_used: 1, stopped_reason: "aborted" }, + messages: [], + usage_total: usage(0), + session_path: undefined, + }, LICH_THEME); + expect(blocks.map((block) => block.lines[0])).toEqual([`${LICH_THEME.response_label} › partial`, "\u00b7 run cancelled"]); + }); + it("omits the final block when content is empty", () => { const blocks = run_notice_blocks({ outcome: { messages: [], final: { role: "assistant", content: "" }, result: undefined, turns_used: 1, stopped_reason: "final" }, From b94119159ee41f9ed06df590027ed701d1236ccb Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 15:34:48 +0000 Subject: [PATCH 28/43] wiki: map #113 phase children and related issues Roadmap map lists #133 (P0), #134 (P1), #117 (P2), #144, #148 and #149 with their wiki pages; the kanban skill names the phase children. Supersedes the stale #135. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- .claude/skills/lich-kanban/SKILL.md | 2 +- wiki/entities/roadmap-issues.md | 10 ++++++++-- wiki/log.md | 1 + 3 files changed, 10 insertions(+), 3 deletions(-) diff --git a/.claude/skills/lich-kanban/SKILL.md b/.claude/skills/lich-kanban/SKILL.md index 57570fc..345e26e 100644 --- a/.claude/skills/lich-kanban/SKILL.md +++ b/.claude/skills/lich-kanban/SKILL.md @@ -98,7 +98,7 @@ git worktree add ".worktrees/issue--" -b "issue--" ## Key issues -- **#113**: plan of record — idea-agnostic harness (Addendum 5); games/content as dogfood verticals; gateway/serve/wiki/DRY addenda. +- **#113**: plan of record — idea-agnostic harness (Addendum 5); games/content as dogfood verticals; gateway/serve/wiki/DRY addenda. Phase children: #133 (P0), #134 (P1), #117 (P2). - **#114**: deferred work, each item with a revisit trigger. - **#79**: Ossuary epic (serve #81–#84, desktop #86–#94). - **#46 / #47**: closed v0.7.0 REVIEW track; leftover test depth is #114 item 10. diff --git a/wiki/entities/roadmap-issues.md b/wiki/entities/roadmap-issues.md index 68f561b..eebbd1f 100644 --- a/wiki/entities/roadmap-issues.md +++ b/wiki/entities/roadmap-issues.md @@ -1,10 +1,10 @@ --- title: Roadmap issues map created: 2026-09-23 -updated: 2026-09-25 +updated: 2026-10-04 type: entity tags: [roadmap, process] -sources: [raw/issues/issue-113.md, raw/issues/issue-114.md, raw/issues/issue-79.md, "#113"] +sources: [raw/issues/issue-113.md, raw/issues/issue-114.md, raw/issues/issue-79.md, "#113", "#133", "#134", "#117", "#144", "#148", "#149"] confidence: high --- @@ -15,6 +15,12 @@ This page is a **map of what each key issue is for**. It does not track status: | Issue | Role | Wiki pages | |---|---|---| | **#113** | Plan of record: architecture and gap audit. **North star (Addendum 5):** idea-agnostic harness that wins on user control (plugins, tools, profiles). Games (A/B/C) and content (D / Addendum 4) are dogfood verticals. Hermes parity, reorganization, doc drafts. Addenda: 1 gateway-as-hub, 2 serve path + DRY, 3 wiki, 4 content vertical, 5 idea-agnostic / extensibility. | [[0001-gateway-as-hub]], [[0002-serve-pr-merge-path]], [[0003-subpath-exports-over-packages]], [[0004-dry-policy]], [[0006-in-repo-llm-wiki]], [[0007-content-as-fourth-goal]], [[0008-idea-agnostic-extensible-harness]] | +| **#133** | #113 phase P0: hygiene and correctness checklist (#113 §5). | [[lich-tools-and-guardrails]], [[lich-plugins-and-hooks]] | +| **#134** | #113 phase P1: event envelope and SessionManager. | [[event-envelope]] | +| **#117** | Global agent identity and Hermes-style profiles under `~/.lich/**`. #133 names it as phase P2. | [[runtime-profile-session]] | +| **#144** | Gateway hub: text adapters on SessionManager, serve as the interactive adapter (from #114 items 2–3). | [[0001-gateway-as-hub]], [[lich-serve]] | +| **#149** | Model roles in config (`models.chat` / `models.compress`), plugin settings, host model access, `before_llm_call`. | [[lich-plugins-and-hooks]] | +| **#148** | Ollama defaults and cloud docs, plus the decision-model plugin prototype. | [[decision-models]], [[decision-lane-example]] | | **#114** | Deferred work with revisit triggers: package split, adapters, file-bus, SDKs, TLS, memory, platform features, play harness, REVIEW test depth (10), media toolchains as optional vertical tooling (11) | [[subpaths-vs-packages]], [[npc-memory-namespaces]], [[0008-idea-agnostic-extensible-harness]] | | **#79** | Ossuary and serve epic (#80–#95). Closed after the wave landed in #123; remaining envelope/SessionManager follow-ups live under #114. | [[ossuary]], [[lich-serve]] | | **#46 / #47** | v0.7.0 REVIEW epic and Tests & CI tracker — **closed**. Leftover T-3/T-4 depth is #114 item 10. | [[lich-tools-and-guardrails]] | diff --git a/wiki/log.md b/wiki/log.md index b2edb3c..879b2b8 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -39,3 +39,4 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-04 update | Decision models and game_bridge pages link the decision_lane prototype | concepts/decision-models.md, entities/game-bridge-example.md - 2026-10-04 update | #161 merged: tools_enabled filters plugin tools and git_commit, throwing before_tool_call blocks, last-turn tool calls not run; holes lists and index updated, code pinned at 029b7e8 | entities/lich-plugins-and-hooks.md, entities/lich-tools-and-guardrails.md, entities/lich-agent-loop.md, index.md - 2026-10-04 update | #163 merged: S-11 items (http_request status, .. guard, terminal_timeout_ms) moved out of open holes, pinned at 7001e31 | entities/lich-tools-and-guardrails.md +- 2026-10-04 update | Roadmap map lists #113 children and related issues (#133 P0, #134 P1, #117 P2, #144, #148, #149); supersedes stale PR #135 | entities/roadmap-issues.md From 4c8b180553f8f4ed98d7cbe13c9e123005df54fa Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 22:24:17 +0000 Subject: [PATCH 29/43] feat(config): global ~/.lich/config.json merged under the project config (#117) Phase 0 + 1 of #117. Discovery is now project .lich/config.json merged over ~/.lich/config.json (shallow, project wins per key). A project providers array replaces the global one and drops the global models unless the project sets its own. The global layer ignores work_dir and session_dir, and its relative plugin paths resolve against ~/.lich/. ~/.config/lich/config.json is read only when ~/.lich/config.json is absent, with a hint to move it; nothing writes there. --config still replaces the whole chain. Bare lich no longer pins an existing config as --config when it skips the wizard, so the merge applies there too. first_run tests use an empty HOME instead of mocking the home lookup. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 8 +++ docs/getting-started.md | 4 +- docs/user-guide/cli.md | 17 +++++- src/cli.ts | 14 ++--- src/cli_config.ts | 101 +++++++++++++++++++++++++++++-- test/cli_config.test.ts | 128 ++++++++++++++++++++++++++++++++++++++-- test/first_run.test.ts | 51 +++++++++++----- 7 files changed, 287 insertions(+), 36 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a61a618..c6d3e35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,14 @@ ## Unreleased +- Global defaults live in `~/.lich/config.json`. A project `.lich/config.json` + is merged over it key by key instead of hiding it (#117). A project + `providers` array replaces the global one and drops the global `models` + unless the project sets its own. `work_dir` and `session_dir` in the global + file are ignored, and its relative plugin paths resolve against `~/.lich/`. + The old `~/.config/lich/config.json` is read only when `~/.lich/config.json` + is absent, with a hint to move it. **Behavior change:** before, a project + file hid the user file entirely; now its unset keys come from the global file. - Ctrl+C in a one-shot run or in `lich chat` cancels the run or turn through its abort signal; chat keeps the session and returns to the prompt. A second Ctrl+C quits at once (exit `130`). In the TUI, Esc cancels a running diff --git a/docs/getting-started.md b/docs/getting-started.md index e7c5617..63be57e 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -40,7 +40,7 @@ lich config > .lich/config.json # edit .lich/config.json and replace "" ``` -`lich config` honors `LICH_PROVIDER_KIND` and `LICH_MODEL` when you have them set, and otherwise prints an ollama-oriented template. The file is picked up automatically from `.lich/config.json` in the working directory (or `~/.config/lich/config.json` as a fallback) — after this, plain `lich "task"` needs no env vars. +`lich config` honors `LICH_PROVIDER_KIND` and `LICH_MODEL` when you have them set, and otherwise prints an ollama-oriented template. The file is picked up automatically from `.lich/config.json` in the working directory, merged over `~/.lich/config.json` if you keep defaults there — after this, plain `lich "task"` needs no env vars. `lich init` writes that same starter file for you (it creates `.lich/` and never overwrites an existing `.lich/config.json`). Bare `lich` on a TTY, with no config in that search chain, runs a setup wizard and writes `.lich/config.json` once before opening the TUI. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. Non-TTY stdin skips the wizard. `.lich/` is gitignored. @@ -50,7 +50,7 @@ lich config > .lich/config.json lich --config ./lich.json "Reply with ok" ``` -The full schema is documented in [the CLI reference](user-guide/cli.md#config-file-reference). Search order: `--config` path first (must exist), then `./.lich/config.json`, then `~/.config/lich/config.json`. +The full schema is documented in [the CLI reference](user-guide/cli.md#config-file-reference). Search order: a `--config` path (must exist) is used alone; otherwise `./.lich/config.json` is merged over `~/.lich/config.json` (see [Global config](user-guide/cli.md#global-config)). ## Your first one-shot diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index 8b6e476..017c589 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -24,7 +24,7 @@ lich --help # usage text lich --version # package.json version (published package and this tree: 0.8.0) ``` -- **Bare `lich`** opens the same TUI as `lich tui`. It does not print usage. On a TTY, if neither `.lich/config.json` nor `~/.config/lich/config.json` exists, a setup wizard runs first (name, provider, optional gateway env-var names, optional plugins) and writes `.lich/config.json` once. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. An existing config in that chain skips the wizard and is not replaced. Non-TTY stdin skips the wizard and prints guidance instead of hanging. `lich --help` still prints usage. +- **Bare `lich`** opens the same TUI as `lich tui`. It does not print usage. On a TTY, if none of `.lich/config.json`, `~/.lich/config.json` or the legacy `~/.config/lich/config.json` exists, a setup wizard runs first (name, provider, optional gateway env-var names, optional plugins) and writes `.lich/config.json` once. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. An existing config in that chain skips the wizard and is not replaced. Non-TTY stdin skips the wizard and prints guidance instead of hanging. `lich --help` still prints usage. - **`lich init`** writes that starter file without prompts, using the same writer as the wizard. Existing flags such as `--model` are written into the file and win over `LICH_MODEL`. It never overwrites an existing `.lich/config.json`. `.lich/` is gitignored. - **One-shot** joins all positional words into a single task, runs the agent loop, prints the final answer to stdout, and exits. Progress (turn numbers, tool results) goes to stderr. Ctrl+C cancels the run (exit `1`); a second Ctrl+C quits at once (exit `130`). - **Chat** is a readline REPL over one long-lived agent: each line is a turn, memory persists across lines, and an empty line, `/exit`, or `/quit` ends the session. After each turn it prints a `[turns N | tokens M]` footer. Ctrl+C cancels the running turn and keeps the session; at the prompt it ends chat. A second Ctrl+C while a turn is still cancelling quits at once (exit `130`). @@ -69,7 +69,7 @@ Passing `--max-turns 0` or a non-integer fails with `--max-turns must be a posit The effective provider for a run is decided in this order: 1. If `--config ` was passed, that file is the whole configuration (it must exist and be a JSON object, or the CLI fails with `cannot use config file : ...`). -2. Otherwise the discovery chain is walked: `./.lich/config.json`, then `~/.config/lich/config.json`. The first file found becomes the config. `LICH_*` env vars are *not* merged into a discovered file. +2. Otherwise the project file `./.lich/config.json` (under `--work-dir` when given) is merged over the global file `~/.lich/config.json`. Either one alone is enough. `LICH_*` env vars are *not* merged into a discovered file. See [Global config](#global-config). 3. If no config file exists, one is built from the environment: `--provider-kind` / `LICH_PROVIDER_KIND` (default `openai_compat`), `--model` / `LICH_MODEL` (required — without it the CLI fails with `no model configured`), `--base-url` / `LICH_BASE_URL`, and `--api-key-env` / `LICH_API_KEY_ENV`, each falling back to the per-kind defaults below. 4. Provider override flags (`--model`, `--provider-kind`, `--base-url`, `--api-key-env`) always win over the chosen source: with a config file present they patch `providers[0]` in place; without one they seed a fresh provider from the environment. @@ -81,6 +81,19 @@ Per-kind defaults: | `anthropic` | `https://api.anthropic.com` | `ANTHROPIC_API_KEY` | | | `ollama` | `http://localhost:11434` | none | No key needed locally; `api_key`/`api_key_env` are sent as a Bearer header when set, which Ollama cloud (`https://ollama.com`, `OLLAMA_API_KEY`) requires. | +## Global config + +`~/.lich/config.json` holds defaults for every project: provider, model, `agent_name`, `max_turns`, gateway settings, plugins and so on. A project `.lich/config.json` overrides it key by key: + +- **Shallow merge, project wins.** A top-level key in the project file replaces the global value whole. `gateway` and `mcp_servers` are not merged entry by entry. +- **`providers`:** a project `providers` array replaces the global one. The global `models` role chains are then dropped (they name the global providers) unless the project sets its own `models`. +- **Per-project keys:** `work_dir` and `session_dir` in the global file are ignored. +- **Paths:** relative `plugins` paths in the global file resolve against `~/.lich/`. MCP `command` and `args` are used as written. +- **Legacy location:** `~/.config/lich/config.json` is still read when `~/.lich/config.json` is absent, with a one-line hint to move it. Lich never writes there. +- **`--config `** replaces the whole chain; nothing is merged. + +`lich init`, the setup wizard and `lich mcp` write only the project file. + ## Config file reference Validated by zod (top-level unknown keys are silently stripped; extra keys inside a `providers[]` entry are passed through; each `mcp_servers` entry is strict). Example: diff --git a/src/cli.ts b/src/cli.ts index cb21642..1bca43c 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -13,7 +13,7 @@ import { parse_agent_config, type AgentConfig } from "./agent/config.js"; import type { EnvelopedAgentEmitter } from "./agent/events.js"; import { LICH_VERSION } from "./index.js"; import { - load_config, + load_layered_config, config_template, existing_config_path, provider_kind_defaults, @@ -384,13 +384,13 @@ function apply_overrides(config: Record, overrides: Record | undefined { - const found = existing_config_path(work_dir); - if (found === undefined) { - return undefined; + const layered = load_layered_config(work_dir); + for (const note of layered?.notes ?? []) { + process.stderr.write(`${note}\n`); } - return load_config(found); + return layered?.config; } function build_config(options: CliOptions): AgentConfig { @@ -603,8 +603,8 @@ async function maybe_first_run(options: CliOptions, work_dir: string): Promise chain.indexOf(entry) === index); } /** Explicit path must exist (else throw); otherwise walk the chain, undefined when absent. */ @@ -61,6 +71,87 @@ export function load_config(explicit?: string): Record | undefi if (found === undefined) { return undefined; } + return read_config_object(found); +} + +export interface LayeredConfig { + config: Record; + /** Files merged, base first. */ + sources: string[]; + /** One-line notices for the user (for example, the legacy location in use). */ + notes: string[]; +} + +/** + * Project `.lich/config.json` merged over the global base (`~/.lich/config.json`, + * else the legacy `~/.config/lich/config.json`). Undefined when neither exists. + */ +export function load_layered_config(work_dir: string): LayeredConfig | undefined { + const project_path = project_config_path(work_dir); + const notes: string[] = []; + let base_path: string | undefined; + for (const candidate of config_search_paths(work_dir).slice(1)) { + if (candidate !== project_path && existsSync(candidate) === true) { + base_path = candidate; + break; + } + } + const legacy = path.resolve(homedir(), LEGACY_USER_CONFIG); + if (base_path === legacy) { + notes.push(`lich: reading ${legacy}; move it to ${global_config_path()} (the old location is read-only)`); + } + const project = existsSync(project_path) === true ? read_config_object(project_path) : undefined; + const base = base_path === undefined ? undefined : global_layer(read_config_object(base_path), path.dirname(base_path)); + if (project === undefined && base === undefined) { + return undefined; + } + const sources = [base_path, project === undefined ? undefined : project_path].filter( + (entry): entry is string => entry !== undefined, + ); + return { config: merge_config_layers(base ?? {}, project ?? {}), sources, notes }; +} + +/** + * Shallow merge, project wins per key. A project `providers` array replaces + * the base's; the base `models` is then dropped unless the project sets its + * own, since its role chains name the base's providers. + */ +export function merge_config_layers( + base: Record, + project: Record, +): Record { + const merged = { ...base, ...project }; + if (Object.hasOwn(project, "providers") === true && Object.hasOwn(project, "models") === false) { + delete merged["models"]; + } + return merged; +} + +/** + * A global file as a base layer: per-project keys are ignored, and relative + * plugin paths resolve against the file's own directory instead of the project. + */ +function global_layer(config: Record, home: string): Record { + const layer = { ...config }; + for (const key of PROJECT_ONLY_KEYS) { + delete layer[key]; + } + const plugins = layer["plugins"]; + if (Array.isArray(plugins) === true) { + layer["plugins"] = plugins.map((entry: unknown) => { + if (typeof entry === "string") { + return path.resolve(home, entry); + } + if (typeof entry === "object" && entry !== null && typeof (entry as { path?: unknown }).path === "string") { + return { ...entry, path: path.resolve(home, (entry as { path: string }).path) }; + } + return entry; + }); + } + return layer; +} + +function read_config_object(found: string): Record { const raw = readFileSync(found, "utf8"); const parsed = safe_json_parse(raw); if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) { diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 71d6353..18d4bd6 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -5,13 +5,16 @@ import { chmodSync, existsSync, linkSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "node:fs"; import { homedir, tmpdir } from "node:os"; import path from "node:path"; -import { afterEach, describe, expect, it } from "vitest"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { config_search_paths, config_template, ensure_lich_config_dir, find_config_file, + global_config_path, load_config, + load_layered_config, + merge_config_layers, project_config_path, write_lich_config, } from "../src/cli_config.js"; @@ -71,10 +74,20 @@ afterEach(() => { }); describe("config_search_paths", () => { - it("lists the cwd .lich config first and the user config home second", () => { + it("lists the cwd .lich config, then ~/.lich/config.json, then the legacy ~/.config/lich file", () => { const paths = config_search_paths(); - expect(paths[0]).toBe(path.resolve(process.cwd(), ".lich/config.json")); - expect(paths[1]).toBe(path.resolve(homedir(), ".config/lich/config.json")); + expect(paths).toEqual([ + path.resolve(process.cwd(), ".lich/config.json"), + path.resolve(homedir(), ".lich/config.json"), + path.resolve(homedir(), ".config/lich/config.json"), + ]); + }); + + it("lists the global file once when the work_dir is the home directory", () => { + expect(config_search_paths(homedir())).toEqual([ + path.resolve(homedir(), ".lich/config.json"), + path.resolve(homedir(), ".config/lich/config.json"), + ]); }); it("resolves the first entry from the requested work_dir", () => { @@ -85,7 +98,7 @@ describe("config_search_paths", () => { it("keeps both entries ordered for a work_dir inside the config home", () => { const paths = config_search_paths(path.join(homedir(), ".config/lich")); expect(paths[0]).toBe(path.resolve(homedir(), ".config/lich/.lich/config.json")); - expect(paths[1]).toBe(path.resolve(homedir(), ".config/lich/config.json")); + expect(paths[2]).toBe(path.resolve(homedir(), ".config/lich/config.json")); }); }); @@ -185,6 +198,111 @@ describe("ensure_lich_config_dir", () => { }); }); +describe("load_layered_config", () => { + let saved_home: string | undefined; + let home: string; + + beforeEach(() => { + saved_home = process.env.HOME; + home = make_temp_dir("home"); + process.env.HOME = home; + }); + + afterEach(() => { + if (saved_home === undefined) { + delete process.env.HOME; + } else { + process.env.HOME = saved_home; + } + }); + + function write_json(file: string, value: unknown): void { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value)); + } + + it("returns undefined when neither a project nor a global config exists", () => { + expect(load_layered_config(make_temp_dir("none"))).toBeUndefined(); + }); + + it("merges the project over ~/.lich/config.json per key", () => { + const work = make_temp_dir("work"); + write_json(global_config_path(), { agent_name: "wight", max_turns: 9, theme: "lich" }); + write_json(project_config_path(work), { max_turns: 3 }); + const layered = load_layered_config(work); + expect(layered?.config).toEqual({ agent_name: "wight", max_turns: 3, theme: "lich" }); + expect(layered?.sources).toEqual([global_config_path(), project_config_path(work)]); + expect(layered?.notes).toEqual([]); + }); + + it("ignores work_dir and session_dir from the global layer and resolves its plugin paths against ~/.lich", () => { + const work = make_temp_dir("work"); + write_json(global_config_path(), { + work_dir: "/elsewhere", + session_dir: "/elsewhere/sessions", + plugins: ["./plugins/a.mjs", { path: "b.mjs", settings: { x: 1 } }, "/abs/c.mjs"], + }); + const config = load_layered_config(work)?.config; + expect(config?.["work_dir"]).toBeUndefined(); + expect(config?.["session_dir"]).toBeUndefined(); + expect(config?.["plugins"]).toEqual([ + path.join(home, ".lich", "plugins", "a.mjs"), + { path: path.join(home, ".lich", "b.mjs"), settings: { x: 1 } }, + "/abs/c.mjs", + ]); + }); + + it("keeps project plugin paths and work_dir as written", () => { + const work = make_temp_dir("work"); + write_json(global_config_path(), { plugins: ["./g.mjs"] }); + write_json(project_config_path(work), { work_dir: work, plugins: ["./p.mjs"] }); + expect(load_layered_config(work)?.config).toEqual({ work_dir: work, plugins: ["./p.mjs"] }); + }); + + it("reads the legacy ~/.config/lich file only when ~/.lich/config.json is absent, with a note", () => { + const work = make_temp_dir("work"); + const legacy = path.join(home, ".config", "lich", "config.json"); + write_json(legacy, { agent_name: "legacy" }); + const old = load_layered_config(work); + expect(old?.config).toEqual({ agent_name: "legacy" }); + expect(old?.notes[0]).toContain(global_config_path()); + write_json(global_config_path(), { agent_name: "global" }); + const current = load_layered_config(work); + expect(current?.config).toEqual({ agent_name: "global" }); + expect(current?.notes).toEqual([]); + }); + + it("does not merge ~/.lich/config.json over itself when the work_dir is the home directory", () => { + write_json(global_config_path(), { agent_name: "home" }); + const layered = load_layered_config(home); + expect(layered?.config).toEqual({ agent_name: "home" }); + expect(layered?.sources).toEqual([global_config_path()]); + }); +}); + +describe("merge_config_layers", () => { + const base_providers = [{ kind: "ollama", name: "a", model: "m" }]; + + it("lets a project providers array replace the base's and drops the base models", () => { + const merged = merge_config_layers( + { providers: base_providers, models: { chat: ["a"] } }, + { providers: [{ kind: "ollama", name: "b", model: "n" }] }, + ); + expect(merged).toEqual({ providers: [{ kind: "ollama", name: "b", model: "n" }] }); + }); + + it("keeps the base models when the project does not set providers, and project models when it does", () => { + expect(merge_config_layers({ providers: base_providers, models: { chat: ["a"] } }, { max_turns: 2 })).toEqual({ + providers: base_providers, + models: { chat: ["a"] }, + max_turns: 2, + }); + expect( + merge_config_layers({ models: { chat: ["a"] } }, { providers: base_providers, models: { compress: ["a"] } }), + ).toEqual({ providers: base_providers, models: { compress: ["a"] } }); + }); +}); + describe("write_lich_config update", () => { it("replaces the file by rename instead of rewriting it in place, keeping its mode", () => { const dir = make_temp_dir("atomic"); diff --git a/test/first_run.test.ts b/test/first_run.test.ts index 8e41b1d..5cb8987 100644 --- a/test/first_run.test.ts +++ b/test/first_run.test.ts @@ -9,13 +9,11 @@ import { TMP_BASE } from "./helpers/tmp_base.js"; const writes = vi.hoisted(() => ({ calls: [] as Array<{ work_dir: string; config: Record }> })); const wizard = vi.hoisted(() => ({ cancel: false, lines: [] as string[] })); -const tui_run = vi.hoisted(() => ({ configs: [] as Array<{ providers?: Array<{ model?: string }> }> })); +const tui_run = vi.hoisted(() => ({ configs: [] as Array<{ agent_name?: string; max_turns?: number; providers?: Array<{ model?: string }> }> })); const gateway_run = vi.hoisted(() => ({ fn: vi.fn(async (_config: unknown, _platforms: readonly string[]) => 0), })); vi.mock("../src/cli_config.js", async (import_original) => { - const path_mod = await import("node:path"); - const fs_mod = await import("node:fs"); const actual = await import_original(); return { ...actual, @@ -23,15 +21,6 @@ vi.mock("../src/cli_config.js", async (import_original) => { writes.calls.push({ work_dir, config }); return actual.write_lich_config(work_dir, config); }, - existing_config_path: (work_dir: string) => { - const [project] = actual.config_search_paths(work_dir); - const found = actual.existing_config_path(work_dir); - if (found !== undefined && found === project) { - return found; - } - const absent_home = path_mod.resolve(work_dir, "missing-home-config.json"); - return fs_mod.existsSync(absent_home) === true ? absent_home : undefined; - }, }; }); @@ -126,6 +115,9 @@ beforeEach(() => { saved_env[key] = process.env[key]; delete process.env[key]; } + // An empty HOME per test, so a real ~/.lich or ~/.config/lich never leaks in. + saved_env["HOME"] = process.env.HOME; + process.env.HOME = make_temp_dir("home"); }); afterEach(() => { @@ -373,10 +365,13 @@ describe("lich init and bare lich", () => { } }); - it("pins a home-config skip into the load instead of a later discovery", async () => { + it("skips setup on a global ~/.lich/config.json and loads it for --work-dir, not the cwd project", async () => { const dir = make_temp_dir("pin-home"); - const home = path.join(dir, "missing-home-config.json"); - writeFileSync(home, JSON.stringify({ providers: [{ kind: "ollama", name: "main", model: "from-home" }] })); + const global_dir = path.join(String(process.env.HOME), ".lich"); + mkdirSync(global_dir, { recursive: true }); + writeFileSync(path.join(global_dir, "config.json"), JSON.stringify({ + providers: [{ kind: "ollama", name: "main", model: "from-home" }], + })); const cwd_dir = make_temp_dir("pin-cwd"); mkdirSync(path.join(cwd_dir, ".lich"), { recursive: true }); writeFileSync(path.join(cwd_dir, ".lich", "config.json"), JSON.stringify({ @@ -399,6 +394,32 @@ describe("lich init and bare lich", () => { } }); + it("bare lich merges the project config over the global one", async () => { + const dir = make_temp_dir("merge"); + const global_dir = path.join(String(process.env.HOME), ".lich"); + mkdirSync(global_dir, { recursive: true }); + writeFileSync(path.join(global_dir, "config.json"), JSON.stringify({ + agent_name: "wight", + max_turns: 7, + providers: [{ kind: "ollama", name: "main", model: "from-home" }], + })); + mkdirSync(path.join(dir, ".lich"), { recursive: true }); + writeFileSync(project_config_path(dir), JSON.stringify({ max_turns: 3 })); + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + expect(await run_cli(["--work-dir", dir])).toBe(0); + expect(writes.calls).toHaveLength(0); + const config = tui_run.configs[0]; + expect(config?.agent_name).toBe("wight"); + expect(config?.max_turns).toBe(3); + expect(config?.providers?.[0]?.model).toBe("from-home"); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + it("lets lich tui pick up the file lich init wrote", async () => { await without_model_env(async () => { const dir = make_temp_dir("tui-pickup"); From dc815f3a1bd2b0f8d1772ed94e5fb40ded1b7384 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 22:30:49 +0000 Subject: [PATCH 30/43] fix(config): read the legacy user config only when ~/.lich/config.json is absent With work_dir = home the global file is the project file; the legacy ~/.config/lich file was then picked as the base under it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/cli_config.ts | 17 ++++++++--------- test/cli_config.test.ts | 3 ++- 2 files changed, 10 insertions(+), 10 deletions(-) diff --git a/src/cli_config.ts b/src/cli_config.ts index 5373648..ad044ef 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -88,17 +88,16 @@ export interface LayeredConfig { */ export function load_layered_config(work_dir: string): LayeredConfig | undefined { const project_path = project_config_path(work_dir); + const global_path = global_config_path(); + const legacy = path.resolve(homedir(), LEGACY_USER_CONFIG); const notes: string[] = []; let base_path: string | undefined; - for (const candidate of config_search_paths(work_dir).slice(1)) { - if (candidate !== project_path && existsSync(candidate) === true) { - base_path = candidate; - break; - } - } - const legacy = path.resolve(homedir(), LEGACY_USER_CONFIG); - if (base_path === legacy) { - notes.push(`lich: reading ${legacy}; move it to ${global_config_path()} (the old location is read-only)`); + if (existsSync(global_path) === true) { + // When work_dir is home, the global file is the project file: no base layer. + base_path = global_path === project_path ? undefined : global_path; + } else if (existsSync(legacy) === true) { + base_path = legacy; + notes.push(`lich: reading ${legacy}; move it to ${global_path} (the old location is read-only)`); } const project = existsSync(project_path) === true ? read_config_object(project_path) : undefined; const base = base_path === undefined ? undefined : global_layer(read_config_object(base_path), path.dirname(base_path)); diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 18d4bd6..2dbcccd 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -272,8 +272,9 @@ describe("load_layered_config", () => { expect(current?.notes).toEqual([]); }); - it("does not merge ~/.lich/config.json over itself when the work_dir is the home directory", () => { + it("does not merge ~/.lich/config.json over itself, or the legacy file under it, when the work_dir is home", () => { write_json(global_config_path(), { agent_name: "home" }); + write_json(path.join(home, ".config", "lich", "config.json"), { theme: "legacy" }); const layered = load_layered_config(home); expect(layered?.config).toEqual({ agent_name: "home" }); expect(layered?.sources).toEqual([global_config_path()]); From 146619b33267c5106e88d1bb8d7cf5c2d7993108 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 00:44:30 +0000 Subject: [PATCH 31/43] feat(config): lich init --global and a "save as global" wizard step (#117) lich init --global writes the starter config to ~/.lich/config.json (never overwrites; no work_dir/session_dir). The setup wizard ends with "Save as the global default?" (default no). Yes writes the answers to ~/.lich/config.json; discovered .lich/plugins entries stay in the project file since they are project paths. The wizard no longer pins its written file as --config, so discovery merges project over global. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 4 +++ docs/getting-started.md | 2 +- docs/user-guide/cli.md | 7 +++-- src/cli.ts | 48 +++++++++++++++++++++++------ src/setup_wizard.ts | 10 ++++++ test/first_run.test.ts | 67 ++++++++++++++++++++++++++++++++++++++++- 6 files changed, 124 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c6d3e35..3c1c11b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased +- `lich init --global` writes the starter config to `~/.lich/config.json` + (never overwrites). The setup wizard ends with "Save as the global default?" + (default no); yes writes the answers to `~/.lich/config.json` and keeps + discovered project plugins in the project file (#117). - Global defaults live in `~/.lich/config.json`. A project `.lich/config.json` is merged over it key by key instead of hiding it (#117). A project `providers` array replaces the global one and drops the global `models` diff --git a/docs/getting-started.md b/docs/getting-started.md index 63be57e..671e64e 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -42,7 +42,7 @@ lich config > .lich/config.json `lich config` honors `LICH_PROVIDER_KIND` and `LICH_MODEL` when you have them set, and otherwise prints an ollama-oriented template. The file is picked up automatically from `.lich/config.json` in the working directory, merged over `~/.lich/config.json` if you keep defaults there — after this, plain `lich "task"` needs no env vars. -`lich init` writes that same starter file for you (it creates `.lich/` and never overwrites an existing `.lich/config.json`). Bare `lich` on a TTY, with no config in that search chain, runs a setup wizard and writes `.lich/config.json` once before opening the TUI. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. Non-TTY stdin skips the wizard. `.lich/` is gitignored. +`lich init` writes that same starter file for you (it creates `.lich/` and never overwrites an existing `.lich/config.json`). Bare `lich` on a TTY, with no config in that search chain, runs a setup wizard and writes `.lich/config.json` once before opening the TUI. Answer yes to its last question to save the answers as your global default in `~/.lich/config.json` (or run `lich init --global`), so new projects skip the wizard. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. Non-TTY stdin skips the wizard. `.lich/` is gitignored. ### Path C: an explicit config file diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index 017c589..334b856 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -7,6 +7,7 @@ ```sh lich # open the TUI; first run on a TTY starts the setup wizard lich init # write .lich/config.json without the wizard (flags apply; never overwrites) +lich init --global # write ~/.lich/config.json, the defaults every project inherits lich "one shot task" # run a single task and print the reply lich chat # interactive chat (commands: /exit, /quit) lich tui # interactive terminal UI (ink) @@ -24,8 +25,8 @@ lich --help # usage text lich --version # package.json version (published package and this tree: 0.8.0) ``` -- **Bare `lich`** opens the same TUI as `lich tui`. It does not print usage. On a TTY, if none of `.lich/config.json`, `~/.lich/config.json` or the legacy `~/.config/lich/config.json` exists, a setup wizard runs first (name, provider, optional gateway env-var names, optional plugins) and writes `.lich/config.json` once. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. An existing config in that chain skips the wizard and is not replaced. Non-TTY stdin skips the wizard and prints guidance instead of hanging. `lich --help` still prints usage. -- **`lich init`** writes that starter file without prompts, using the same writer as the wizard. Existing flags such as `--model` are written into the file and win over `LICH_MODEL`. It never overwrites an existing `.lich/config.json`. `.lich/` is gitignored. +- **Bare `lich`** opens the same TUI as `lich tui`. It does not print usage. On a TTY, if none of `.lich/config.json`, `~/.lich/config.json` or the legacy `~/.config/lich/config.json` exists, a setup wizard runs first (name, provider, optional gateway env-var names, optional plugins) and writes `.lich/config.json` once. Its last question, "Save as the global default?" (default no), writes the answers to `~/.lich/config.json` instead, so later projects skip the wizard; plugins found under the project's `.lich/plugins` still go to the project file. `LICH_MODEL` / `--model` prefills the model prompt; it does not skip the wizard. An existing config in that chain skips the wizard and is not replaced. Non-TTY stdin skips the wizard and prints guidance instead of hanging. `lich --help` still prints usage. +- **`lich init`** writes that starter file without prompts, using the same writer as the wizard. Existing flags such as `--model` are written into the file and win over `LICH_MODEL`. It never overwrites an existing `.lich/config.json`. `.lich/` is gitignored. `lich init --global` writes the same starter to `~/.lich/config.json` instead (never overwriting it, and without `work_dir` or `session_dir`). - **One-shot** joins all positional words into a single task, runs the agent loop, prints the final answer to stdout, and exits. Progress (turn numbers, tool results) goes to stderr. Ctrl+C cancels the run (exit `1`); a second Ctrl+C quits at once (exit `130`). - **Chat** is a readline REPL over one long-lived agent: each line is a turn, memory persists across lines, and an empty line, `/exit`, or `/quit` ends the session. After each turn it prints a `[turns N | tokens M]` footer. Ctrl+C cancels the running turn and keeps the session; at the prompt it ends chat. A second Ctrl+C while a turn is still cancelling quits at once (exit `130`). - **TUI** launches the ink interface. See the [TUI guide](tui.md). @@ -92,7 +93,7 @@ Per-kind defaults: - **Legacy location:** `~/.config/lich/config.json` is still read when `~/.lich/config.json` is absent, with a one-line hint to move it. Lich never writes there. - **`--config `** replaces the whole chain; nothing is merged. -`lich init`, the setup wizard and `lich mcp` write only the project file. +`lich init` and `lich mcp` write only the project file. `lich init --global` and the wizard's "save as global" answer write `~/.lich/config.json`. ## Config file reference diff --git a/src/cli.ts b/src/cli.ts index 1bca43c..2c8040c 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -4,6 +4,7 @@ * config files, and environment-based provider resolution. */ import { existsSync, readFileSync, realpathSync } from "node:fs"; +import { homedir } from "node:os"; import { createInterface } from "node:readline"; import path from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; @@ -16,6 +17,7 @@ import { load_layered_config, config_template, existing_config_path, + global_config_path, provider_kind_defaults, starter_config_object, write_lich_config, @@ -26,7 +28,7 @@ import { empty_mcp_flags, take_mcp_flag, type McpCliFlags } from "./cli_mcp_flag import { package_root_from_module_url, run_ossuary } from "./cli_ossuary.js"; import { run_update } from "./cli_update.js"; import type { Message } from "./providers/types.js"; -import { ask_line as ask_wizard_line, build_setup_config, collect_setup_answers } from "./setup_wizard.js"; +import { ask_line as ask_wizard_line, ask_save_global, build_setup_config, collect_setup_answers } from "./setup_wizard.js"; import { is_enoent } from "./util/fs.js"; import { load_theme, notice_flavor } from "./util/theme.js"; @@ -44,6 +46,8 @@ interface CliOptions { positionals: string[]; mcp_flags: McpCliFlags; serve_flags: ServeCliFlags; + /** `lich init --global`: write ~/.lich/config.json instead of the project file. */ + global?: boolean; } const FLAG_KEYS: Record = { @@ -66,6 +70,7 @@ function usage_text(): string { "Usage:", " lich open the TUI (first run: setup wizard, then TUI)", " lich init write .lich/config.json without the wizard (flags apply; never overwrites)", + " lich init --global write ~/.lich/config.json, the defaults every project inherits", ' lich "one shot task" run a single task and print the reply', " lich chat interactive chat (commands: /exit, /quit)", " lich tui interactive terminal UI (ink)", @@ -223,6 +228,7 @@ export function parse_args(argv: string[]): CliOptions { }; const mcp = argv.includes("mcp"); const serve = first_positional_of(argv) === "serve"; + const init = first_positional_of(argv) === "init"; for (let index = 0; index < argv.length; index += 1) { const arg = argv[index]; if (arg === undefined) { @@ -261,6 +267,10 @@ export function parse_args(argv: string[]): CliOptions { continue; } } + if (init === true && arg === "--global") { + options.global = true; + continue; + } if (serve === true) { const consumed = take_serve_flag(argv, index, options.serve_flags); if (consumed !== undefined) { @@ -582,16 +592,30 @@ function work_dir_of(options: CliOptions): string { return options.overrides["work_dir"] ?? process.cwd(); } -async function offer_wizard(work_dir: string, hint?: string): Promise { +async function offer_wizard(work_dir: string, hint?: string): Promise<"written" | "cancelled"> { const rl = createInterface({ input: process.stdin, output: process.stdout }); try { - const answers = await collect_setup_answers(work_dir, (prompt) => ask_wizard_line(rl, prompt), hint); + const ask = (prompt: string): Promise => ask_wizard_line(rl, prompt); + const answers = await collect_setup_answers(work_dir, ask, hint); if (answers === undefined) { return "cancelled"; } - const result = write_lich_config(work_dir, build_setup_config(answers)); - process.stdout.write(`${result.message}\n`); - return result.path; + const save_global = await ask_save_global(ask, global_config_path()); + if (save_global === undefined) { + return "cancelled"; + } + const config = build_setup_config(answers); + if (save_global === false) { + process.stdout.write(`${write_lich_config(work_dir, config).message}\n`); + return "written"; + } + // Discovered plugins are project paths (.lich/plugins/...), so they stay in the project file. + const { plugins, ...global_config } = config; + process.stdout.write(`${write_lich_config(homedir(), global_config).message}\n`); + if (Array.isArray(plugins) === true && plugins.length > 0) { + process.stdout.write(`${write_lich_config(work_dir, { plugins }).message}\n`); + } + return "written"; } finally { rl.close(); } @@ -612,7 +636,7 @@ async function maybe_first_run(options: CliOptions, work_dir: string): Promise { + const answer = await ask(`Save as the global default in ${global_path} (every project inherits it)? [y/N]: `); + if (answer === undefined) { + return undefined; + } + const yes = answer.trim().toLowerCase(); + return yes === "y" || yes === "yes"; +} + /** All steps skippable. Undefined means Ctrl+C or a rejected env name: caller must not write. */ export async function collect_setup_answers(work_dir: string, ask: AskLine, model_hint?: string): Promise { const agent_name = await ask_name(ask); diff --git a/test/first_run.test.ts b/test/first_run.test.ts index 5cb8987..4be7897 100644 --- a/test/first_run.test.ts +++ b/test/first_run.test.ts @@ -58,7 +58,7 @@ vi.mock("../src/gateway.js", () => ({ })); import { parse_agent_config } from "../src/agent/config.js"; -import { config_search_paths, existing_config_path, project_config_path, write_lich_config } from "../src/cli_config.js"; +import { config_search_paths, existing_config_path, global_config_path, project_config_path, write_lich_config } from "../src/cli_config.js"; import { read_platform_token } from "../src/gateway/token_env.js"; import { build_setup_config, collect_setup_answers } from "../src/setup_wizard.js"; import { run_cli } from "../src/cli.js"; @@ -420,6 +420,71 @@ describe("lich init and bare lich", () => { } }); + it("wizard can save the answers as the global default and keep project plugins in the project file", async () => { + const dir = make_temp_dir("wizard-global"); + mkdirSync(path.join(dir, ".lich", "plugins"), { recursive: true }); + writeFileSync(path.join(dir, ".lich", "plugins", "demo.ts"), "export const plugin = { name: 'demo', tools: [] };\n"); + // name, kind, model, base url, key env, gateway, plugin demo.ts, save global + wizard.lines = ["ada", "", "global-model", "", "", "", "y", "y"]; + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + expect(await run_cli(["--work-dir", dir])).toBe(0); + const global = JSON.parse(readFileSync(global_config_path(), "utf8")) as Record; + expect(global["agent_name"]).toBe("ada"); + expect(global["plugins"]).toBeUndefined(); + expect(JSON.parse(readFileSync(project_config_path(dir), "utf8"))).toEqual({ plugins: [".lich/plugins/demo.ts"] }); + const config = tui_run.configs[0]; + expect(config?.agent_name).toBe("ada"); + expect(config?.providers?.[0]?.model).toBe("global-model"); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + + it("wizard writes only the global file when saving globally with no plugins, and only the project file by default", async () => { + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + const global_dir = make_temp_dir("wizard-global-only"); + wizard.lines = ["", "", "m1", "", "", "", "yes"]; + expect(await run_cli(["--work-dir", global_dir])).toBe(0); + expect(existsSync(global_config_path())).toBe(true); + expect(existsSync(project_config_path(global_dir))).toBe(false); + + rmSync(global_config_path()); + const project_dir = make_temp_dir("wizard-project-only"); + wizard.lines = ["", "", "m2", "", "", ""]; + expect(await run_cli(["--work-dir", project_dir])).toBe(0); + expect(existsSync(project_config_path(project_dir))).toBe(true); + expect(existsSync(global_config_path())).toBe(false); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + + it("lich init --global writes ~/.lich/config.json without work_dir, and never overwrites it", async () => { + await without_model_env(async () => { + const dir = make_temp_dir("init-global"); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + expect(await run_cli(["init", "--global", "--model", "g-model", "--work-dir", dir])).toBe(0); + const global = JSON.parse(readFileSync(global_config_path(), "utf8")) as Record; + expect((global["providers"] as Array<{ model: string }>)[0]?.model).toBe("g-model"); + expect(global["work_dir"]).toBeUndefined(); + expect(existsSync(project_config_path(dir))).toBe(false); + const before = readFileSync(global_config_path(), "utf8"); + expect(await run_cli(["init", "--global", "--model", "other"])).toBe(0); + expect(readFileSync(global_config_path(), "utf8")).toBe(before); + } finally { + stdout.mockRestore(); + } + await expect(run_cli(["tui", "--global"])).rejects.toThrow("unknown flag: --global"); + }); + }); + it("lets lich tui pick up the file lich init wrote", async () => { await without_model_env(async () => { const dir = make_temp_dir("tui-pickup"); From 2faf90cee55828eef4b50c28d2144528f67958ff Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 00:50:23 +0000 Subject: [PATCH 32/43] fix(wizard): write the config once, plugins included, when work_dir is home With work_dir = home the project file is ~/.lich/config.json, so the plugins-only project write hit "already exists" and dropped plugins. Test also asserts the merged TUI plugins. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/cli.ts | 4 +++- test/first_run.test.ts | 22 +++++++++++++++++++++- 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/src/cli.ts b/src/cli.ts index 2c8040c..b9f31b0 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -18,6 +18,7 @@ import { config_template, existing_config_path, global_config_path, + project_config_path, provider_kind_defaults, starter_config_object, write_lich_config, @@ -605,7 +606,8 @@ async function offer_wizard(work_dir: string, hint?: string): Promise<"written" return "cancelled"; } const config = build_setup_config(answers); - if (save_global === false) { + // With work_dir = home the project file is the global file: write it once, plugins included. + if (save_global === false || project_config_path(work_dir) === global_config_path()) { process.stdout.write(`${write_lich_config(work_dir, config).message}\n`); return "written"; } diff --git a/test/first_run.test.ts b/test/first_run.test.ts index 4be7897..4ff100e 100644 --- a/test/first_run.test.ts +++ b/test/first_run.test.ts @@ -9,7 +9,7 @@ import { TMP_BASE } from "./helpers/tmp_base.js"; const writes = vi.hoisted(() => ({ calls: [] as Array<{ work_dir: string; config: Record }> })); const wizard = vi.hoisted(() => ({ cancel: false, lines: [] as string[] })); -const tui_run = vi.hoisted(() => ({ configs: [] as Array<{ agent_name?: string; max_turns?: number; providers?: Array<{ model?: string }> }> })); +const tui_run = vi.hoisted(() => ({ configs: [] as Array<{ agent_name?: string; max_turns?: number; plugins?: unknown[]; providers?: Array<{ model?: string }> }> })); const gateway_run = vi.hoisted(() => ({ fn: vi.fn(async (_config: unknown, _platforms: readonly string[]) => 0), })); @@ -437,6 +437,26 @@ describe("lich init and bare lich", () => { const config = tui_run.configs[0]; expect(config?.agent_name).toBe("ada"); expect(config?.providers?.[0]?.model).toBe("global-model"); + expect(config?.plugins).toEqual([".lich/plugins/demo.ts"]); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + + it("wizard writes the global file once, plugins included, when the work_dir is home", async () => { + const home = String(process.env.HOME); + mkdirSync(path.join(home, ".lich", "plugins"), { recursive: true }); + writeFileSync(path.join(home, ".lich", "plugins", "demo.ts"), "export const plugin = { name: 'demo', tools: [] };\n"); + wizard.lines = ["", "", "home-model", "", "", "", "y", "y"]; + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + expect(await run_cli(["--work-dir", home])).toBe(0); + const written = JSON.parse(readFileSync(global_config_path(), "utf8")) as Record; + expect(written["plugins"]).toEqual([".lich/plugins/demo.ts"]); + expect((written["providers"] as Array<{ model: string }>)[0]?.model).toBe("home-model"); + expect(tui_run.configs[0]?.plugins).toEqual([".lich/plugins/demo.ts"]); } finally { stdout.mockRestore(); restore_tty(); From eff0ea1b9115ebdb17cdab07640b4a13330e926c Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 00:54:45 +0000 Subject: [PATCH 33/43] fix(wizard): store absolute plugin paths when writing ~/.lich/config.json from home Other projects load that file as the global layer and resolve relative plugin paths against ~/.lich, so .lich/plugins/x would become ~/.lich/.lich/plugins/x. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/cli.ts | 11 +++++++++-- test/first_run.test.ts | 12 +++++++++--- 2 files changed, 18 insertions(+), 5 deletions(-) diff --git a/src/cli.ts b/src/cli.ts index b9f31b0..d3a0310 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -606,8 +606,15 @@ async function offer_wizard(work_dir: string, hint?: string): Promise<"written" return "cancelled"; } const config = build_setup_config(answers); - // With work_dir = home the project file is the global file: write it once, plugins included. - if (save_global === false || project_config_path(work_dir) === global_config_path()) { + if (project_config_path(work_dir) === global_config_path()) { + // With work_dir = home the project file is the global file: write it once. Other + // projects resolve its relative plugin paths against ~/.lich, so store them absolute. + const plugins = Array.isArray(config["plugins"]) === true ? (config["plugins"] as string[]) : []; + config["plugins"] = plugins.map((entry) => path.resolve(work_dir, entry)); + process.stdout.write(`${write_lich_config(work_dir, config).message}\n`); + return "written"; + } + if (save_global === false) { process.stdout.write(`${write_lich_config(work_dir, config).message}\n`); return "written"; } diff --git a/test/first_run.test.ts b/test/first_run.test.ts index 4ff100e..264b9b8 100644 --- a/test/first_run.test.ts +++ b/test/first_run.test.ts @@ -444,7 +444,7 @@ describe("lich init and bare lich", () => { } }); - it("wizard writes the global file once, plugins included, when the work_dir is home", async () => { + it("wizard writes the global file once, with absolute plugin paths, when the work_dir is home", async () => { const home = String(process.env.HOME); mkdirSync(path.join(home, ".lich", "plugins"), { recursive: true }); writeFileSync(path.join(home, ".lich", "plugins", "demo.ts"), "export const plugin = { name: 'demo', tools: [] };\n"); @@ -454,9 +454,15 @@ describe("lich init and bare lich", () => { try { expect(await run_cli(["--work-dir", home])).toBe(0); const written = JSON.parse(readFileSync(global_config_path(), "utf8")) as Record; - expect(written["plugins"]).toEqual([".lich/plugins/demo.ts"]); + // Absolute, so other projects that load this file as their global layer find the plugin too. + const absolute = path.join(home, ".lich", "plugins", "demo.ts"); + expect(written["plugins"]).toEqual([absolute]); expect((written["providers"] as Array<{ model: string }>)[0]?.model).toBe("home-model"); - expect(tui_run.configs[0]?.plugins).toEqual([".lich/plugins/demo.ts"]); + expect(tui_run.configs[0]?.plugins).toEqual([absolute]); + tui_run.configs = []; + const other = make_temp_dir("other-project"); + expect(await run_cli(["--work-dir", other])).toBe(0); + expect(tui_run.configs[0]?.plugins).toEqual([absolute]); } finally { stdout.mockRestore(); restore_tty(); From eaadb017db1c0adc53355e5ae0295e58b85a8d6a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 02:26:54 +0000 Subject: [PATCH 34/43] feat(config): named profiles in ~/.lich/profiles (#117) Phase 2 of #117. ~/.lich/profiles/.json, plus an optional .md used as the system prompt, merges between the global and the project config (global < profile < project). Selection: --profile, then LICH_PROFILE, then a project `profile` key, then the default in ~/.lich/config.json. Profile plugin paths resolve against ~/.lich/profiles; work_dir/session_dir in a profile are ignored. New `lich profile list|show|create|use`: create runs the wizard into the profile file (never overwrites), use sets `profile` in the global file. A requested profile skips the first-run wizard; --profile with --config is an error. File tools now refuse .lich/profiles/. tui/chat/serve/gateway no longer replace every config error with "no model configured"; only that case gets the hint. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 8 +++ docs/.vitepress/config.mts | 1 + docs/user-guide/cli.md | 2 + docs/user-guide/profiles.md | 59 +++++++++++++++ src/cli.ts | 33 +++++++-- src/cli_config.ts | 69 +++++++++++++++--- src/cli_profile.ts | 140 ++++++++++++++++++++++++++++++++++++ src/tools/guard.ts | 6 +- test/cli_config.test.ts | 86 ++++++++++++++++++++-- test/first_run.test.ts | 68 +++++++++++++++++- test/tools.test.ts | 12 ++++ 11 files changed, 463 insertions(+), 21 deletions(-) create mode 100644 docs/user-guide/profiles.md create mode 100644 src/cli_profile.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 3c1c11b..dd66972 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,14 @@ ## Unreleased +- Profiles (#117): `~/.lich/profiles/.json` (plus an optional `.md` + used as the system prompt) is merged between the global and project config. + Select one with `--profile`, `LICH_PROFILE`, a project `profile` key, or the + default set by `lich profile use`. New `lich profile list|show|create|use`. + File tools now refuse `.lich/profiles/`. See `docs/user-guide/profiles.md`. +- `lich tui`, `chat`, `serve` and `gateway` now show the real config error + (for example `profile not found`) instead of always saying "no model + configured". - `lich init --global` writes the starter config to `~/.lich/config.json` (never overwrites). The setup wizard ends with "Save as the global default?" (default no); yes writes the answers to `~/.lich/config.json` and keeps diff --git a/docs/.vitepress/config.mts b/docs/.vitepress/config.mts index e2446f4..44dd445 100644 --- a/docs/.vitepress/config.mts +++ b/docs/.vitepress/config.mts @@ -26,6 +26,7 @@ export default defineConfig({ { text: 'Introduction', link: '/' }, { text: 'Getting started', link: '/getting-started' }, { text: 'CLI reference', link: '/user-guide/cli' }, + { text: 'Profiles', link: '/user-guide/profiles' }, { text: 'TUI guide', link: '/user-guide/tui' }, { text: 'Ossuary guide', link: '/user-guide/ossuary' }, { text: 'Gateway guide', link: '/user-guide/gateway' }, diff --git a/docs/user-guide/cli.md b/docs/user-guide/cli.md index 334b856..a471b25 100644 --- a/docs/user-guide/cli.md +++ b/docs/user-guide/cli.md @@ -8,6 +8,7 @@ lich # open the TUI; first run on a TTY starts the setup wizard lich init # write .lich/config.json without the wizard (flags apply; never overwrites) lich init --global # write ~/.lich/config.json, the defaults every project inherits +lich profile list # named configs in ~/.lich/profiles (also show/create/use) lich "one shot task" # run a single task and print the reply lich chat # interactive chat (commands: /exit, /quit) lich tui # interactive terminal UI (ink) @@ -91,6 +92,7 @@ Per-kind defaults: - **Per-project keys:** `work_dir` and `session_dir` in the global file are ignored. - **Paths:** relative `plugins` paths in the global file resolve against `~/.lich/`. MCP `command` and `args` are used as written. - **Legacy location:** `~/.config/lich/config.json` is still read when `~/.lich/config.json` is absent, with a one-line hint to move it. Lich never writes there. +- **Profiles:** a selected profile (`--profile`, `LICH_PROFILE`, or `lich profile use`) is merged between the global and project files. See [Profiles](profiles.md). - **`--config `** replaces the whole chain; nothing is merged. `lich init` and `lich mcp` write only the project file. `lich init --global` and the wizard's "save as global" answer write `~/.lich/config.json`. diff --git a/docs/user-guide/profiles.md b/docs/user-guide/profiles.md new file mode 100644 index 0000000..1a28ef8 --- /dev/null +++ b/docs/user-guide/profiles.md @@ -0,0 +1,59 @@ +# Profiles + +A profile is a named config, such as `coder` or `bard`, that you can switch between without editing project files. Profiles live under `~/.lich/profiles/`, next to the global config. + +``` +~/.lich/ +├── config.json # global defaults (see the CLI reference: Global config) +└── profiles/ + ├── coder.json # any config keys: model, agent_name, max_turns, plugins, ... + └── coder.md # optional "soul": becomes the system prompt +``` + +## Layers + +A run merges up to three files, each overriding the one before it key by key: + +1. `~/.lich/config.json` (global) +2. `~/.lich/profiles/.json` (the selected profile) +3. `/.lich/config.json` (project) + +The merge rules are the same as for the global config. It is a shallow merge. A later `providers` array replaces the earlier one and drops the earlier `models` unless the later layer sets its own. `work_dir` and `session_dir` in a profile are ignored. Relative `plugins` paths in a profile resolve against `~/.lich/profiles/`. + +A profile only needs the keys that differ from your global config. For example, a profile that just changes the model: + +```json +{ "agent_name": "coder", "providers": [{ "kind": "ollama", "name": "main", "model": "qwen3:8b" }] } +``` + +`--config ` still replaces every layer and cannot be combined with `--profile`. + +## The soul file + +When `.md` exists and is not blank, its contents become the profile's `system_prompt`, replacing any `system_prompt` in `.json`. A project `system_prompt` still wins over it. A profile can be only a soul file, with no JSON. + +## Choosing a profile + +The first of these that is set wins: + +1. `--profile ` on the command line +2. the `LICH_PROFILE` environment variable +3. a `profile` key in the project `.lich/config.json` +4. the default profile: a `profile` key in `~/.lich/config.json`, set by `lich profile use` + +A named profile that does not exist is an error (`profile not found: `). Profile names use lowercase letters, digits, `-` and `_`. When `--profile` or `LICH_PROFILE` is set, bare `lich` skips the setup wizard. + +## Commands + +| Command | Effect | +| --- | --- | +| `lich profile list` | List profiles. `*` marks the default; `(soul)` marks a profile with a `.md` file. | +| `lich profile show ` | Print the profile's JSON and its soul file's path and length. | +| `lich profile create ` | Run the setup wizard and write `~/.lich/profiles/.json`. Needs a TTY; never overwrites. | +| `lich profile use ` | Make `` the default by setting `profile` in `~/.lich/config.json` (other keys are kept). | + +To clear the default, remove the `profile` key from `~/.lich/config.json`. + +## Safety + +File tools cannot read or write `.lich/profiles/` under the working directory, just as they cannot touch `.lich/config.json`. When the working directory is your home directory, these are your global identity files. Profiles hold env var *names* for keys (`api_key_env`, `token_envs`), never the secrets themselves. diff --git a/src/cli.ts b/src/cli.ts index d3a0310..6665f50 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -25,6 +25,7 @@ import { type ProviderKind, } from "./cli_config.js"; import { run_mcp } from "./cli_mcp.js"; +import { run_profile } from "./cli_profile.js"; import { empty_mcp_flags, take_mcp_flag, type McpCliFlags } from "./cli_mcp_flags.js"; import { package_root_from_module_url, run_ossuary } from "./cli_ossuary.js"; import { run_update } from "./cli_update.js"; @@ -62,6 +63,7 @@ const FLAG_KEYS: Record = { "--session-dir": "session_dir", "--log-level": "log_level", "--theme": "theme", + "--profile": "profile", }; function usage_text(): string { @@ -83,11 +85,14 @@ function usage_text(): string { " lich mcp list list mcp servers in .lich/config.json", " lich mcp add add a catalog or --command/--url server (disabled)", " lich mcp enable / disable / remove ", + " lich profile list list profiles in ~/.lich/profiles (* = default)", + " lich profile create / show / use ", " lich --help show this help", " lich --version print version", "", "Flags (before or after the subcommand):", " --config JSON config file parsed by parse_agent_config", + " --profile merge ~/.lich/profiles/.json between global and project config (or LICH_PROFILE)", " --work-dir working directory for tools", " --max-turns loop turn budget (default 25)", " --model model name (default from LICH_MODEL)", @@ -129,6 +134,7 @@ function non_tui_resume_mode(first: string | undefined): string | undefined { first === "init" || first === "config" || first === "mcp" || + first === "profile" || first === "update" || first === "chat" || first === "gateway" || @@ -142,6 +148,11 @@ function non_tui_resume_mode(first: string | undefined): string | undefined { /** Mode-aware config failure message; one-shot keeps the generic variant. */ function error_for_mode(mode: string, base_message: string): string { + const modes = ["tui", "gateway", "serve", "chat"]; + // Only the missing-model case gets the hint; other errors (bad file, unknown profile) pass through. + if (modes.includes(mode) === true && base_message.startsWith("no model configured") === false) { + return `lich ${mode}: ${base_message}`; + } if (mode === "tui") { return "lich tui: no model configured — set LICH_MODEL (e.g. glm-5.3-flash:cloud), pass --model, or create .lich/config.json (`lich config` prints a template)"; } @@ -395,9 +406,12 @@ function apply_overrides(config: Record, overrides: Record | undefined { - const layered = load_layered_config(work_dir); +/** + * Project `.lich/config.json` (under `--work-dir`, else cwd) merged over the + * global `~/.lich/config.json`, with the selected profile in between. + */ +function load_discovered_config(work_dir: string, profile?: string): Record | undefined { + const layered = load_layered_config(work_dir, profile); for (const note of layered?.notes ?? []) { process.stderr.write(`${note}\n`); } @@ -405,9 +419,14 @@ function load_discovered_config(work_dir: string): Record | und } function build_config(options: CliOptions): AgentConfig { + if (options.config_path !== undefined && options.overrides["profile"] !== undefined) { + throw new Error("--profile cannot be combined with --config (--config replaces every layer)"); + } const base = options.config_path === undefined - ? load_discovered_config(work_dir_of(options)) ?? { providers: [env_provider(options.overrides)] } + ? load_discovered_config(work_dir_of(options), options.overrides["profile"]) ?? { + providers: [env_provider(options.overrides)], + } : load_config_file(options.config_path); apply_overrides(base, options.overrides); const providers = base["providers"]; @@ -631,7 +650,8 @@ async function offer_wizard(work_dir: string, hint?: string): Promise<"written" } async function maybe_first_run(options: CliOptions, work_dir: string): Promise { - if (options.config_path !== undefined) { + // An explicit file or a requested profile is the setup; build_config reports a missing one. + if (options.config_path !== undefined || options.overrides["profile"] !== undefined || (process.env.LICH_PROFILE ?? "").length > 0) { return undefined; } const existing = existing_config_path(work_dir); @@ -798,6 +818,9 @@ export async function run_cli(argv: string[]): Promise { if (first === "mcp") { return run_mcp(work_dir_of(options), options.positionals, options.mcp_flags); } + if (first === "profile") { + return run_profile(options.positionals); + } if (first === "update") { if (options.positionals.length > 1) { throw new Error("update takes no arguments"); diff --git a/src/cli_config.ts b/src/cli_config.ts index ad044ef..75a0e36 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -13,6 +13,7 @@ const CONFIG_RELPATH = `${LICH_DIRNAME}/config.json`; const LEGACY_USER_CONFIG = ".config/lich/config.json"; /** Keys that stay per project: a global layer never sets them. */ const PROJECT_ONLY_KEYS = ["work_dir", "session_dir"] as const; +const PROFILE_NAME = /^[a-z0-9][a-z0-9_-]*$/; export type ProviderKind = "openai_compat" | "anthropic" | "ollama"; @@ -80,13 +81,50 @@ export interface LayeredConfig { sources: string[]; /** One-line notices for the user (for example, the legacy location in use). */ notes: string[]; + /** The profile merged in, when one was selected. */ + profile?: string; +} + +/** `~/.lich/profiles`: one `.json` (and optional `.md` soul) per profile. */ +export function profiles_dir(): string { + return path.resolve(homedir(), LICH_DIRNAME, "profiles"); +} + +/** The JSON and soul paths for a profile name; throws on a name that is not a plain slug. */ +export function profile_paths(name: string): { json: string; soul: string } { + if (PROFILE_NAME.test(name) === false) { + throw new Error(`invalid profile name "${name}" (use lowercase letters, digits, - and _)`); + } + return { json: path.join(profiles_dir(), `${name}.json`), soul: path.join(profiles_dir(), `${name}.md`) }; +} + +/** + * A profile as a layer: its JSON (paths resolved against `~/.lich/profiles`) with + * a non-empty soul file as `system_prompt`. Throws when neither file exists. + */ +function profile_layer(name: string): { layer: Record; sources: string[] } { + const paths = profile_paths(name); + const has_json = existsSync(paths.json); + const has_soul = existsSync(paths.soul); + if (has_json === false && has_soul === false) { + throw new Error(`profile not found: ${name} (expected ${paths.json})`); + } + const layer = has_json === true ? global_layer(read_config_object(paths.json), profiles_dir()) : {}; + delete layer["profile"]; + const soul = has_soul === true ? readFileSync(paths.soul, "utf8").trim() : ""; + if (soul.length > 0) { + layer["system_prompt"] = soul; + } + return { layer, sources: [paths.json, paths.soul].filter((file) => existsSync(file) === true) }; } /** * Project `.lich/config.json` merged over the global base (`~/.lich/config.json`, - * else the legacy `~/.config/lich/config.json`). Undefined when neither exists. + * else the legacy `~/.config/lich/config.json`), with a selected profile in + * between. The profile is `profile_flag`, else `LICH_PROFILE`, else a `profile` + * key in the project file, else one in the global file. Undefined when no file exists. */ -export function load_layered_config(work_dir: string): LayeredConfig | undefined { +export function load_layered_config(work_dir: string, profile_flag?: string): LayeredConfig | undefined { const project_path = project_config_path(work_dir); const global_path = global_config_path(); const legacy = path.resolve(homedir(), LEGACY_USER_CONFIG); @@ -101,13 +139,28 @@ export function load_layered_config(work_dir: string): LayeredConfig | undefined } const project = existsSync(project_path) === true ? read_config_object(project_path) : undefined; const base = base_path === undefined ? undefined : global_layer(read_config_object(base_path), path.dirname(base_path)); - if (project === undefined && base === undefined) { + const selected = first_profile_name([profile_flag, process.env.LICH_PROFILE, project?.["profile"], base?.["profile"]]); + const profile = selected === undefined ? undefined : profile_layer(selected); + if (project === undefined && base === undefined && profile === undefined) { return undefined; } - const sources = [base_path, project === undefined ? undefined : project_path].filter( - (entry): entry is string => entry !== undefined, - ); - return { config: merge_config_layers(base ?? {}, project ?? {}), sources, notes }; + const sources = [ + ...(base_path === undefined ? [] : [base_path]), + ...(profile?.sources ?? []), + ...(project === undefined ? [] : [project_path]), + ]; + const config = merge_config_layers(merge_config_layers(base ?? {}, profile?.layer ?? {}), project ?? {}); + delete config["profile"]; + return { config, sources, notes, ...(selected === undefined ? {} : { profile: selected }) }; +} + +function first_profile_name(candidates: readonly unknown[]): string | undefined { + for (const candidate of candidates) { + if (typeof candidate === "string" && candidate.length > 0) { + return candidate; + } + } + return undefined; } /** @@ -150,7 +203,7 @@ function global_layer(config: Record, home: string): Record { +export function read_config_object(found: string): Record { const raw = readFileSync(found, "utf8"); const parsed = safe_json_parse(raw); if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) { diff --git a/src/cli_profile.ts b/src/cli_profile.ts new file mode 100644 index 0000000..8cd6912 --- /dev/null +++ b/src/cli_profile.ts @@ -0,0 +1,140 @@ +/** + * `lich profile`: named configs under ~/.lich/profiles. A profile is merged + * between the global ~/.lich/config.json and the project config; `.md` + * next to it, when present, becomes its system prompt. + */ +import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs"; +import { homedir } from "node:os"; +import { createInterface } from "node:readline"; +import { global_config_path, profile_paths, profiles_dir, read_config_object, write_lich_config } from "./cli_config.js"; +import { ask_line, build_setup_config, collect_setup_answers } from "./setup_wizard.js"; + +function write_line(line: string): void { + process.stdout.write(`${line}\n`); +} + +function require_name(name: string | undefined, action: string): string { + if (name === undefined || name.length === 0) { + throw new Error(`profile ${action} requires a name`); + } + profile_paths(name); + return name; +} + +function read_global(): Record { + return existsSync(global_config_path()) === true ? read_config_object(global_config_path()) : {}; +} + +function sticky_profile(): string | undefined { + const value = read_global()["profile"]; + return typeof value === "string" && value.length > 0 ? value : undefined; +} + +function profile_names(): string[] { + let entries: string[]; + try { + entries = readdirSync(profiles_dir()); + } catch { + return []; + } + const names = new Set(); + for (const entry of entries) { + const match = /^([a-z0-9][a-z0-9_-]*)\.(json|md)$/.exec(entry); + if (match?.[1] !== undefined) { + names.add(match[1]); + } + } + return [...names].sort(); +} + +function list_profiles(): void { + const names = profile_names(); + if (names.length === 0) { + write_line(`no profiles in ${profiles_dir()}`); + return; + } + const active = sticky_profile(); + for (const name of names) { + const soul = existsSync(profile_paths(name).soul) === true ? " (soul)" : ""; + write_line(`${name === active ? "*" : " "} ${name}${soul}`); + } +} + +function show_profile(name: string): void { + const paths = profile_paths(name); + if (existsSync(paths.json) === false && existsSync(paths.soul) === false) { + throw new Error(`profile not found: ${name}`); + } + if (existsSync(paths.json) === true) { + write_line(paths.json); + write_line(readFileSync(paths.json, "utf8").trimEnd()); + } + if (existsSync(paths.soul) === true) { + write_line(`${paths.soul} (system prompt, ${readFileSync(paths.soul, "utf8").trim().length} chars)`); + } +} + +/** Run the setup wizard and write `~/.lich/profiles/.json`; never overwrites. */ +async function create_profile(name: string): Promise { + const paths = profile_paths(name); + if (existsSync(paths.json) === true) { + throw new Error(`profile ${name} already exists at ${paths.json}`); + } + if (process.stdin.isTTY !== true) { + throw new Error("profile create needs a TTY; write the JSON by hand instead"); + } + const rl = createInterface({ input: process.stdin, output: process.stdout }); + let config: Record; + try { + // A directory with no .lich/plugins: project plugins do not belong in a profile. + const answers = await collect_setup_answers(profiles_dir(), (prompt) => ask_line(rl, prompt)); + if (answers === undefined) { + process.stderr.write("lich: profile create cancelled; nothing written\n"); + return 1; + } + config = build_setup_config(answers); + } finally { + rl.close(); + } + mkdirSync(profiles_dir(), { recursive: true }); + // "wx": a profile created meanwhile is never overwritten. + writeFileSync(paths.json, `${JSON.stringify(config, null, 2)}\n`, { encoding: "utf8", flag: "wx" }); + write_line(`wrote ${paths.json}`); + return 0; +} + +/** Record `name` as the sticky default (`profile` key in ~/.lich/config.json). */ +function use_profile(name: string): void { + const paths = profile_paths(name); + if (existsSync(paths.json) === false && existsSync(paths.soul) === false) { + throw new Error(`profile not found: ${name}`); + } + write_lich_config(homedir(), { ...read_global(), profile: name }, true); + write_line(`default profile: ${name} (in ${global_config_path()})`); +} + +export async function run_profile(positionals: readonly string[]): Promise { + const action = positionals[1]; + if (positionals.length > 3) { + throw new Error("profile takes one name"); + } + if (action === "list") { + if (positionals[2] !== undefined) { + throw new Error("profile list takes no name"); + } + list_profiles(); + return 0; + } + if (action === "show") { + show_profile(require_name(positionals[2], "show")); + return 0; + } + if (action === "create") { + return create_profile(require_name(positionals[2], "create")); + } + if (action === "use") { + use_profile(require_name(positionals[2], "use")); + return 0; + } + throw new Error("profile requires list, show, create, or use"); +} diff --git a/src/tools/guard.ts b/src/tools/guard.ts index 438d73e..f237b3b 100644 --- a/src/tools/guard.ts +++ b/src/tools/guard.ts @@ -68,7 +68,8 @@ function reject_symlink_leaf(resolved: string, target: string, base_dir: string) } /** - * Deny `.lich/config.json` to file tools; allow `.lich/` writes only under + * Deny `.lich/config.json` and `.lich/profiles/` to file tools (with work_dir = + * home these are the global identity files); allow `.lich/` writes only under * `skills/` and `plugins/`; deny `.env*` basenames on writes. */ export function assert_file_tool_access(work_dir: string, resolved: string, mode: "read" | "write"): void { @@ -78,6 +79,9 @@ export function assert_file_tool_access(work_dir: string, resolved: string, mode if (parts[0] === ".lich" && parts[1] === "config.json" && parts.length === 2) { throw new Error("forbidden_path: .lich/config.json"); } + if (parts[0] === ".lich" && parts[1] === "profiles") { + throw new Error("forbidden_path: .lich/profiles"); + } if (mode === "write" && parts[0] === ".lich") { const allowed = parts[1] === "skills" || parts[1] === "plugins"; if (allowed === false) { diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 2dbcccd..4e26b17 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -202,17 +202,23 @@ describe("load_layered_config", () => { let saved_home: string | undefined; let home: string; + let saved_profile: string | undefined; + beforeEach(() => { saved_home = process.env.HOME; + saved_profile = process.env.LICH_PROFILE; home = make_temp_dir("home"); process.env.HOME = home; + delete process.env.LICH_PROFILE; }); afterEach(() => { - if (saved_home === undefined) { - delete process.env.HOME; - } else { - process.env.HOME = saved_home; + for (const [key, value] of [["HOME", saved_home], ["LICH_PROFILE", saved_profile]] as const) { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } } }); @@ -281,6 +287,78 @@ describe("load_layered_config", () => { }); }); +describe("load_layered_config profiles", () => { + let saved: { home?: string; profile?: string }; + let home: string; + + beforeEach(() => { + saved = { home: process.env.HOME, profile: process.env.LICH_PROFILE }; + home = make_temp_dir("home"); + process.env.HOME = home; + delete process.env.LICH_PROFILE; + }); + + afterEach(() => { + for (const [key, value] of [["HOME", saved.home], ["LICH_PROFILE", saved.profile]] as const) { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } + } + }); + + function write_file(file: string, body: string): void { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, body); + } + + const profiles = (): string => path.join(home, ".lich", "profiles"); + + it("merges global < profile < project, resolving profile plugin paths against ~/.lich/profiles", () => { + const work = make_temp_dir("work"); + write_file(global_config_path(), JSON.stringify({ agent_name: "global", max_turns: 9, theme: "lich" })); + write_file(path.join(profiles(), "coder.json"), JSON.stringify({ agent_name: "coder", max_turns: 5, plugins: ["p.mjs"] })); + write_file(project_config_path(work), JSON.stringify({ max_turns: 2 })); + const layered = load_layered_config(work, "coder"); + expect(layered?.config).toEqual({ agent_name: "coder", max_turns: 2, theme: "lich", plugins: [path.join(profiles(), "p.mjs")] }); + expect(layered?.profile).toBe("coder"); + expect(layered?.sources).toEqual([global_config_path(), path.join(profiles(), "coder.json"), project_config_path(work)]); + }); + + it("uses a non-empty soul as the system prompt, which a project system_prompt still overrides", () => { + const work = make_temp_dir("work"); + write_file(path.join(profiles(), "bard.json"), JSON.stringify({ system_prompt: "from json" })); + write_file(path.join(profiles(), "bard.md"), " You are a bard.\n"); + expect(load_layered_config(work, "bard")?.config["system_prompt"]).toBe("You are a bard."); + write_file(path.join(profiles(), "bard.md"), " \n"); + expect(load_layered_config(work, "bard")?.config["system_prompt"]).toBe("from json"); + write_file(project_config_path(work), JSON.stringify({ system_prompt: "project" })); + write_file(path.join(profiles(), "bard.md"), "You are a bard."); + expect(load_layered_config(work, "bard")?.config["system_prompt"]).toBe("project"); + }); + + it("picks the flag, then LICH_PROFILE, then the project profile key, then the global one, and drops the key", () => { + const work = make_temp_dir("work"); + for (const name of ["a", "b", "c", "d"]) { + write_file(path.join(profiles(), `${name}.json`), JSON.stringify({ agent_name: name })); + } + write_file(global_config_path(), JSON.stringify({ profile: "d" })); + expect(load_layered_config(work)?.config).toEqual({ agent_name: "d" }); + write_file(project_config_path(work), JSON.stringify({ profile: "c" })); + expect(load_layered_config(work)?.config).toEqual({ agent_name: "c" }); + process.env.LICH_PROFILE = "b"; + expect(load_layered_config(work)?.config).toEqual({ agent_name: "b" }); + expect(load_layered_config(work, "a")?.config).toEqual({ agent_name: "a" }); + }); + + it("throws for a missing profile or a name that is not a plain slug", () => { + const work = make_temp_dir("work"); + expect(() => load_layered_config(work, "ghost")).toThrow("profile not found: ghost"); + expect(() => load_layered_config(work, "../x")).toThrow("invalid profile name"); + }); +}); + describe("merge_config_layers", () => { const base_providers = [{ kind: "ollama", name: "a", model: "m" }]; diff --git a/test/first_run.test.ts b/test/first_run.test.ts index 264b9b8..e81a1df 100644 --- a/test/first_run.test.ts +++ b/test/first_run.test.ts @@ -17,9 +17,9 @@ vi.mock("../src/cli_config.js", async (import_original) => { const actual = await import_original(); return { ...actual, - write_lich_config: (work_dir: string, config: Record) => { + write_lich_config: (work_dir: string, config: Record, update?: boolean) => { writes.calls.push({ work_dir, config }); - return actual.write_lich_config(work_dir, config); + return actual.write_lich_config(work_dir, config, update); }, }; }); @@ -111,7 +111,7 @@ beforeEach(() => { wizard.lines = []; tui_run.configs = []; gateway_run.fn.mockClear(); - for (const key of ["LICH_MODEL", "LICH_PROVIDER_KIND"] as const) { + for (const key of ["LICH_MODEL", "LICH_PROVIDER_KIND", "LICH_PROFILE"] as const) { saved_env[key] = process.env[key]; delete process.env[key]; } @@ -511,6 +511,68 @@ describe("lich init and bare lich", () => { }); }); + it("lich profile create runs the wizard into ~/.lich/profiles/.json and never overwrites", async () => { + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + wizard.lines = ["coder", "", "code-model", "", "", ""]; + expect(await run_cli(["profile", "create", "coder"])).toBe(0); + const file = path.join(String(process.env.HOME), ".lich", "profiles", "coder.json"); + const written = JSON.parse(readFileSync(file, "utf8")) as Record; + expect(written["agent_name"]).toBe("coder"); + expect(written["plugins"]).toEqual([]); + await expect(run_cli(["profile", "create", "coder"])).rejects.toThrow("already exists"); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + + it("lich profile use records the default in ~/.lich/config.json, keeping its other keys, and list marks it", async () => { + const home = String(process.env.HOME); + mkdirSync(path.join(home, ".lich", "profiles"), { recursive: true }); + writeFileSync(path.join(home, ".lich", "profiles", "coder.json"), JSON.stringify({ agent_name: "coder" })); + writeFileSync(path.join(home, ".lich", "profiles", "bard.md"), "You are a bard."); + writeFileSync(global_config_path(), JSON.stringify({ max_turns: 4 })); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + expect(await run_cli(["profile", "use", "coder"])).toBe(0); + expect(JSON.parse(readFileSync(global_config_path(), "utf8"))).toEqual({ max_turns: 4, profile: "coder" }); + stdout.mockClear(); + expect(await run_cli(["profile", "list"])).toBe(0); + expect(stdout.mock.calls.map((call) => String(call[0])).join("")).toBe(" bard (soul)\n* coder\n"); + await expect(run_cli(["profile", "use", "ghost"])).rejects.toThrow("profile not found: ghost"); + } finally { + stdout.mockRestore(); + } + }); + + it("bare lich --profile skips the wizard and runs with the profile; --profile with --config is rejected", async () => { + const home = String(process.env.HOME); + mkdirSync(path.join(home, ".lich", "profiles"), { recursive: true }); + writeFileSync(path.join(home, ".lich", "profiles", "coder.json"), JSON.stringify({ + agent_name: "coder", + providers: [{ kind: "ollama", name: "main", model: "code-model" }], + })); + const dir = make_temp_dir("profile-run"); + const restore_tty = set_tty(true); + const stdout = vi.spyOn(process.stdout, "write").mockImplementation(() => true); + try { + wizard.lines = ["should-not-be-read"]; + expect(await run_cli(["--work-dir", dir, "--profile", "coder"])).toBe(0); + expect(writes.calls).toHaveLength(0); + expect(tui_run.configs[0]?.agent_name).toBe("coder"); + expect(tui_run.configs[0]?.providers?.[0]?.model).toBe("code-model"); + await expect(run_cli(["tui", "--profile", "coder", "--config", global_config_path()])).rejects.toThrow( + "--profile cannot be combined with --config", + ); + await expect(run_cli(["tui", "--profile", "ghost"])).rejects.toThrow("lich tui: profile not found: ghost"); + } finally { + stdout.mockRestore(); + restore_tty(); + } + }); + it("lets lich tui pick up the file lich init wrote", async () => { await without_model_env(async () => { const dir = make_temp_dir("tui-pickup"); diff --git a/test/tools.test.ts b/test/tools.test.ts index 1618561..11cbe25 100644 --- a/test/tools.test.ts +++ b/test/tools.test.ts @@ -454,6 +454,18 @@ describe("forbidden file paths", () => { expect(write.error?.startsWith("forbidden_path")).toBe(true); }); + it("denies .lich/profiles to read and write (global identity files when work_dir is home)", async () => { + await mkdir(path.join(tmp_root, ".lich", "profiles"), { recursive: true }); + await writeFile(path.join(tmp_root, ".lich", "profiles", "coder.md"), "soul\n", "utf8"); + const registry = new ToolRegistry(); + register_builtin_tools(registry); + const executor = make_executor(registry); + const read = await executor.execute("read_file", { path: ".lich/profiles/coder.md" }); + expect(read.error?.startsWith("forbidden_path: .lich/profiles")).toBe(true); + const write = await executor.execute("write_file", { path: ".lich/profiles/coder.json", content: "{}\n" }); + expect(write.error?.startsWith("forbidden_path")).toBe(true); + }); + it("allows writes under .lich/skills and .lich/plugins; denies other .lich writes and .env*", async () => { const registry = new ToolRegistry(); register_builtin_tools(registry); From ae7e9d84f4952f09cb36ff76480a0b4a6df084aa Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 02:33:27 +0000 Subject: [PATCH 35/43] fix(profiles): profile over the home config, stable errors, guard list_dir/disk_usage - work_dir = home: ~/.lich/config.json is the project file, so the profile now merges over it instead of under it. - "profile not found" / "already exists" errors carry no absolute path. - list_dir and disk_usage check the root with assert_file_tool_access, and list_dir skips denied entries, so .lich/profiles is not listed. - Docs: LICH_PROFILE is ignored with --config. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/user-guide/profiles.md | 2 +- src/cli_config.ts | 8 ++++++-- src/cli_profile.ts | 2 +- src/tools/builtin/disk_usage.ts | 3 ++- src/tools/builtin/list_dir.ts | 21 +++++++++++++++++---- src/tools/guard.ts | 10 ++++++++++ test/cli_config.test.ts | 10 ++++++++-- test/tools.test.ts | 8 ++++++++ 8 files changed, 53 insertions(+), 11 deletions(-) diff --git a/docs/user-guide/profiles.md b/docs/user-guide/profiles.md index 1a28ef8..f769574 100644 --- a/docs/user-guide/profiles.md +++ b/docs/user-guide/profiles.md @@ -26,7 +26,7 @@ A profile only needs the keys that differ from your global config. For example, { "agent_name": "coder", "providers": [{ "kind": "ollama", "name": "main", "model": "qwen3:8b" }] } ``` -`--config ` still replaces every layer and cannot be combined with `--profile`. +`--config ` still replaces every layer. It cannot be combined with `--profile`, and `LICH_PROFILE` is ignored when it is given. ## The soul file diff --git a/src/cli_config.ts b/src/cli_config.ts index 75a0e36..96adcfb 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -107,7 +107,7 @@ function profile_layer(name: string): { layer: Record; sources: const has_json = existsSync(paths.json); const has_soul = existsSync(paths.soul); if (has_json === false && has_soul === false) { - throw new Error(`profile not found: ${name} (expected ${paths.json})`); + throw new Error(`profile not found: ${name}`); } const layer = has_json === true ? global_layer(read_config_object(paths.json), profiles_dir()) : {}; delete layer["profile"]; @@ -149,7 +149,11 @@ export function load_layered_config(work_dir: string, profile_flag?: string): La ...(profile?.sources ?? []), ...(project === undefined ? [] : [project_path]), ]; - const config = merge_config_layers(merge_config_layers(base ?? {}, profile?.layer ?? {}), project ?? {}); + // With work_dir = home the project file is the global file, so the profile goes over it. + const config = + project_path === global_path + ? merge_config_layers(project ?? {}, profile?.layer ?? {}) + : merge_config_layers(merge_config_layers(base ?? {}, profile?.layer ?? {}), project ?? {}); delete config["profile"]; return { config, sources, notes, ...(selected === undefined ? {} : { profile: selected }) }; } diff --git a/src/cli_profile.ts b/src/cli_profile.ts index 8cd6912..64890dd 100644 --- a/src/cli_profile.ts +++ b/src/cli_profile.ts @@ -78,7 +78,7 @@ function show_profile(name: string): void { async function create_profile(name: string): Promise { const paths = profile_paths(name); if (existsSync(paths.json) === true) { - throw new Error(`profile ${name} already exists at ${paths.json}`); + throw new Error(`profile ${name} already exists`); } if (process.stdin.isTTY !== true) { throw new Error("profile create needs a TTY; write the JSON by hand instead"); diff --git a/src/tools/builtin/disk_usage.ts b/src/tools/builtin/disk_usage.ts index 45c5869..751d221 100644 --- a/src/tools/builtin/disk_usage.ts +++ b/src/tools/builtin/disk_usage.ts @@ -3,7 +3,7 @@ import { readdir } from "node:fs/promises"; import type { Dirent } from "node:fs"; import path from "node:path"; import type { JsonSchemaObject } from "../../util/json_schema.js"; -import { capture_errors, optional_number_arg, optional_string_arg, resolve_safe_path } from "../guard.js"; +import { assert_file_tool_access, capture_errors, optional_number_arg, optional_string_arg, resolve_safe_path } from "../guard.js"; import type { Tool, ToolContext } from "../types.js"; import { clamp_int_arg } from "./fetch_url.js"; @@ -68,6 +68,7 @@ async function run_disk_usage(args: Record, work_dir: string): const target = optional_string_arg(args, "path", "."); const max_entries = clamp_int_arg(args, "max_entries", DEFAULT_MAX_ENTRIES, MAX_MAX_ENTRIES); const root = resolve_safe_path(work_dir, target); + assert_file_tool_access(work_dir, root, "read"); const usage = await measure_entries(root); if (usage === null) { throw new Error("du_unavailable"); diff --git a/src/tools/builtin/list_dir.ts b/src/tools/builtin/list_dir.ts index f581c7b..d984917 100644 --- a/src/tools/builtin/list_dir.ts +++ b/src/tools/builtin/list_dir.ts @@ -2,7 +2,14 @@ import { readdir, stat } from "node:fs/promises"; import type { Dirent } from "node:fs"; import path from "node:path"; import type { JsonSchemaObject } from "../../util/json_schema.js"; -import { capture_errors, optional_number_arg, optional_string_arg, resolve_safe_path } from "../guard.js"; +import { + assert_file_tool_access, + capture_errors, + file_tool_denied, + optional_number_arg, + optional_string_arg, + resolve_safe_path, +} from "../guard.js"; import type { Tool } from "../types.js"; const MAX_ENTRIES = 500; @@ -53,12 +60,17 @@ async function push_entries( queue: Array<{ dir: string; remaining: number }>, entries: Dirent[], current: { dir: string; remaining: number }, + work_dir: string, ): Promise { for (const entry of sort_entries(entries)) { if (lines.length >= MAX_ENTRIES) { return true; } const full = path.join(current.dir, entry.name); + // Never list what file tools may not read (.lich/config.json, .lich/profiles). + if (file_tool_denied(work_dir, full, "read") === true) { + continue; + } if (entry.isDirectory() === true) { lines.push(`d ${entry.name}/`); if (current.remaining > 1) { @@ -71,7 +83,7 @@ async function push_entries( return false; } -async function collect_lines(root: string, max_depth: number): Promise { +async function collect_lines(root: string, max_depth: number, work_dir: string): Promise { const lines: string[] = []; const queue: Array<{ dir: string; remaining: number }> = [{ dir: root, remaining: max_depth }]; let truncated = false; @@ -84,7 +96,7 @@ async function collect_lines(root: string, max_depth: number): Promise if (entries === undefined) { continue; } - truncated = await push_entries(lines, queue, entries, current); + truncated = await push_entries(lines, queue, entries, current, work_dir); } if (truncated === true) { lines.push(`(... truncated at ${MAX_ENTRIES} entries)`); @@ -105,7 +117,8 @@ export const list_dir_tool: Tool = { const target = optional_string_arg(args, "path", "."); const depth = clamp_depth(optional_number_arg(args, "depth", 1)); const root = resolve_safe_path(context.work_dir, target); - const lines = await collect_lines(root, depth); + assert_file_tool_access(context.work_dir, root, "read"); + const lines = await collect_lines(root, depth, context.work_dir); return { ok: true, output: lines.join("\n") }; }), }; \ No newline at end of file diff --git a/src/tools/guard.ts b/src/tools/guard.ts index f237b3b..f7ddb43 100644 --- a/src/tools/guard.ts +++ b/src/tools/guard.ts @@ -96,6 +96,16 @@ export function assert_file_tool_access(work_dir: string, resolved: string, mode } } +/** True when `assert_file_tool_access` would refuse `resolved`. */ +export function file_tool_denied(work_dir: string, resolved: string, mode: "read" | "write"): boolean { + try { + assert_file_tool_access(work_dir, resolved, mode); + return false; + } catch { + return true; + } +} + /** Read a required non-empty string argument, or throw `missing_arg`. */ export function require_string_arg(args: Record, key: string): string { const value = args[key]; diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 4e26b17..71d237c 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -352,9 +352,15 @@ describe("load_layered_config profiles", () => { expect(load_layered_config(work, "a")?.config).toEqual({ agent_name: "a" }); }); - it("throws for a missing profile or a name that is not a plain slug", () => { + it("puts the profile over ~/.lich/config.json when the work_dir is home", () => { + write_file(global_config_path(), JSON.stringify({ agent_name: "home", max_turns: 9 })); + write_file(path.join(profiles(), "coder.json"), JSON.stringify({ agent_name: "coder" })); + expect(load_layered_config(home, "coder")?.config).toEqual({ agent_name: "coder", max_turns: 9 }); + }); + + it("throws for a missing profile or a name that is not a plain slug, without absolute paths", () => { const work = make_temp_dir("work"); - expect(() => load_layered_config(work, "ghost")).toThrow("profile not found: ghost"); + expect(() => load_layered_config(work, "ghost")).toThrow(/^profile not found: ghost$/); expect(() => load_layered_config(work, "../x")).toThrow("invalid profile name"); }); }); diff --git a/test/tools.test.ts b/test/tools.test.ts index 11cbe25..1bc0df5 100644 --- a/test/tools.test.ts +++ b/test/tools.test.ts @@ -464,6 +464,14 @@ describe("forbidden file paths", () => { expect(read.error?.startsWith("forbidden_path: .lich/profiles")).toBe(true); const write = await executor.execute("write_file", { path: ".lich/profiles/coder.json", content: "{}\n" }); expect(write.error?.startsWith("forbidden_path")).toBe(true); + const listed = await executor.execute("list_dir", { path: ".lich/profiles" }); + expect(listed.error?.startsWith("forbidden_path: .lich/profiles")).toBe(true); + const walked = await executor.execute("list_dir", { path: ".lich", depth: 3 }); + expect(walked.ok).toBe(true); + expect(walked.output).not.toContain("profiles"); + expect(walked.output).not.toContain("coder.md"); + const usage = await executor.execute("disk_usage", { path: ".lich/profiles" }); + expect(usage.error?.startsWith("forbidden_path: .lich/profiles")).toBe(true); }); it("allows writes under .lich/skills and .lich/plugins; denies other .lich writes and .env*", async () => { From 2d64bbf3e145d031ceb7741d936699546fe5b699 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 02:37:26 +0000 Subject: [PATCH 36/43] fix(profiles): keep the legacy base at home; disk_usage skips denied children The home-dir branch now applies only when ~/.lich/config.json exists, so a legacy ~/.config/lich file stays the base. disk_usage leaves out entries file tools may not read (.lich/profiles, .lich/config.json). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- src/cli_config.ts | 4 ++-- src/tools/builtin/disk_usage.ts | 24 +++++++++++++++++++----- test/cli_config.test.ts | 5 ++++- test/tools.test.ts | 5 +++++ 4 files changed, 30 insertions(+), 8 deletions(-) diff --git a/src/cli_config.ts b/src/cli_config.ts index 96adcfb..a22bfb5 100644 --- a/src/cli_config.ts +++ b/src/cli_config.ts @@ -151,8 +151,8 @@ export function load_layered_config(work_dir: string, profile_flag?: string): La ]; // With work_dir = home the project file is the global file, so the profile goes over it. const config = - project_path === global_path - ? merge_config_layers(project ?? {}, profile?.layer ?? {}) + project_path === global_path && project !== undefined + ? merge_config_layers(project, profile?.layer ?? {}) : merge_config_layers(merge_config_layers(base ?? {}, profile?.layer ?? {}), project ?? {}); delete config["profile"]; return { config, sources, notes, ...(selected === undefined ? {} : { profile: selected }) }; diff --git a/src/tools/builtin/disk_usage.ts b/src/tools/builtin/disk_usage.ts index 751d221..48b90e7 100644 --- a/src/tools/builtin/disk_usage.ts +++ b/src/tools/builtin/disk_usage.ts @@ -3,7 +3,14 @@ import { readdir } from "node:fs/promises"; import type { Dirent } from "node:fs"; import path from "node:path"; import type { JsonSchemaObject } from "../../util/json_schema.js"; -import { assert_file_tool_access, capture_errors, optional_number_arg, optional_string_arg, resolve_safe_path } from "../guard.js"; +import { + assert_file_tool_access, + capture_errors, + file_tool_denied, + optional_number_arg, + optional_string_arg, + resolve_safe_path, +} from "../guard.js"; import type { Tool, ToolContext } from "../types.js"; import { clamp_int_arg } from "./fetch_url.js"; @@ -38,12 +45,19 @@ function run_du(entry_path: string): Promise { }); } -/** Measure every depth-1 entry iteratively; null signals du itself is missing. */ -async function measure_entries(root: string): Promise { +/** + * Measure every depth-1 entry iteratively, skipping what file tools may not read + * (.lich/config.json, .lich/profiles); null signals du itself is missing. + */ +async function measure_entries(root: string, work_dir: string): Promise { const entries = await readdir(root, { withFileTypes: true }); const measured: DirEntry[] = []; for (const entry of entries as Dirent[]) { - const bytes = await run_du(path.join(root, entry.name)); + const full = path.join(root, entry.name); + if (file_tool_denied(work_dir, full, "read") === true) { + continue; + } + const bytes = await run_du(full); if (bytes < 0) { return null; } @@ -69,7 +83,7 @@ async function run_disk_usage(args: Record, work_dir: string): const max_entries = clamp_int_arg(args, "max_entries", DEFAULT_MAX_ENTRIES, MAX_MAX_ENTRIES); const root = resolve_safe_path(work_dir, target); assert_file_tool_access(work_dir, root, "read"); - const usage = await measure_entries(root); + const usage = await measure_entries(root, work_dir); if (usage === null) { throw new Error("du_unavailable"); } diff --git a/test/cli_config.test.ts b/test/cli_config.test.ts index 71d237c..6b7844e 100644 --- a/test/cli_config.test.ts +++ b/test/cli_config.test.ts @@ -353,8 +353,11 @@ describe("load_layered_config profiles", () => { }); it("puts the profile over ~/.lich/config.json when the work_dir is home", () => { - write_file(global_config_path(), JSON.stringify({ agent_name: "home", max_turns: 9 })); write_file(path.join(profiles(), "coder.json"), JSON.stringify({ agent_name: "coder" })); + // Without ~/.lich/config.json the legacy file is still the base. + write_file(path.join(home, ".config", "lich", "config.json"), JSON.stringify({ agent_name: "legacy", theme: "lich" })); + expect(load_layered_config(home, "coder")?.config).toEqual({ agent_name: "coder", theme: "lich" }); + write_file(global_config_path(), JSON.stringify({ agent_name: "home", max_turns: 9 })); expect(load_layered_config(home, "coder")?.config).toEqual({ agent_name: "coder", max_turns: 9 }); }); diff --git a/test/tools.test.ts b/test/tools.test.ts index 1bc0df5..31fbee2 100644 --- a/test/tools.test.ts +++ b/test/tools.test.ts @@ -472,6 +472,11 @@ describe("forbidden file paths", () => { expect(walked.output).not.toContain("coder.md"); const usage = await executor.execute("disk_usage", { path: ".lich/profiles" }); expect(usage.error?.startsWith("forbidden_path: .lich/profiles")).toBe(true); + const parent = await executor.execute("disk_usage", { path: ".lich" }); + if (parent.error !== "du_unavailable") { + expect(parent.ok).toBe(true); + expect(parent.output).not.toContain("profiles"); + } }); it("allows writes under .lich/skills and .lich/plugins; denies other .lich writes and .env*", async () => { From 49574528833a11b738bcc19b87c568782cbcdb1a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 02:42:26 +0000 Subject: [PATCH 37/43] wiki: config layers page after #170-#172 New entities/lich-config.md: global < profile < project merge, writers, file-tool guard, and config profile vs runtime Profile. Guardrails page and runtime-profile-session updated; index providers line fixed (#165). SCHEMA gains the `config` tag. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/SCHEMA.md | 2 +- wiki/concepts/runtime-profile-session.md | 4 +- wiki/entities/lich-config.md | 55 ++++++++++++++++++++++ wiki/entities/lich-tools-and-guardrails.md | 4 +- wiki/index.md | 5 +- wiki/log.md | 2 + 6 files changed, 66 insertions(+), 6 deletions(-) create mode 100644 wiki/entities/lich-config.md diff --git a/wiki/SCHEMA.md b/wiki/SCHEMA.md index 8119a41..2ec2241 100644 --- a/wiki/SCHEMA.md +++ b/wiki/SCHEMA.md @@ -81,7 +81,7 @@ Raw files carry `source_url`, `ingested` and `sha256`. The hash covers the body Add a tag here before you use it. - **Layers:** core, runtime, surface, gateway, serve, ossuary, cli, tui -- **Subsystems:** providers, tools, plugins, mcp, sessions, events, context, memory, skills, security +- **Subsystems:** providers, tools, plugins, mcp, sessions, events, context, memory, skills, security, config - **Games:** games, npc, play, editor, engines, content - **Comparisons:** hermes, ecosystem - **Meta:** decision, roadmap, process, docs, performance, research diff --git a/wiki/concepts/runtime-profile-session.md b/wiki/concepts/runtime-profile-session.md index 2dc6ad1..1503c58 100644 --- a/wiki/concepts/runtime-profile-session.md +++ b/wiki/concepts/runtime-profile-session.md @@ -1,7 +1,7 @@ --- title: Runtime / Profile / Session split created: 2026-09-23 -updated: 2026-09-23 +updated: 2026-10-05 type: concept tags: [core, runtime, npc, gateway] sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, "#113"] @@ -30,4 +30,6 @@ confidence: medium **Prerequisites:** remove process-global state (docs root, log level, Ollama id counter), and add the [[event-envelope]]. +**Not the same as config profiles:** #117 shipped *config* profiles, named file layers under `~/.lich/profiles` chosen per CLI run ([[lich-config]]). The runtime Profile here is a per-session object inside one process and is not built yet. + Related: [[npc-memory-namespaces]], [[embedded-safety-profile]]. diff --git a/wiki/entities/lich-config.md b/wiki/entities/lich-config.md new file mode 100644 index 0000000..ecccbca --- /dev/null +++ b/wiki/entities/lich-config.md @@ -0,0 +1,55 @@ +--- +title: Lich config layers (global, profile, project) +created: 2026-10-05 +updated: 2026-10-05 +type: entity +tags: [config, cli, security] +sources: ["#117", "#170", "#171", "#172"] +confidence: high +--- + +# Lich config layers + +How the CLI turns files into one config before `parse_agent_config`. The user docs are `docs/user-guide/cli.md` (Global config) and `docs/user-guide/profiles.md`. This page records the design and the reasons for it. + +## Layers + +Without `--config`, `load_layered_config` (`src/cli_config.ts:127@ab0f508`) merges, base first: + +1. **Global:** `~/.lich/config.json` (`src/cli_config.ts:44@ab0f508`). The legacy `~/.config/lich/config.json` is read only when the global file is absent, with a one-line note; nothing writes it. +2. **Profile:** `~/.lich/profiles/.json` (`src/cli_config.ts:89@ab0f508`), plus `.md` as `system_prompt` when it is not blank (`src/cli_config.ts:105@ab0f508`). +3. **Project:** `/.lich/config.json`. + +`--config ` replaces every layer, and combining it with `--profile` is an error. + +**Merge** (`merge_config_layers`, `src/cli_config.ts:175@ab0f508`): shallow, later layer wins per key. A later `providers` array replaces the earlier one and drops the earlier `models` unless the later layer sets its own, because role chains name providers. + +**Global and profile layers** (`global_layer`, `src/cli_config.ts:190@ab0f508`): `work_dir` and `session_dir` are ignored. Relative `plugins` paths resolve against the file's own directory (`~/.lich` or `~/.lich/profiles`), not the project. + +**Profile selection:** `--profile`, then `LICH_PROFILE`, then a `profile` key in the project file, then one in the global file (set by `lich profile use`, `src/cli_profile.ts:107@ab0f508`). The `profile` key is removed from the merged config. `LICH_PROFILE` is ignored when `--config` is given. + +**Work_dir = home:** the project file *is* `~/.lich/config.json`. It is not merged over itself, and a selected profile goes over it rather than under it. + +## Writers + +- `write_lich_config` (`src/cli_config.ts:277@ab0f508`) is the only config writer. Create mode uses an exclusive open; update mode writes a temp file and renames it into place, keeping the file mode (`src/cli_config.ts:307@ab0f508`, #166). +- `lich init` writes the project file; `lich init --global` writes the global one, without `work_dir`/`session_dir` (`src/cli.ts:694@ab0f508`). +- The setup wizard runs only when no file exists anywhere in the chain. Its last question can save the answers globally; discovered project plugins then stay in the project file, or are stored absolute when the project file is the global file (`src/cli.ts:615@ab0f508`). +- `lich profile create` writes a profile through the wizard and never overwrites. `lich mcp` edits only the project file. + +## Guard + +File tools refuse `.lich/config.json` and `.lich/profiles/` for reads and writes (`src/tools/guard.ts:83@ab0f508`). `list_dir` and `disk_usage` check their root and skip denied entries (`file_tool_denied`, `src/tools/guard.ts:100@ab0f508`). With `work_dir` = home these paths are the global identity files. + +## Two kinds of "profile" + +- **Config profile (#117, shipped):** a named file layer chosen per CLI run. It holds any config keys, including identity and model. +- **Runtime Profile (#113 §2b, not built):** cheap, data-only, chosen per session inside one Runtime ([[runtime-profile-session]]). + +A config profile could later seed a runtime Profile, but today they are separate things. Name new code accordingly. + +## Errors + +`tui`, `chat`, `serve` and `gateway` give the "no model configured" hint only for that error. Others pass through as `lich : ` (`src/cli.ts:150@ab0f508`). Messages carry no absolute paths (`.cursor/review-rules.md`). + +Related: [[lich-tools-and-guardrails]], [[hermes-agent]] (its profiles are separate home directories; Lich layers files instead). diff --git a/wiki/entities/lich-tools-and-guardrails.md b/wiki/entities/lich-tools-and-guardrails.md index 4d11620..d5773ad 100644 --- a/wiki/entities/lich-tools-and-guardrails.md +++ b/wiki/entities/lich-tools-and-guardrails.md @@ -1,7 +1,7 @@ --- title: Lich tools, executor and guardrails created: 2026-09-23 -updated: 2026-10-04 +updated: 2026-10-05 type: entity tags: [tools, security, runtime] sources: [raw/audits/2026-09-23-core-engine-audit.md, raw/audits/2026-09-23-game-surface-audit.md, "#161", "#163"] @@ -32,7 +32,7 @@ It also: ## Guardrails (the "wards") -- **File tools:** realpath confinement to `work_dir`, and writes to `.lich/config.json` are denied. +- **File tools:** realpath confinement to `work_dir`. `.lich/config.json` and `.lich/profiles/` are denied for reads and writes, and `list_dir`/`disk_usage` skip them (`src/tools/guard.ts:83@ab0f508`, #172; see [[lich-config]]). - **Network tools:** an SSRF guard blocks private and loopback URLs unless `LICH_ALLOW_PRIVATE_URLS=1`, and redirects are re-checked. - **`terminal` is not sandboxed.** It runs `bash -lc` in `work_dir` with secret env vars scrubbed. The docs say this plainly. diff --git a/wiki/index.md b/wiki/index.md index 48d6dd9..d39e51e 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -1,7 +1,7 @@ --- title: Wiki index type: index -updated: 2026-10-04 +updated: 2026-10-05 --- # Lich Wiki: Index @@ -15,9 +15,10 @@ Start here. Read [SCHEMA.md](SCHEMA.md) for the conventions and [log.md](log.md) ## Entities: Lich subsystems - [[lich-agent-loop]]: `run_conversation` + `Agent`. A dependency-injected TAO loop. Its P0 gaps are that events aren't scoped to a run, there is no streaming, it has global state, and each agent is heavyweight. -- [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover and per-role chains (`models.chat` / `models.compress`, #154). They have no streaming, `tool_choice` or cache_control, and ~150 lines of their helpers are duplicated. +- [[lich-providers]]: openai_compat, anthropic and ollama clients without SDKs, plus failover and per-role chains (`models.chat` / `models.compress`, #154). They have no streaming, `tool_choice` or cache_control; their shared HTTP/error helpers live in `providers/http.ts` (#165). - [[lich-tools-and-guardrails]]: builtins, an executor that never throws, and the wards. Known holes: `terminal` isn't sandboxed and hooks have no timeout; `tools_enabled` covers plugin tools since #161. - [[lich-plugins-and-hooks]]: tool-call hooks with veto, the gatekeeper's single gated `git_commit`, and (#157) per-plugin settings, granted model roles and a `before_llm_call` note hook. There is no `build_system_prompt` hook yet; a throwing `before_tool_call` blocks the call (#161), other hooks fail open. +- [[lich-config]]: global `~/.lich/config.json` < named profile < project file, shallow merge; `lich init --global`, `lich profile`, and file tools barred from `.lich/profiles` (#170–#172). - [[lich-sessions]]: JSONL phylacteries used as combat logs. There is no search, and gateway files are supersets of each other. - [[lich-mcp]]: an MCP client and catalog. Redot is a real entry and Godot has none. The code is spread over 21 micro-files. - [[lich-gateway]]: familiars routed into one shared Agent. The per-chat bus and read-only defaults make it a good hub. diff --git a/wiki/log.md b/wiki/log.md index 879b2b8..7a78502 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -40,3 +40,5 @@ Append-only. One line per action: `- YYYY-MM-DD | | `. Ops - 2026-10-04 update | #161 merged: tools_enabled filters plugin tools and git_commit, throwing before_tool_call blocks, last-turn tool calls not run; holes lists and index updated, code pinned at 029b7e8 | entities/lich-plugins-and-hooks.md, entities/lich-tools-and-guardrails.md, entities/lich-agent-loop.md, index.md - 2026-10-04 update | #163 merged: S-11 items (http_request status, .. guard, terminal_timeout_ms) moved out of open holes, pinned at 7001e31 | entities/lich-tools-and-guardrails.md - 2026-10-04 update | Roadmap map lists #113 children and related issues (#133 P0, #134 P1, #117 P2, #144, #148, #149); supersedes stale PR #135 | entities/roadmap-issues.md +- 2026-10-05 create | Config layers page after #170–#172: global < profile < project merge, writers, guard, and config profile vs runtime Profile; pinned at ab0f508 | entities/lich-config.md, index.md, SCHEMA.md (tag: config) +- 2026-10-05 update | Guardrails: file tools deny .lich/profiles (reads too) and list_dir/disk_usage skip it; runtime-profile-session notes the config-profile name clash; index providers line no longer claims duplicated helpers (#165) | entities/lich-tools-and-guardrails.md, concepts/runtime-profile-session.md, index.md From 0d9590e17d64698a45ba9220e40e4e144593f3b9 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 03:29:39 +0000 Subject: [PATCH 38/43] wiki: scope config writer and error-path claims on lich-config Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- wiki/entities/lich-config.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/wiki/entities/lich-config.md b/wiki/entities/lich-config.md index ecccbca..5969ab9 100644 --- a/wiki/entities/lich-config.md +++ b/wiki/entities/lich-config.md @@ -32,7 +32,7 @@ Without `--config`, `load_layered_config` (`src/cli_config.ts:127@ab0f508`) merg ## Writers -- `write_lich_config` (`src/cli_config.ts:277@ab0f508`) is the only config writer. Create mode uses an exclusive open; update mode writes a temp file and renames it into place, keeping the file mode (`src/cli_config.ts:307@ab0f508`, #166). +- `write_lich_config` (`src/cli_config.ts:277@ab0f508`) is the only writer of `.lich/config.json` files (project and global); profile JSON is written by `lich profile create` (`src/cli_profile.ts:101@ab0f508`). Create mode uses an exclusive open; update mode writes a temp file and renames it into place, keeping the file mode (`src/cli_config.ts:307@ab0f508`, #166). - `lich init` writes the project file; `lich init --global` writes the global one, without `work_dir`/`session_dir` (`src/cli.ts:694@ab0f508`). - The setup wizard runs only when no file exists anywhere in the chain. Its last question can save the answers globally; discovered project plugins then stay in the project file, or are stored absolute when the project file is the global file (`src/cli.ts:615@ab0f508`). - `lich profile create` writes a profile through the wizard and never overwrites. `lich mcp` edits only the project file. @@ -50,6 +50,6 @@ A config profile could later seed a runtime Profile, but today they are separate ## Errors -`tui`, `chat`, `serve` and `gateway` give the "no model configured" hint only for that error. Others pass through as `lich : ` (`src/cli.ts:150@ab0f508`). Messages carry no absolute paths (`.cursor/review-rules.md`). +`tui`, `chat`, `serve` and `gateway` give the "no model configured" hint only for that error. Others pass through as `lich : ` (`src/cli.ts:150@ab0f508`). The profile errors (`profile not found`, `profile already exists`) carry no absolute paths (`.cursor/review-rules.md`). `config not found` and `invalid config json` still name the file. Related: [[lich-tools-and-guardrails]], [[hermes-agent]] (its profiles are separate home directories; Lich layers files instead). From b95516b5d10a42f9f32ad50755b867c9ceb3e58f Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 03:29:01 +0000 Subject: [PATCH 39/43] fix(gateway): one session queue, LRU eviction, telegram-only /start, awaited shutdown, webhook usage/502 (#144) Slice 1 of #144 (A1 plus the G-10 and webhook fixes): - GatewayBus queues conversations on SessionManager instead of its own promise chains (one queue implementation). - max_conversations evicts the least recently used conversation. - Only telegram's /start (/start, /start@bot, /start ) becomes "hello"; /started and other platforms pass through. - Shutdown awaits adapter stop (bounded by 5 s) before exit. - bus.reply() returns { text, usage, failed }; the webhook returns the run's usage and 502 {"error"} on agent failure. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 9 ++ docs/getting-started.md | 2 +- docs/user-guide/gateway.md | 13 ++- docs/user-guide/godot.md | 11 +-- examples/persona_orchestrator/README.md | 2 +- src/gateway/bus.ts | 70 +++++++-------- src/gateway/runner.ts | 39 ++++++-- src/gateway/types.ts | 22 ++++- src/gateway/webhook.ts | 10 ++- test/gateway.test.ts | 108 ++++++++++++++++++++++- test/gateway_conversation_bounds.test.ts | 5 +- test/gateway_shutdown.test.ts | 51 +++++++++++ 12 files changed, 271 insertions(+), 71 deletions(-) create mode 100644 test/gateway_shutdown.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index dd66972..602c43f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,15 @@ ## Unreleased +- Gateway (#144): conversations queue on the shared `SessionManager` instead of + their own promise chains. Beyond `max_conversations`, the least recently used + conversation is evicted (it was the oldest inserted). Only Telegram's `/start` + command (`/start`, `/start@bot`, `/start `) is rewritten to "hello"; + `/started ...` and other platforms pass through. Shutdown waits for adapters + to stop (up to 5 s) before exiting. +- **Breaking (webhook):** `POST /message` returns the run's token `usage` + instead of `null`, and an agent failure is `502 {"error":"agent error: ..."}` + instead of `200` with the error in `reply`. - Profiles (#117): `~/.lich/profiles/.json` (plus an optional `.md` used as the system prompt) is merged between the global and project config. Select one with `--profile`, `LICH_PROFILE`, a project `profile` key, or the diff --git a/docs/getting-started.md b/docs/getting-started.md index 671e64e..3c28953 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -94,7 +94,7 @@ curl -s -X POST http://localhost:8089/message \ Observed response shape (the `reply` text is the model's answer; `usage` is `null` on this endpoint): ```json -{"reply":"Hello! How can I help you today? I can assist with coding, file management, running commands, web searches, and more — just let me know what you'd like to do.","usage":null} +{"reply":"Hello! How can I help you today? I can assist with coding, file management, running commands, web searches, and more — just let me know what you'd like to do.","usage":{"prompt_tokens":812,"completion_tokens":41,"total_tokens":853}} ``` Stop the gateway with Ctrl+C (SIGINT) or `kill` (SIGTERM); both shut down adapters cleanly. diff --git a/docs/user-guide/gateway.md b/docs/user-guide/gateway.md index 94f5829..3026d46 100644 --- a/docs/user-guide/gateway.md +++ b/docs/user-guide/gateway.md @@ -4,7 +4,7 @@ ## How it works -`lich gateway ` runs a long-lived process that forwards inbound chat messages to **one shared agent** and routes replies back. Per-conversation memory is keyed `platform:chat_id` (Telegram/Discord chat ids, Twitch channel names, webhook `chat_id` field): each conversation keeps its own bounded history capped at 40 messages (oldest evicted; conversations beyond 200 are evicted oldest-first). Messages for the same conversation are serialized, so overlapping messages never interleave histories; different conversations can run concurrently. Failures become a safe one-line reply: `agent error: `. +`lich gateway ` runs a long-lived process that forwards inbound chat messages to **one shared agent** and routes replies back. Per-conversation memory is keyed `platform:chat_id` (Telegram/Discord chat ids, Twitch channel names, webhook `chat_id` field): each conversation keeps its own bounded history capped at 40 messages (oldest evicted; beyond 200 conversations, the least recently used one is evicted). Messages for the same conversation are serialized, so overlapping messages never interleave histories; different conversations can run concurrently. Failures become a safe one-line reply: `agent error: `. ```mermaid flowchart LR @@ -25,7 +25,7 @@ lich gateway telegram discord twitch # no webhook server lich gateway # defaults to webhook ``` -The gateway is silent after startup: Telegram/Discord/Twitch respond only in chats, channels, or servers the bot can see or has joined, and the webhook only serves HTTP. Telegram media messages arrive as the placeholder text `media not supported yet`; other non-text events are ignored. Telegram `/start` is answered like a plain "hello". +The gateway is silent after startup: Telegram/Discord/Twitch respond only in chats, channels, or servers the bot can see or has joined, and the webhook only serves HTTP. Telegram media messages arrive as the placeholder text `media not supported yet`; other non-text events are ignored. Telegram `/start` (also `/start@bot` and `/start `) is answered like a plain "hello"; other text, and other platforms, are passed through as written. **Security defaults:** the webhook binds loopback only; Telegram/Discord/Twitch default-deny until you configure allowlists; the gateway agent uses a read-only tool subset (no `terminal`, no file writes) unless you override `gateway.tools_enabled`. @@ -40,7 +40,7 @@ lich gateway webhook ```sh curl -s -X POST http://127.0.0.1:8089/message \ -H "content-type: application/json" -d '{"text": "hello"}' -# -> {"reply":"...","usage":null} +# -> {"reply":"...","usage":{"prompt_tokens":...,"completion_tokens":...,"total_tokens":...}} curl -s http://127.0.0.1:8089/health # -> {"status":"ok"} @@ -203,10 +203,10 @@ Request: Only `text` is required. Any `platform` field in the body is ignored; the conversation key is always `webhook:`. Success (`200`): ```json -{"reply":"Hello! How can I help you today? ...","usage":null} +{"reply":"Hello! How can I help you today? ...","usage":{"prompt_tokens":812,"completion_tokens":24,"total_tokens":836}} ``` -`reply` is the agent's final answer; `usage` is always `null` on this endpoint (webhook replies are formatted without usage stats, unlike the chat/TUI footers). Errors: +`reply` is the agent's final answer; `usage` is the run's token totals. Errors: | Status | When | | --- | --- | @@ -214,8 +214,7 @@ Only `text` is required. Any `platform` field in the body is ignored; the conver | `401` | `LICH_GATEWAY_TOKEN` is set and the `x-lich-token` header does not match. | | `404` | Anything other than `POST /message` or `GET /health`. | | `500` | Internal dispatch failure: `{"error":"internal error"}`. | - -Agent-level failures (e.g. every provider failed) return `200` with `reply` set to a sanitized one-line `agent error: ...` string, so callers always get a deliverable text. +| `502` | The agent run failed (e.g. every provider failed): `{"error":"agent error: ..."}`, a sanitized one-line message. | ### `GET /health` diff --git a/docs/user-guide/godot.md b/docs/user-guide/godot.md index 2e4fcc7..be54ee7 100644 --- a/docs/user-guide/godot.md +++ b/docs/user-guide/godot.md @@ -44,10 +44,10 @@ curl -s -X POST http://127.0.0.1:8089/message \ -d '{"text":"round 1: hero1 at full. goblin is the only living enemy.","chat_id":"run-1"}' ``` -Success is exactly one JSON object. This endpoint always sends `usage: null` — it does not forward provider token counts: +Success is exactly one JSON object, with the run's token totals in `usage`: ```json -{"reply":"...","usage":null} +{"reply":"...","usage":{"prompt_tokens":812,"completion_tokens":24,"total_tokens":836}} ``` | Status | Body | @@ -56,10 +56,11 @@ Success is exactly one JSON object. This endpoint always sends `usage: null` — | `401` | `{"error":"unauthorized"}` when `LICH_GATEWAY_TOKEN` is set and `x-lich-token` does not match | | `404` | `{"error":"not found"}` for any other method or path | | `500` | `{"error":"internal error"}` if the handler throws before a response is sent | +| `502` | `{"error":"agent error: ..."}` when the agent run fails (a sanitized one-line message) | -A failed run is still `200`. `reply` is then a sanitized `agent error: ...` line. There is no streaming, pagination, or cursor. Full platform notes: [webhook API](gateway.md#webhook-api-reference). +There is no streaming, pagination, or cursor. Full platform notes: [webhook API](gateway.md#webhook-api-reference). -Memory is keyed `platform:chat_id`, capped at 40 messages (oldest dropped) and 200 conversations (oldest dropped). For a roguelike, `chat_id` = the run id gives the commander that process's memory of the run. A new run id starts a fresh history. That history is in memory only — restarting the gateway clears it. Restate facts the digest still needs. Durable notes are a different file, below. +Memory is keyed `platform:chat_id`, capped at 40 messages (oldest dropped) and 200 conversations (least recently used dropped). For a roguelike, `chat_id` = the run id gives the commander that process's memory of the run. A new run id starts a fresh history. That history is in memory only — restarting the gateway clears it. Restate facts the digest still needs. Durable notes are a different file, below. ## Wire the example plugin @@ -157,6 +158,6 @@ Plugins run in-process with the agent's privileges (files, network, environment) ## Limits - A long run evicts gateway history past 40 messages. Put habits that still matter in the digest, or in `memory.jsonl` if they must survive a restart. -- More than 200 concurrent `chat_id`s on one process drops the oldest conversation. Fine for one developer machine; a host of many runs should know the cap. +- More than 200 concurrent `chat_id`s on one process drops the least recently used conversation. Fine for one developer machine; a host of many runs should know the cap. - Combat must finish if the gateway is down, the call times out, or `orders.jsonl` is empty or garbage. The game's fallback is the game's — this repo does not ship one. - Same-`chat_id` calls run one after another. Different `chat_id`s run concurrently. The plugin itself makes no concurrency guarantee; one bridge per `chat_id` is the intended pattern. diff --git a/examples/persona_orchestrator/README.md b/examples/persona_orchestrator/README.md index 34bc95b..b15bb41 100644 --- a/examples/persona_orchestrator/README.md +++ b/examples/persona_orchestrator/README.md @@ -49,7 +49,7 @@ curl -s -X POST http://127.0.0.1:8090/message \ -d '{"text":"round 1: hero1 at full. goblin is the only living enemy.","chat_id":"npc:commander:run-1"}' ``` -Success is `{reply, usage}`. `usage` is the run's `usage_total`. The CLI webhook still sends `usage: null`; a Godot client that only reads `reply` needs no change. Missing `text` is `400 {"error":"text is required"}`. A bad token is `401`. Wrong `Content-Type` is `415`. Oversized body is `413`. Unknown or missing `chat_id` is still `200` with `reply` starting `agent error: unknown persona` and `usage: null` — this example does not default `chat_id` to `"default"`, because a persona cannot be inferred. +Success is `{reply, usage}`. `usage` is the run's `usage_total`. The CLI webhook sends the same shape, and returns `502` on an agent failure; this example keeps `200` with an `agent error:` reply. Missing `text` is `400 {"error":"text is required"}`. A bad token is `401`. Wrong `Content-Type` is `415`. Oversized body is `413`. Unknown or missing `chat_id` is still `200` with `reply` starting `agent error: unknown persona` and `usage: null` — this example does not default `chat_id` to `"default"`, because a persona cannot be inferred. From a source checkout, with the working directory at the repo root: diff --git a/src/gateway/bus.ts b/src/gateway/bus.ts index b531fec..dad688e 100644 --- a/src/gateway/bus.ts +++ b/src/gateway/bus.ts @@ -1,17 +1,19 @@ /** * GatewayBus: conversation-keyed runner over one shared Agent. * - * Per-conversation history lives in a bounded Map; concurrent messages for - * the same conversation are serialized through a promise chain so history - * never interleaves. Agent failures become sanitized reply strings. + * Per-conversation history lives in a bounded Map evicted by least recent + * use; messages for the same conversation are serialized on the shared + * SessionManager so history never interleaves. Agent failures become + * sanitized reply strings. */ import type { Agent } from "../agent/agent.js"; import type { AgentConfig } from "../agent/config.js"; import { history_after_run_error } from "../agent/loop.js"; import type { Message } from "../providers/types.js"; +import { create_session_manager, type SessionManager } from "../session/manager.js"; import { logger } from "../util/log.js"; import { check_gateway_sender } from "./access.js"; -import { sanitize_agent_error } from "./types.js"; +import { sanitize_agent_error, type GatewayReply } from "./types.js"; export interface GatewayBusOptions { history_cap?: number; @@ -33,7 +35,7 @@ export class GatewayBus { private readonly agent_factory: () => Agent; private agent: Agent | undefined; private readonly histories: Map = new Map(); - private readonly chains: Map> = new Map(); + private readonly sessions: SessionManager = create_session_manager(); private readonly history_cap: number; private readonly max_conversations: number; private stop_logging: (() => void) | undefined; @@ -48,32 +50,19 @@ export class GatewayBus { } } - /** Serializes runs per conversation and resolves to the reply text. */ + /** Serializes runs per conversation and resolves to the reply text (undefined when empty or denied). */ async handle(platform: string, chat_id: string, user_id: string, text: string): Promise { + const reply = await this.reply(platform, chat_id, user_id, text); + return reply === undefined || reply.text.length === 0 ? undefined : reply.text; + } + + /** Like `handle`, with the run's usage and whether it failed; undefined when the sender is denied. */ + async reply(platform: string, chat_id: string, user_id: string, text: string): Promise { if (check_gateway_sender(this.config, platform, chat_id, user_id) === false) { return undefined; } const key = conversation_key(platform, chat_id); - const previous = this.chains.get(key) ?? Promise.resolve(); - const run = previous.then(() => this.run_once(key, platform, chat_id, user_id, text)); - let tracked: Promise = Promise.resolve(); - tracked = run.then( - () => { - this.release_chain(key, tracked); - }, - () => { - this.release_chain(key, tracked); - }, - ); - this.chains.set(key, tracked); - return run; - } - - /** Drops a settled chain entry unless a newer message re-queued the key. */ - private release_chain(key: string, tracked: Promise): void { - if (this.chains.get(key) === tracked) { - this.chains.delete(key); - } + return this.sessions.enqueue(key, () => this.run_once(key, platform, chat_id, user_id, text)); } /** Unsubscribes the debug tool logger (bus owns no other resources). */ @@ -88,24 +77,30 @@ export class GatewayBus { chat_id: string, user_id: string, text: string, - ): Promise { - const input = text.startsWith("/start") === true ? "hello" : text; + ): Promise { + const input = is_telegram_start(platform, text) === true ? "hello" : text; const history = this.history_for(key); const agent = this.ensure_agent(); try { const result = await agent.run({ input, history, label: `gw:${platform}:${chat_id}` }); - this.histories.set(key, cap_history(result.messages, this.history_cap)); - return final_reply_text(result.outcome.final?.content); + this.store_history(key, cap_history(result.messages, this.history_cap)); + return { text: result.outcome.final?.content ?? "", usage: result.usage_total }; } catch (error) { logger.error(`gateway bus run failed for ${key} (user ${user_id})`, error); const kept = history_after_run_error(error); if (kept !== undefined) { - this.histories.set(key, cap_history(kept, this.history_cap)); + this.store_history(key, cap_history(kept, this.history_cap)); } - return sanitize_agent_error(error); + return { text: sanitize_agent_error(error), failed: true }; } } + /** Re-inserts the key so Map order tracks the most recent use. */ + private store_history(key: string, messages: Message[]): void { + this.histories.delete(key); + this.histories.set(key, messages); + } + private ensure_agent(): Agent { if (this.agent === undefined) { this.agent = this.agent_factory(); @@ -113,7 +108,7 @@ export class GatewayBus { return this.agent; } - /** Oldest-first eviction keeps the conversation map bounded. */ + /** Least-recently-used eviction keeps the conversation map bounded. */ private history_for(key: string): Message[] { while (this.histories.size >= this.max_conversations && this.histories.has(key) === false) { const oldest = this.histories.keys().next(); @@ -141,6 +136,11 @@ function conversation_key(platform: string, chat_id: string): string { return `${platform}:${chat_id}`; } +/** Telegram's bot-start command (`/start`, `/start@bot`, `/start `); not `/started` or other platforms. */ +function is_telegram_start(platform: string, text: string): boolean { + return platform === "telegram" && /^\/start(@\w+)?(\s|$)/.test(text); +} + /** * Newest `cap` messages, starting at the first user message in the window. * When the window has no user message, drop a leading tool-result fragment so @@ -195,7 +195,3 @@ function assistant_owning_tools(messages: readonly Message[], start: number): nu const calls = parent.tool_calls; return calls !== undefined && calls.length > 0 ? index : undefined; } - -function final_reply_text(content: string | undefined): string | undefined { - return content === undefined || content.length === 0 ? undefined : content; -} \ No newline at end of file diff --git a/src/gateway/runner.ts b/src/gateway/runner.ts index e3f0e47..298abe5 100644 --- a/src/gateway/runner.ts +++ b/src/gateway/runner.ts @@ -49,7 +49,7 @@ function is_known_platform(platform: string): boolean { function build_adapters(config: AgentConfig, bus: GatewayBus, platforms: readonly string[]): PlatformAdapter[] { const params: AdapterParams = { config, - handle_message: (platform, chat_id, user_id, text) => bus.handle(platform, chat_id, user_id, text), + handle_message: (platform, chat_id, user_id, text) => bus.reply(platform, chat_id, user_id, text), get_agent: () => { throw new Error("get_agent is reserved for future use"); }, @@ -109,16 +109,37 @@ async function start_all_adapters(adapters: readonly PlatformAdapter[]): Promise } } +/** Upper bound on waiting for adapters to stop, so a hung platform socket cannot block exit. */ +const SHUTDOWN_TIMEOUT_MS = 5000; + +/** Stops every adapter and waits for them (bounded), then releases the bus and agent. */ +export async function shutdown_gateway( + agent: Agent, + bus: GatewayBus, + adapters: readonly PlatformAdapter[], + timeout_ms = SHUTDOWN_TIMEOUT_MS, +): Promise { + const stops = Promise.allSettled( + adapters.map((adapter) => + adapter.stop().catch((error: unknown) => logger.warn(`gateway adapter stop failed: ${adapter.name}`, error)), + ), + ); + let timer: ReturnType | undefined; + const timeout = new Promise((resolve) => { + timer = setTimeout(() => { + logger.warn(`gateway adapters did not stop within ${timeout_ms}ms; exiting anyway`); + resolve(); + }, timeout_ms); + }); + await Promise.race([stops, timeout]); + clearTimeout(timer); + bus.stop(); + agent.close(); +} + function install_signal_handlers(agent: Agent, bus: GatewayBus, adapters: readonly PlatformAdapter[]): void { const shutdown = (): void => { - for (const adapter of adapters) { - void adapter - .stop() - .catch((error: unknown) => logger.warn(`gateway adapter stop failed: ${adapter.name}`, error)); - } - bus.stop(); - agent.close(); - process.exit(0); + void shutdown_gateway(agent, bus, adapters).finally(() => process.exit(0)); }; process.once("SIGINT", shutdown); process.once("SIGTERM", shutdown); diff --git a/src/gateway/types.ts b/src/gateway/types.ts index 94bd150..f2b20cd 100644 --- a/src/gateway/types.ts +++ b/src/gateway/types.ts @@ -5,6 +5,7 @@ */ import type { Agent } from "../agent/agent.js"; import type { AgentConfig } from "../agent/config.js"; +import type { Usage } from "../providers/types.js"; import { logger } from "../util/log.js"; export type PlatformName = "webhook" | "telegram" | "discord" | "twitch"; @@ -29,13 +30,28 @@ export interface PlatformAdapter { stop(): Promise; } -/** Handles one inbound message and resolves to the reply text. */ +/** One run's reply: text, token usage, and whether the agent failed (text is then the sanitized error). */ +export interface GatewayReply { + text: string; + usage?: Usage; + failed?: boolean; +} + +/** Handles one inbound message and resolves to the reply (text alone, or text plus usage/failure). */ export type InboundHandler = ( platform: string, chat_id: string, user_id: string, text: string, -) => Promise; +) => Promise; + +/** Reply text for chat platforms, whichever form the handler returned. */ +export function reply_text(reply: string | GatewayReply | undefined): string { + if (reply === undefined) { + return ""; + } + return typeof reply === "string" ? reply : reply.text; +} /** Dependencies handed to every adapter factory. */ export interface AdapterParams { @@ -97,7 +113,7 @@ export async function run_inbound_message( text: string, ): Promise { try { - return (await handle(platform, chat_id, user_id, text)) ?? ""; + return reply_text(await handle(platform, chat_id, user_id, text)); } catch (error) { logger.error(`gateway ${platform} message handling failed`, error); return sanitize_agent_error(error); diff --git a/src/gateway/webhook.ts b/src/gateway/webhook.ts index e66e747..fafb775 100644 --- a/src/gateway/webhook.ts +++ b/src/gateway/webhook.ts @@ -7,7 +7,7 @@ import { createServer, type IncomingMessage, type Server, type ServerResponse } import { logger } from "../util/log.js"; import { format_agent_reply } from "./format.js"; import { read_platform_token } from "./token_env.js"; -import type { AdapterParams, PlatformAdapter } from "./types.js"; +import { reply_text, type AdapterParams, type PlatformAdapter } from "./types.js"; export const DEFAULT_GATEWAY_PORT = 8089; export const DEFAULT_GATEWAY_HOST = "127.0.0.1"; @@ -105,7 +105,13 @@ async function handle_message_post( const chat_id = payload.chat_id ?? "default"; const user_id = payload.user_id ?? "anonymous"; const reply = await params.handle_message(platform, String(chat_id), String(user_id), String(text)); - respond_json_text(response, 200, format_agent_reply(reply ?? "", undefined, "webhook")); + if (typeof reply === "object" && reply.failed === true) { + // The text is already the sanitized agent error. + send_json(response, 502, { error: reply.text }); + return; + } + const usage = typeof reply === "object" ? reply.usage : undefined; + respond_json_text(response, 200, format_agent_reply(reply_text(reply), usage, "webhook")); } function content_length_exceeds(request: IncomingMessage, max_bytes: number): boolean { diff --git a/test/gateway.test.ts b/test/gateway.test.ts index f600656..0b8aa57 100644 --- a/test/gateway.test.ts +++ b/test/gateway.test.ts @@ -21,6 +21,7 @@ import { note_partial_messages } from "../src/agent/loop.js"; import { GatewayBus } from "../src/gateway/bus.js"; import type { Agent, AgentRunResult } from "../src/agent/agent.js"; import type { Message, Usage } from "../src/providers/types.js"; +import type { GatewayReply } from "../src/gateway/types.js"; import { TMP_BASE } from "./helpers/tmp_base.js"; const usage_zero: Usage = { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }; @@ -383,7 +384,7 @@ describe("gateway bus", () => { } }); - it("releases settled promise chains so the chains map stays bounded (G-6)", async () => { + it("releases settled session queues so they stay bounded (G-6)", async () => { const work_dir = temp_work_dir(); try { const bus = new GatewayBus({ @@ -392,8 +393,8 @@ describe("gateway bus", () => { }); await bus.handle("webhook", "c1", "u1", "one"); await bus.handle("webhook", "c2", "u1", "two"); - const chains = (bus as unknown as { chains: Map> }).chains; - expect(chains.size).toBe(0); + const sessions = (bus as unknown as { sessions: { pending_count(): number } }).sessions; + expect(sessions.pending_count()).toBe(0); } finally { rmSync(work_dir, { recursive: true, force: true }); } @@ -421,6 +422,73 @@ describe("gateway bus", () => { } }); + it("evicts the least recently used conversation, not the oldest inserted (G-10)", async () => { + const work_dir = temp_work_dir(); + try { + const records: RunRecord[] = []; + const bus = new GatewayBus( + { config: config_for(work_dir), agent_factory: () => recording_agent(records) }, + { max_conversations: 2 }, + ); + await bus.handle("webhook", "a", "u1", "a1"); + await bus.handle("webhook", "b", "u1", "b1"); + await bus.handle("webhook", "a", "u1", "a2"); + await bus.handle("webhook", "c", "u1", "c1"); + await bus.handle("webhook", "a", "u1", "a3"); + await bus.handle("webhook", "b", "u1", "b2"); + const history_of = (input: string): readonly Message[] => records.find((run) => run.input === input)?.history ?? []; + // a was used after b, so c evicted b: a keeps its history and b starts over. + expect(history_of("a3").length).toBeGreaterThan(0); + expect(history_of("b2")).toEqual([]); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); + + it("rewrites only telegram's /start command to hello (G-10)", async () => { + const work_dir = temp_work_dir(); + try { + const records: RunRecord[] = []; + const bus = new GatewayBus({ + config: config_for(work_dir, { allowed_users: { telegram: ["u1"] } }), + agent_factory: () => recording_agent(records), + }); + await bus.handle("telegram", "t1", "u1", "/start"); + await bus.handle("telegram", "t2", "u1", "/start@lich_bot deep-link"); + await bus.handle("telegram", "t3", "u1", "/started a thing"); + await bus.handle("webhook", "w1", "u1", "/start"); + expect(records.map((run) => run.input)).toEqual(["hello", "hello", "/started a thing", "/start"]); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); + + it("reply() carries the run's usage, and marks agent failures", async () => { + const work_dir = temp_work_dir(); + try { + let fail = false; + const bus = new GatewayBus({ + config: config_for(work_dir), + agent_factory: () => + ({ + run: async (options: { input: string; history?: readonly Message[] }): Promise => { + if (fail === true) { + throw new Error("provider down"); + } + return { ...reply_result("ok", options.history ?? [], options.input), usage_total: usage_small }; + }, + }) as unknown as Agent, + }); + expect(await bus.reply("webhook", "c1", "u1", "hi")).toEqual({ text: "ok", usage: usage_small }); + fail = true; + const failed = await bus.reply("webhook", "c1", "u1", "again"); + expect(failed?.failed).toBe(true); + expect(failed?.text.startsWith("agent error: provider down")).toBe(true); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); + it("keeps completed tool turns when a later model call throws", async () => { const work_dir = temp_work_dir(); try { @@ -567,7 +635,7 @@ describe("webhook adapter", () => { chat_id: string, user_id: string, text: string, - ) => Promise, + ) => Promise, ) { return create_webhook_adapter({ config: config_for(work_dir), @@ -584,6 +652,38 @@ describe("webhook adapter", () => { }); } + it("returns the run's usage, and HTTP 502 with the error when the agent fails", async () => { + const work_dir = temp_work_dir(); + let port: number | undefined; + const adapter = make_adapter( + work_dir, + "unused", + (seen) => { + port = seen; + }, + async (_platform, _chat, _user, text) => + text === "fail" ? { text: "agent error: provider down", failed: true } : { text: `ok:${text}`, usage: usage_small }, + ); + try { + await adapter.start(); + const post = (text: string) => + fetch(`http://127.0.0.1:${String(port)}/message`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ text }), + }); + const ok = await post("hi"); + expect(ok.status).toBe(200); + expect(await ok.json()).toEqual({ reply: "ok:hi", usage: usage_small }); + const failed = await post("fail"); + expect(failed.status).toBe(502); + expect(await failed.json()).toEqual({ error: "agent error: provider down" }); + } finally { + await adapter.stop(); + rmSync(work_dir, { recursive: true, force: true }); + } + }); + it("serves POST /message, GET /health, and 404 for other paths", async () => { const work_dir = temp_work_dir(); let port: number | undefined; diff --git a/test/gateway_conversation_bounds.test.ts b/test/gateway_conversation_bounds.test.ts index fe69f4a..f625564 100644 --- a/test/gateway_conversation_bounds.test.ts +++ b/test/gateway_conversation_bounds.test.ts @@ -148,8 +148,9 @@ describe("gateway max_conversations", () => { const other = bus.handle("webhook", "b", "u", "b1"); const b1 = await wait_for_start(started, "b1"); - const chains = (bus as unknown as { chains: Map> }).chains; - expect(chains.has("webhook:a")).toBe(true); + // Both conversations still hold a queue: evicting a's history does not drop its run queue. + const sessions = (bus as unknown as { sessions: { pending_count(): number } }).sessions; + expect(sessions.pending_count()).toBe(2); const third = bus.handle("webhook", "a", "u", "a3"); await new Promise((resolve) => setImmediate(resolve)); diff --git a/test/gateway_shutdown.test.ts b/test/gateway_shutdown.test.ts new file mode 100644 index 0000000..2802e43 --- /dev/null +++ b/test/gateway_shutdown.test.ts @@ -0,0 +1,51 @@ +/** + * Gateway shutdown (G-10): adapters are stopped and awaited (bounded by a + * timeout) before the bus and agent are released. + */ +import { describe, expect, it, vi } from "vitest"; +import type { Agent } from "../src/agent/agent.js"; +import type { GatewayBus } from "../src/gateway/bus.js"; +import { shutdown_gateway } from "../src/gateway/runner.js"; +import type { PlatformAdapter } from "../src/gateway/types.js"; +import { logger } from "../src/util/log.js"; + +function tracked(order: string[]): { agent: Agent; bus: GatewayBus } { + return { + agent: { close: () => order.push("agent.close") } as unknown as Agent, + bus: { stop: () => order.push("bus.stop") } as unknown as GatewayBus, + }; +} + +describe("shutdown_gateway", () => { + it("waits for every adapter to stop before releasing the bus and agent", async () => { + const order: string[] = []; + const { agent, bus } = tracked(order); + const slow: PlatformAdapter = { + name: "slow", + start: async () => undefined, + stop: () => new Promise((resolve) => setTimeout(() => resolve(order.push("slow.stop") as unknown as void), 20)), + }; + const broken: PlatformAdapter = { + name: "broken", + start: async () => undefined, + stop: async () => { + throw new Error("socket gone"); + }, + }; + vi.spyOn(logger, "warn").mockImplementation(() => undefined); + await shutdown_gateway(agent, bus, [slow, broken]); + expect(order).toEqual(["slow.stop", "bus.stop", "agent.close"]); + vi.restoreAllMocks(); + }); + + it("stops waiting after the timeout when an adapter hangs", async () => { + const order: string[] = []; + const { agent, bus } = tracked(order); + const hung: PlatformAdapter = { name: "hung", start: async () => undefined, stop: () => new Promise(() => undefined) }; + const warn = vi.spyOn(logger, "warn").mockImplementation(() => undefined); + await shutdown_gateway(agent, bus, [hung], 20); + expect(order).toEqual(["bus.stop", "agent.close"]); + expect(String(warn.mock.calls[0]?.[0])).toContain("did not stop within 20ms"); + vi.restoreAllMocks(); + }); +}); From 53d81181cdb38501c03c8f8f758e555b319465af Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 05:50:35 +0000 Subject: [PATCH 40/43] fix(gateway): enforce max_conversations on store; doc webhook 502 Concurrent new chats could each pass the pre-run eviction and all store, leaving the map over the cap. store_history now trims least recently used entries after each write. Adds a keep-alive stop regression test. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- docs/user-guide/gateway.md | 2 +- src/gateway/bus.ts | 12 +++++++++- test/gateway.test.ts | 30 ++++++++++++++++++++++++ test/gateway_conversation_bounds.test.ts | 23 ++++++++++++++++++ 4 files changed, 65 insertions(+), 2 deletions(-) diff --git a/docs/user-guide/gateway.md b/docs/user-guide/gateway.md index 3026d46..458cbc5 100644 --- a/docs/user-guide/gateway.md +++ b/docs/user-guide/gateway.md @@ -4,7 +4,7 @@ ## How it works -`lich gateway ` runs a long-lived process that forwards inbound chat messages to **one shared agent** and routes replies back. Per-conversation memory is keyed `platform:chat_id` (Telegram/Discord chat ids, Twitch channel names, webhook `chat_id` field): each conversation keeps its own bounded history capped at 40 messages (oldest evicted; beyond 200 conversations, the least recently used one is evicted). Messages for the same conversation are serialized, so overlapping messages never interleave histories; different conversations can run concurrently. Failures become a safe one-line reply: `agent error: `. +`lich gateway ` runs a long-lived process that forwards inbound chat messages to **one shared agent** and routes replies back. Per-conversation memory is keyed `platform:chat_id` (Telegram/Discord chat ids, Twitch channel names, webhook `chat_id` field): each conversation keeps its own bounded history capped at 40 messages (oldest evicted; beyond 200 conversations, the least recently used one is evicted). Messages for the same conversation are serialized, so overlapping messages never interleave histories; different conversations can run concurrently. Failures become a safe one-line reply, `agent error: `; the webhook returns it as HTTP `502 {"error":"agent error: ..."}`. ```mermaid flowchart LR diff --git a/src/gateway/bus.ts b/src/gateway/bus.ts index dad688e..61e6573 100644 --- a/src/gateway/bus.ts +++ b/src/gateway/bus.ts @@ -95,10 +95,20 @@ export class GatewayBus { } } - /** Re-inserts the key so Map order tracks the most recent use. */ + /** + * Re-inserts the key so Map order tracks the most recent use, then trims to + * the cap: concurrent new chats can each pass `history_for` before any stores. + */ private store_history(key: string, messages: Message[]): void { this.histories.delete(key); this.histories.set(key, messages); + while (this.histories.size > this.max_conversations) { + const oldest = this.histories.keys().next(); + if (oldest.done === true || oldest.value === key) { + break; + } + this.histories.delete(oldest.value); + } } private ensure_agent(): Agent { diff --git a/test/gateway.test.ts b/test/gateway.test.ts index 0b8aa57..c56c261 100644 --- a/test/gateway.test.ts +++ b/test/gateway.test.ts @@ -652,6 +652,36 @@ describe("webhook adapter", () => { }); } + it("stops promptly while a keep-alive client is still connected", async () => { + const work_dir = temp_work_dir(); + let port: number | undefined; + const adapter = make_adapter(work_dir, "echo", (seen) => { + port = seen; + }); + try { + await adapter.start(); + const { request } = await import("node:http"); + const agent = new (await import("node:http")).Agent({ keepAlive: true }); + await new Promise((resolve, reject) => { + const req = request( + { host: "127.0.0.1", port, path: "/health", agent }, + (response) => { + response.resume(); + response.on("end", () => resolve()); + }, + ); + req.on("error", reject); + req.end(); + }); + const started = Date.now(); + await adapter.stop(); + expect(Date.now() - started).toBeLessThan(1000); + agent.destroy(); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); + it("returns the run's usage, and HTTP 502 with the error when the agent fails", async () => { const work_dir = temp_work_dir(); let port: number | undefined; diff --git a/test/gateway_conversation_bounds.test.ts b/test/gateway_conversation_bounds.test.ts index f625564..0ca46b1 100644 --- a/test/gateway_conversation_bounds.test.ts +++ b/test/gateway_conversation_bounds.test.ts @@ -123,6 +123,29 @@ describe("gateway max_conversations", () => { expect(c2?.history_len).toBeGreaterThan(0); }); + it("stays within max_conversations when new chats run concurrently", async () => { + const work_dir = make_temp_dir(); + const started: HeldRun[] = []; + const bus = new GatewayBus( + { + config: parse_agent_config({ + providers: [{ kind: "openai_compat", name: "main", model: "mock-model" }], + work_dir, + log_level: "error", + }), + agent_factory: () => holding_agent(started), + }, + { max_conversations: 2, history_cap: 40 }, + ); + const replies = ["a", "b", "c", "d"].map((chat) => bus.handle("webhook", chat, "u", `${chat}1`)); + for (const chat of ["a", "b", "c", "d"]) { + (await wait_for_start(started, `${chat}1`)).release(); + } + await Promise.all(replies); + const histories = (bus as unknown as { histories: Map }).histories; + expect([...histories.keys()]).toEqual(["webhook:c", "webhook:d"]); + }); + it("keeps an in-flight chain after history eviction so later turns stay serialized", async () => { const work_dir = make_temp_dir(); const started: HeldRun[] = []; From 33dbf68d98bc7dbe166e163247f1a70ea8adf265 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 14:33:23 +0000 Subject: [PATCH 41/43] feat(gateway): GatewayPolicy and declared adapter capabilities (#144) Slice 2 of #144 (section A, items 2-3): - access.ts gains GatewayPolicy (allowlists + gateway toolset), built once from config; the runner builds it and hands it to the bus, which accepts an injected policy. - PlatformAdapter declares capabilities { kind: "text", max_reply_chars }. telegram 4096, discord 2000 (was a bare literal), twitch 450, webhook uncapped; idle adapters keep theirs. The runner logs them at start. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 6 ++++ docs/architecture/extending.md | 17 +++++++--- src/gateway/access.ts | 19 ++++++++++- src/gateway/bus.ts | 10 +++--- src/gateway/discord.ts | 9 +++-- src/gateway/runner.ts | 10 +++--- src/gateway/telegram.ts | 6 ++-- src/gateway/twitch.ts | 6 ++-- src/gateway/types.ts | 14 +++++++- src/gateway/webhook.ts | 2 ++ test/gateway.test.ts | 61 +++++++++++++++++++++++++++++++++- test/gateway_shutdown.test.ts | 4 ++- 12 files changed, 140 insertions(+), 24 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 602c43f..2b7e86a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,12 @@ ## Unreleased +- Gateway (#144): sender allowlists and the gateway toolset are one + `GatewayPolicy`, built from config at startup (`src/gateway/access.ts`). + Platform adapters declare `capabilities` (`{ kind: "text", max_reply_chars }`); + the runner logs them. **Breaking for custom adapters:** `PlatformAdapter` + requires `capabilities`, and `create_idle_adapter` takes it as a third + argument. - Gateway (#144): conversations queue on the shared `SessionManager` instead of their own promise chains. Beyond `max_conversations`, the least recently used conversation is evicted (it was the oldest inserted). Only Telegram's `/start` diff --git a/docs/architecture/extending.md b/docs/architecture/extending.md index eb82bf6..0b07753 100644 --- a/docs/architecture/extending.md +++ b/docs/architecture/extending.md @@ -156,16 +156,23 @@ A platform adapter is an implementation of the `PlatformAdapter` contract ```ts export interface PlatformAdapter { readonly name: string; + readonly capabilities: AdapterCapabilities; // { kind: "text", max_reply_chars?: number } start(): Promise; stop(): Promise; } ``` +`capabilities` declares what the adapter can deliver. Every platform today is +`kind: "text"` (one plain reply per message); set `max_reply_chars` to the +platform's message cap and split replies to it. The runner logs it at start. + Inbound flow: the adapter normalizes a platform event into `(platform, chat_id, user_id, text)` and calls `run_inbound_message` with the -shared `InboundHandler`; the handler is the bus's `handle`, which resolves to -the reply text; the adapter then sends the reply through the platform's send -API. The bus owns history, serialization, and error sanitization - adapters +shared `InboundHandler`; the handler is the bus's `reply`, and +`run_inbound_message` reduces its result to the reply text; the adapter then sends the reply through the platform's send +API. The bus owns history, serialization, and error sanitization, and checks +each sender against the `GatewayPolicy` built from config at startup +(`src/gateway/access.ts`: allowlists plus the gateway toolset) - adapters stay thin. Recipe (mirroring [`telegram.ts`](../../src/gateway/telegram.ts), the @@ -189,7 +196,7 @@ await send_reply(reply); 3. **Handle missing credentials with the idle-adapter pattern.** When the platform's token env var is absent, return - `create_idle_adapter("platform", "ENV_VAR not set")` instead of throwing. + `create_idle_adapter("platform", "ENV_VAR not set", CAPABILITIES)` instead of throwing. The adapter logs once why it is idle and no-ops on start/stop, so the gateway keeps serving the platforms that *do* have credentials - `run_gateway` only needs one valid platform. @@ -198,7 +205,7 @@ await send_reply(reply); structural `RawSocket` type works on both Bun and Node runtimes), `sanitize_agent_error` for reply strings on failure, and the whitespace splitter `split_text` from `src/gateway/format.ts` for size-capped - transports (telegram caps at 4096, discord at 2000, twitch at 512). + transports (telegram caps at 4096, discord at 2000, twitch at 450 to stay under its 500-char limit). **Testing.** `test/gateway.test.ts` covers the webhook adapter over a real `node:http` server (a `on_listening` test hook reports the bound port), the diff --git a/src/gateway/access.ts b/src/gateway/access.ts index 3533ccd..b852c3f 100644 --- a/src/gateway/access.ts +++ b/src/gateway/access.ts @@ -1,5 +1,7 @@ /** - * Gateway access: public-platform allowlists and the safe default toolset. + * Gateway access: public-platform allowlists and the safe default toolset, + * gathered into one GatewayPolicy built at startup (decision 0001: Profiles + * and Policy evolve from this file). */ import type { AgentConfig } from "../agent/config.js"; import { logger } from "../util/log.js"; @@ -42,6 +44,21 @@ export function is_gateway_sender_allowed( return user_ok && chat_ok; } +/** What the gateway lets in and what its agent may use; built once from config. */ +export interface GatewayPolicy { + /** Tools the shared gateway agent registers. */ + readonly tools_enabled: "all" | readonly string[]; + /** True when the sender may talk to the agent; logs a denial. */ + allows(platform: string, chat_id: string, user_id: string): boolean; +} + +export function create_gateway_policy(config: AgentConfig): GatewayPolicy { + return { + tools_enabled: gateway_tools_enabled(config), + allows: (platform, chat_id, user_id) => check_gateway_sender(config, platform, chat_id, user_id), + }; +} + /** Logs and returns false when the sender is not on the allowlist. */ export function check_gateway_sender( config: AgentConfig, diff --git a/src/gateway/bus.ts b/src/gateway/bus.ts index 61e6573..9447650 100644 --- a/src/gateway/bus.ts +++ b/src/gateway/bus.ts @@ -12,7 +12,7 @@ import { history_after_run_error } from "../agent/loop.js"; import type { Message } from "../providers/types.js"; import { create_session_manager, type SessionManager } from "../session/manager.js"; import { logger } from "../util/log.js"; -import { check_gateway_sender } from "./access.js"; +import { create_gateway_policy, type GatewayPolicy } from "./access.js"; import { sanitize_agent_error, type GatewayReply } from "./types.js"; export interface GatewayBusOptions { @@ -22,6 +22,8 @@ export interface GatewayBusOptions { export interface BusParams { config: AgentConfig; + /** Who may talk to the agent; defaults to `create_gateway_policy(config)`. */ + policy?: GatewayPolicy; agent_factory: () => Agent; /** Subscribe to agent tool events for debug logging (creates the agent). */ wire_tool_logging?: boolean; @@ -31,7 +33,7 @@ const DEFAULT_HISTORY_CAP = 40; const DEFAULT_MAX_CONVERSATIONS = 200; export class GatewayBus { - private readonly config: AgentConfig; + private readonly policy: GatewayPolicy; private readonly agent_factory: () => Agent; private agent: Agent | undefined; private readonly histories: Map = new Map(); @@ -41,7 +43,7 @@ export class GatewayBus { private stop_logging: (() => void) | undefined; constructor(params: BusParams, options?: GatewayBusOptions) { - this.config = params.config; + this.policy = params.policy ?? create_gateway_policy(params.config); this.agent_factory = params.agent_factory; this.history_cap = options?.history_cap ?? DEFAULT_HISTORY_CAP; this.max_conversations = options?.max_conversations ?? DEFAULT_MAX_CONVERSATIONS; @@ -58,7 +60,7 @@ export class GatewayBus { /** Like `handle`, with the run's usage and whether it failed; undefined when the sender is denied. */ async reply(platform: string, chat_id: string, user_id: string, text: string): Promise { - if (check_gateway_sender(this.config, platform, chat_id, user_id) === false) { + if (this.policy.allows(platform, chat_id, user_id) === false) { return undefined; } const key = conversation_key(platform, chat_id); diff --git a/src/gateway/discord.ts b/src/gateway/discord.ts index 2e84949..d0dcb92 100644 --- a/src/gateway/discord.ts +++ b/src/gateway/discord.ts @@ -7,13 +7,15 @@ import { logger } from "../util/log.js"; import { sleep } from "../util/sleep.js"; import { platform_token_env, read_platform_token } from "./token_env.js"; import type { AdapterParams, PlatformAdapter, RawSocket } from "./types.js"; -import { create_idle_adapter, open_socket, run_inbound_message } from "./types.js"; +import { create_idle_adapter, open_socket, run_inbound_message, type AdapterCapabilities } from "./types.js"; const DISCORD_API = "https://discord.com/api/v10"; const GATEWAY_URL = "wss://gateway.discord.gg/?v=10&encoding=json"; /** Guild messages + message content + direct messages. */ const INTENTS = 512 | 32768 | 4096; export const DISCORD_BACKOFF_MS = [5000, 10000, 20000, 30000] as const; +export const DISCORD_MAX_MESSAGE_CHARS = 2000; +export const DISCORD_CAPABILITIES: AdapterCapabilities = { kind: "text", max_reply_chars: DISCORD_MAX_MESSAGE_CHARS }; /** Auth / sharding / API-version failures — reconnecting cannot recover. */ const DISCORD_FATAL_CLOSE = new Set([4004, 4010, 4011, 4012, 4013, 4014]); /** Live heartbeat timers keyed by socket, cleared when the session ends. */ @@ -41,13 +43,14 @@ interface DiscordSessionState { export function create_discord_adapter(params: AdapterParams): PlatformAdapter { const token = read_platform_token(params.config, "discord"); if (token === undefined) { - return create_idle_adapter("discord", `${platform_token_env(params.config, "discord")} not set`); + return create_idle_adapter("discord", `${platform_token_env(params.config, "discord")} not set`, DISCORD_CAPABILITIES); } let running = false; let socket: RawSocket | undefined; let stop_controller = new AbortController(); return { name: "discord", + capabilities: DISCORD_CAPABILITIES, start: async () => { running = true; stop_controller = new AbortController(); @@ -226,7 +229,7 @@ async function rest_send_message(params: AdapterParams, channel_id: string, text if (token === undefined || channel_id === "") { return; } - for (const chunk of split_chunks(text, 2000)) { + for (const chunk of split_chunks(text, DISCORD_MAX_MESSAGE_CHARS)) { const response = await fetch(`${DISCORD_API}/channels/${channel_id}/messages`, { method: "POST", headers: { authorization: `Bot ${token}`, "content-type": "application/json" }, diff --git a/src/gateway/runner.ts b/src/gateway/runner.ts index 298abe5..8cb3653 100644 --- a/src/gateway/runner.ts +++ b/src/gateway/runner.ts @@ -5,7 +5,7 @@ import { create_agent_with_plugins, type Agent } from "../agent/agent.js"; import type { AgentConfig } from "../agent/config.js"; import { logger } from "../util/log.js"; -import { gateway_tools_enabled } from "./access.js"; +import { create_gateway_policy } from "./access.js"; import { GatewayBus } from "./bus.js"; import { create_discord_adapter } from "./discord.js"; import { create_telegram_adapter } from "./telegram.js"; @@ -15,9 +15,10 @@ import { create_webhook_adapter } from "./webhook.js"; /** Preload plugins, then hand that same agent to the bus factory. */ export async function create_gateway_bus(config: AgentConfig): Promise<{ agent: Agent; bus: GatewayBus }> { - const agent_config = { ...config, tools_enabled: gateway_tools_enabled(config) }; + const policy = create_gateway_policy(config); + const agent_config = { ...config, tools_enabled: policy.tools_enabled }; const agent = await create_agent_with_plugins(agent_config); - const bus = new GatewayBus({ config, agent_factory: () => agent, wire_tool_logging: true }); + const bus = new GatewayBus({ config, policy, agent_factory: () => agent, wire_tool_logging: true }); return { agent, bus }; } @@ -102,7 +103,8 @@ async function start_all_adapters(adapters: readonly PlatformAdapter[]): Promise for (const adapter of adapters) { try { await adapter.start(); - logger.info(`gateway adapter started: ${adapter.name}`); + const cap = adapter.capabilities.max_reply_chars; + logger.info(`gateway adapter started: ${adapter.name} (${adapter.capabilities.kind}${cap === undefined ? "" : `, replies split at ${cap} chars`})`); } catch (error) { logger.error(`gateway adapter failed to start: ${adapter.name}`, error); } diff --git a/src/gateway/telegram.ts b/src/gateway/telegram.ts index e7ffe9a..28465fc 100644 --- a/src/gateway/telegram.ts +++ b/src/gateway/telegram.ts @@ -7,9 +7,10 @@ import { sleep } from "../util/sleep.js"; import { logger } from "../util/log.js"; import { platform_token_env, read_platform_token } from "./token_env.js"; import type { AdapterParams, PlatformAdapter } from "./types.js"; -import { create_idle_adapter, run_inbound_message } from "./types.js"; +import { create_idle_adapter, run_inbound_message, type AdapterCapabilities } from "./types.js"; export const TELEGRAM_MAX_MESSAGE_CHARS = 4096; +export const TELEGRAM_CAPABILITIES: AdapterCapabilities = { kind: "text", max_reply_chars: TELEGRAM_MAX_MESSAGE_CHARS }; interface TelegramUpdate { update_id?: number; @@ -25,11 +26,12 @@ export const TELEGRAM_BACKOFF_MS = [2000, 4000, 8000, 16000, 30000] as const; export function create_telegram_adapter(params: AdapterParams): PlatformAdapter { const token = read_platform_token(params.config, "telegram"); if (token === undefined) { - return create_idle_adapter("telegram", `${platform_token_env(params.config, "telegram")} not set`); + return create_idle_adapter("telegram", `${platform_token_env(params.config, "telegram")} not set`, TELEGRAM_CAPABILITIES); } let running = false; return { name: "telegram", + capabilities: TELEGRAM_CAPABILITIES, start: async () => { running = true; void poll_loop(params, token, () => running); diff --git a/src/gateway/twitch.ts b/src/gateway/twitch.ts index d4ceb51..38501a9 100644 --- a/src/gateway/twitch.ts +++ b/src/gateway/twitch.ts @@ -8,11 +8,12 @@ import { logger } from "../util/log.js"; import { sleep } from "../util/sleep.js"; import { platform_token_env, read_platform_token } from "./token_env.js"; import type { AdapterParams, PlatformAdapter, RawSocket } from "./types.js"; -import { create_idle_adapter, open_socket, run_inbound_message } from "./types.js"; +import { create_idle_adapter, open_socket, run_inbound_message, type AdapterCapabilities } from "./types.js"; export const TWITCH_IRC_URL = "wss://irc-ws.chat.twitch.tv:443"; /** Content budget under Twitch's 500-char message / 512-byte IRC line caps. */ export const TWITCH_MESSAGE_CAP = 450; +export const TWITCH_CAPABILITIES: AdapterCapabilities = { kind: "text", max_reply_chars: TWITCH_MESSAGE_CAP }; export const TWITCH_BACKOFF_MS = [5000, 10000, 20000, 30000] as const; const TWITCH_CHUNK_GAP_MS = 1600; @@ -34,13 +35,14 @@ export function create_twitch_adapter(params: AdapterParams): PlatformAdapter { if (twitch === undefined) { const token_env = platform_token_env(params.config, "twitch"); const reason = token_env === "LICH_TWITCH_OAUTH_TOKEN" ? "LICH_TWITCH_OAUTH_TOKEN / NICK not set" : `${token_env} / LICH_TWITCH_NICK not set`; - return create_idle_adapter("twitch", reason); + return create_idle_adapter("twitch", reason, TWITCH_CAPABILITIES); } let running = false; let socket: RawSocket | undefined; let stop_controller = new AbortController(); return { name: "twitch", + capabilities: TWITCH_CAPABILITIES, start: async () => { running = true; stop_controller = new AbortController(); diff --git a/src/gateway/types.ts b/src/gateway/types.ts index f2b20cd..d8bfae2 100644 --- a/src/gateway/types.ts +++ b/src/gateway/types.ts @@ -23,9 +23,20 @@ export interface ReplySink { send(text: string): Promise; } +/** + * What an adapter can deliver (decision 0001: adapters declare capabilities). + * Every platform adapter today is "text": a plain reply per message, split + * into chunks of at most `max_reply_chars` when the platform caps length. + */ +export interface AdapterCapabilities { + readonly kind: "text"; + readonly max_reply_chars?: number; +} + /** Lifecycle contract implemented by every platform adapter. */ export interface PlatformAdapter { readonly name: string; + readonly capabilities: AdapterCapabilities; start(): Promise; stop(): Promise; } @@ -81,10 +92,11 @@ export function sanitize_agent_error(error: unknown): string { } /** Adapter that logs why it is idle once and otherwise does nothing. */ -export function create_idle_adapter(name: string, reason: string): PlatformAdapter { +export function create_idle_adapter(name: string, reason: string, capabilities: AdapterCapabilities): PlatformAdapter { logger.warn(`gateway ${name} adapter idle: ${reason}`); return { name, + capabilities, start: async () => undefined, stop: async () => undefined, }; diff --git a/src/gateway/webhook.ts b/src/gateway/webhook.ts index fafb775..b1ba21d 100644 --- a/src/gateway/webhook.ts +++ b/src/gateway/webhook.ts @@ -35,6 +35,8 @@ export function create_webhook_adapter(params: WebhookAdapterParams): PlatformAd return { name: "webhook", + // One JSON reply per request; no length cap. + capabilities: { kind: "text" }, start: async () => { assert_bind_allowed(host, token); server = createServer((request, response) => { diff --git a/test/gateway.test.ts b/test/gateway.test.ts index c56c261..79d3166 100644 --- a/test/gateway.test.ts +++ b/test/gateway.test.ts @@ -4,15 +4,20 @@ * No real network to telegram/discord/twitch — webhook binds an ephemeral * port (0) on loopback only. */ -import { afterEach, describe, expect, it } from "vitest"; +import { afterEach, describe, expect, it, vi } from "vitest"; import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; import { join } from "node:path"; import { parse_agent_config } from "../src/agent/config.js"; import { + create_gateway_policy, DEFAULT_GATEWAY_TOOLS_ENABLED, gateway_tools_enabled, is_gateway_sender_allowed, + type GatewayPolicy, } from "../src/gateway/access.js"; +import { create_discord_adapter } from "../src/gateway/discord.js"; +import { create_twitch_adapter } from "../src/gateway/twitch.js"; +import { logger } from "../src/util/log.js"; import { assert_bind_allowed, create_webhook_adapter, MAX_WEBHOOK_BODY_BYTES } from "../src/gateway/webhook.js"; import { format_agent_reply, split_text } from "../src/gateway/format.js"; import { create_telegram_adapter } from "../src/gateway/telegram.js"; @@ -607,6 +612,60 @@ describe("gateway access", () => { rmSync(work_dir, { recursive: true, force: true }); } }); + + it("builds one GatewayPolicy from config: toolset plus the sender allowlists", () => { + const work_dir = temp_work_dir(); + try { + const policy = create_gateway_policy(config_for(work_dir, { allowed_users: { discord: ["u1"] } })); + expect(policy.tools_enabled).toEqual([...DEFAULT_GATEWAY_TOOLS_ENABLED]); + expect(policy.allows("webhook", "c", "anyone")).toBe(true); + expect(policy.allows("discord", "c", "u1")).toBe(true); + expect(policy.allows("discord", "c", "u2")).toBe(false); + expect(policy.allows("telegram", "c", "u1")).toBe(false); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); + + it("lets the bus use an injected policy instead of the config's", async () => { + const work_dir = temp_work_dir(); + try { + const records: RunRecord[] = []; + const deny_all: GatewayPolicy = { tools_enabled: [], allows: () => false }; + const bus = new GatewayBus({ config: config_for(work_dir), policy: deny_all, agent_factory: () => recording_agent(records) }); + expect(await bus.handle("webhook", "c1", "u1", "hello")).toBeUndefined(); + expect(records).toEqual([]); + } finally { + rmSync(work_dir, { recursive: true, force: true }); + } + }); +}); + +describe("adapter capabilities", () => { + it("declares each platform's reply cap, also on idle adapters", () => { + const saved = { ...process.env }; + for (const key of ["LICH_TELEGRAM_BOT_TOKEN", "LICH_DISCORD_BOT_TOKEN", "LICH_TWITCH_OAUTH_TOKEN", "LICH_TWITCH_NICK"]) { + delete process.env[key]; + } + const params = { + config: parse_agent_config({ providers: [{ kind: "openai_compat", name: "m", model: "x" }] }), + handle_message: async () => "", + get_agent: (): Agent => { + throw new Error("not used"); + }, + reply_router: () => undefined, + }; + vi.spyOn(logger, "warn").mockImplementation(() => undefined); + try { + expect(create_telegram_adapter(params).capabilities).toEqual({ kind: "text", max_reply_chars: 4096 }); + expect(create_discord_adapter(params).capabilities).toEqual({ kind: "text", max_reply_chars: 2000 }); + expect(create_twitch_adapter(params).capabilities).toEqual({ kind: "text", max_reply_chars: TWITCH_MESSAGE_CAP }); + expect(create_webhook_adapter({ ...params, port: 0, host: "127.0.0.1" }).capabilities).toEqual({ kind: "text" }); + } finally { + vi.restoreAllMocks(); + process.env = saved; + } + }); }); describe("webhook adapter", () => { diff --git a/test/gateway_shutdown.test.ts b/test/gateway_shutdown.test.ts index 2802e43..a6a835f 100644 --- a/test/gateway_shutdown.test.ts +++ b/test/gateway_shutdown.test.ts @@ -22,11 +22,13 @@ describe("shutdown_gateway", () => { const { agent, bus } = tracked(order); const slow: PlatformAdapter = { name: "slow", + capabilities: { kind: "text" }, start: async () => undefined, stop: () => new Promise((resolve) => setTimeout(() => resolve(order.push("slow.stop") as unknown as void), 20)), }; const broken: PlatformAdapter = { name: "broken", + capabilities: { kind: "text" }, start: async () => undefined, stop: async () => { throw new Error("socket gone"); @@ -41,7 +43,7 @@ describe("shutdown_gateway", () => { it("stops waiting after the timeout when an adapter hangs", async () => { const order: string[] = []; const { agent, bus } = tracked(order); - const hung: PlatformAdapter = { name: "hung", start: async () => undefined, stop: () => new Promise(() => undefined) }; + const hung: PlatformAdapter = { name: "hung", capabilities: { kind: "text" }, start: async () => undefined, stop: () => new Promise(() => undefined) }; const warn = vi.spyOn(logger, "warn").mockImplementation(() => undefined); await shutdown_gateway(agent, bus, [hung], 20); expect(order).toEqual(["bus.stop", "agent.close"]); From 13551c8995b504394793a393c19a3613a63d7e7b Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 15:46:22 +0000 Subject: [PATCH 42/43] feat(serve): queue frames per session on each connection (#144) Frames are now queued by params.session_id instead of one tail per connection, so a long prompt.submit no longer holds up other sessions or session-less frames (health, session.list, session.create) on the same WebSocket. One session's frames keep their order; prompt.abort still bypasses every queue. Idle lanes are dropped. Docs: the stale "runs are serialized" note is replaced (events are per-run since #136). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 5 +++ docs/architecture/serve.md | 13 +++++--- src/serve/server.ts | 31 ++++++++++++++---- test/serve_prompts.test.ts | 64 +++++++++++++++++++++++++++++++++++++- 4 files changed, 102 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2b7e86a..175d29f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,11 @@ ## Unreleased +- `lich serve` queues frames per session on each connection instead of + per connection (#144). A long `prompt.submit` no longer holds up other + sessions or session-less frames such as `health` on the same socket. Frames + for one session keep their order; replies across sessions may interleave, + so clients match them by JSON-RPC `id` (Ossuary already does). - Gateway (#144): sender allowlists and the gateway toolset are one `GatewayPolicy`, built from config at startup (`src/gateway/access.ts`). Platform adapters declare `capabilities` (`{ kind: "text", max_reply_chars }`); diff --git a/docs/architecture/serve.md b/docs/architecture/serve.md index a1ca313..78056d9 100644 --- a/docs/architecture/serve.md +++ b/docs/architecture/serve.md @@ -84,8 +84,9 @@ about it from the next `session.clear` or `prompt.submit` failing with `not_found`. Evicted transcripts stay on disk and can be resumed again. `prompt.submit` runs the server Agent with that bag's history and -`AgentRunOptions.session` (one JSONL file per serve session). Runs are serialized -so AgentEvent fan-out stays correctly tagged with `session_id`. While a run is +`AgentRunOptions.session` (one JSONL file per serve session). Runs of one +session are serialized; different sessions run concurrently. Each run's events +arrive through its own `on_event` and are tagged with that run's `session_id`. While a run is in flight, `prompt.abort` aborts it via `AbortSignal`. The submit result mirrors `AgentRunResult` (`reply`, `usage`, `session_path`, `turns_used`, `stopped_reason`). @@ -113,8 +114,12 @@ same WebSocket that issued `prompt.submit` while the call is still in flight. do not assume `file://` / `app://` behavior here. - Frame size capped at ~1 MiB (`maxPayload`). - `health` returns `{ status: "ok", version }` (`LICH_VERSION` from `src/version.ts`). -- Frames are handled per-connection in arrival order: pipelined requests get - in-order replies even when earlier requests hit slower filesystem awaits. +- Frames are queued per session within a connection: frames whose + `params.session_id` match are handled in arrival order and get in-order + replies, while other sessions, and frames that name no session (such as + `health`, `session.list` or `session.create`), do not wait behind them. + Replies from different sessions can therefore interleave; match them by + JSON-RPC `id`. `prompt.abort` is dispatched immediately so it can cancel an in-flight `prompt.submit` on the same socket instead of waiting behind that run. A submit already accepted on that socket, but still waiting behind another diff --git a/src/serve/server.ts b/src/serve/server.ts index 082117d..a8f5ebd 100644 --- a/src/serve/server.ts +++ b/src/serve/server.ts @@ -233,11 +233,13 @@ function attach_client( prompts: ServePromptService | undefined, inflight: Set>, ): void { - // Serialize frames per connection: pipelined requests get in-order replies, - // and the tail catch keeps any rejection from becoming an unhandled one. - // prompt.abort is not on that tail. It must run while prompt.submit is still + // Serialize frames per session on this connection: frames for one session_id + // get in-order replies, while different sessions (and session-less frames + // such as health or session.list) no longer wait behind each other's runs. + // Replies across lanes can interleave; clients match them by JSON-RPC id. + // prompt.abort is on no lane. It must run while prompt.submit is still // awaiting the model; waiting would deadlock a hung run with its own cancel. - let tail: Promise = Promise.resolve(); + const tails = new Map>(); const track = (chain: Promise): void => { inflight.add(chain); void chain.finally(() => { @@ -256,12 +258,29 @@ function attach_client( // Arm the controller before this frame waits on `tail`, so prompt.abort // (which skips the tail) can cancel a submit that has not started yet. const prepared = arm_queued_submit(data, prompts); - const chain = tail.then(() => run_frame(data, prepared)); - tail = chain; + const lane = frame_lane(data); + const chain = (tails.get(lane) ?? Promise.resolve()).then(() => run_frame(data, prepared)); + tails.set(lane, chain); track(chain); + // Drop an idle lane so the map does not grow with every session ever used. + void chain.finally(() => { + if (tails.get(lane) === chain) { + tails.delete(lane); + } + }); }); } +/** Lane for a frame: its params.session_id, or "" for frames that name no session. */ +function frame_lane(data: RawData): string { + const params = parse_frame_object(data)?.params; + if (typeof params !== "object" || params === null || Array.isArray(params)) { + return ""; + } + const session_id = (params as { session_id?: unknown }).session_id; + return typeof session_id === "string" && session_id.length > 0 ? `session:${session_id}` : ""; +} + interface PreparedSubmit { session_id: string; controller: AbortController; diff --git a/test/serve_prompts.test.ts b/test/serve_prompts.test.ts index f1a4dc7..b384e30 100644 --- a/test/serve_prompts.test.ts +++ b/test/serve_prompts.test.ts @@ -1104,11 +1104,12 @@ describe("serve prompt over websocket", () => { return body; }); await started; + // Same session_id, so this frame shares the submit's lane and must wait behind it. const health_promise = request_rpc(ws, { jsonrpc: "2.0", id: 4, method: "health", - params: { note: "prompt.abort" }, + params: { session_id, note: "prompt.abort" }, }).then((body) => { order.push(4); return body; @@ -1142,6 +1143,67 @@ describe("serve prompt over websocket", () => { } }); + it("runs different sessions on one connection concurrently, and session-less frames do not wait", async () => { + const work_dir = await make_temp_dir("serve-ws-lanes"); + const session_dir = path.join(work_dir, "sessions"); + let release_hang!: () => void; + const hang = new Promise((resolve) => { + release_hang = resolve; + }); + let hang_started!: () => void; + const started = new Promise((resolve) => { + hang_started = resolve; + }); + const fetch_fn: typeof fetch = async (_url, init) => { + const body = JSON.parse(String(init?.body)) as { messages: Array<{ content?: string }> }; + if (body.messages.some((message) => message.content === "hang")) { + hang_started(); + await hang; + } + return new Response( + JSON.stringify({ + model: "mock-model", + choices: [{ message: { role: "assistant", content: "done" }, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + { status: 200 }, + ); + }; + const agent = mock_agent(work_dir, fetch_fn); + const server = create_serve_server({ port: 0, boot_stdout: null, session_dir, agent, version: "9.9.9" }); + servers.push(server); + const boot = await server.start(); + const ws = await open_ws(`ws://127.0.0.1:${boot.port}/?token=${encodeURIComponent(boot.token)}`); + try { + const create = async (id: number): Promise => + ((await request_rpc(ws, { jsonrpc: "2.0", id, method: "session.create", params: { source: "test" } })).result as { + session_id: string; + }).session_id; + const a = await create(1); + const b = await create(2); + const order: string[] = []; + const slow = request_rpc(ws, { jsonrpc: "2.0", id: 3, method: "prompt.submit", params: { session_id: a, text: "hang" } }).then( + (body) => { + order.push("a"); + return body; + }, + ); + await started; + const fast = await request_rpc(ws, { jsonrpc: "2.0", id: 4, method: "prompt.submit", params: { session_id: b, text: "hi" } }); + order.push("b"); + const health = await request_rpc(ws, { jsonrpc: "2.0", id: 5, method: "health" }); + order.push("health"); + expect(fast.result).toMatchObject({ session_id: b, stopped_reason: "final" }); + expect(health.result).toBeDefined(); + release_hang(); + expect((await slow).result).toMatchObject({ session_id: a, stopped_reason: "final" }); + expect(order).toEqual(["b", "health", "a"]); + } finally { + release_hang(); + ws.close(); + } + }); + it("stop() aborts an in-flight prompt instead of draining the model call", async () => { const work_dir = await make_temp_dir("serve-stop-abort"); const session_dir = path.join(work_dir, "sessions"); From fbc70309afd28c10a856dc51cde69671c2bb8827 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 16:10:43 +0000 Subject: [PATCH 43/43] feat(serve): report protocol version and capabilities in health (#144) health now returns protocol_version (SERVE_PROTOCOL_VERSION = 1, bumped only on breaking changes) and capabilities { methods, notifications }. methods lists prompt.* only when the server has an Agent, so clients can check before calling instead of discovering an application error. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01LHQAfXHofy6QkAgmfmJava --- CHANGELOG.md | 3 +++ docs/architecture/serve.md | 10 ++++++++-- src/index.ts | 2 ++ src/serve/protocol.ts | 18 ++++++++++++++++++ src/serve/rpc.ts | 19 +++++++++++++++++-- test/serve_cli.test.ts | 2 +- test/serve_rpc.test.ts | 4 ++-- test/serve_transport.test.ts | 26 +++++++++++++++++++++++--- 8 files changed, 74 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 175d29f..b79942a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,9 @@ ## Unreleased +- `lich serve` `health` also returns `protocol_version` (`1`, bumped only on + breaking changes) and `capabilities: { methods, notifications }` (#144). + `methods` lists `prompt.*` only when the server has an Agent. - `lich serve` queues frames per session on each connection instead of per connection (#144). A long `prompt.submit` no longer holds up other sessions or session-less frames such as `health` on the same socket. Frames diff --git a/docs/architecture/serve.md b/docs/architecture/serve.md index 78056d9..1aaa413 100644 --- a/docs/architecture/serve.md +++ b/docs/architecture/serve.md @@ -42,7 +42,7 @@ method `event` (no `id`). | Method | Params | Result | | --- | --- | --- | -| `health` | `{}` | `{ status: "ok", version }` (`LICH_VERSION`) | +| `health` | `{}` | `{ status: "ok", version, protocol_version, capabilities }` | | `session.create` | `{ label?, source }` | `{ session_id }` | | `session.list` | `{}` | `{ sessions: [{ id, mtime_ms }, ...] }` | | `session.clear` | `{ session_id }` | `{ session_id }` | @@ -50,6 +50,13 @@ method `event` (no `id`). | `prompt.submit` | `{ session_id, text }` | reply, usage, `session_path`, `stopped_reason`, … | | `prompt.abort` | `{ session_id }` | `{ session_id, aborted }` | +`health` reports `version` (`LICH_VERSION`), `protocol_version` +(`SERVE_PROTOCOL_VERSION`, currently `1`) and +`capabilities: { methods, notifications }`. `protocol_version` changes only +on breaking changes; new features are added to `capabilities`, so clients +check it before calling a method. `methods` lists `prompt.submit` and +`prompt.abort` only when the server has an Agent. + `health` and `session.list` have no param fields. Typed clients still send `params: {}`, because `ServeRequest` requires `params`. JSON-RPC 2.0 also allows omitting `params`; serve handlers accept that omission the same as `{}`. @@ -113,7 +120,6 @@ same WebSocket that issued `prompt.submit` while the call is still in flight. - Origin allowlisting is deferred until Electron’s page origin policy is decided; do not assume `file://` / `app://` behavior here. - Frame size capped at ~1 MiB (`maxPayload`). -- `health` returns `{ status: "ok", version }` (`LICH_VERSION` from `src/version.ts`). - Frames are queued per session within a connection: frames whose `params.session_id` match are handled in arrival order and get in-order replies, while other sessions, and frames that name no session (such as diff --git a/src/index.ts b/src/index.ts index 2249415..7c1e8c5 100644 --- a/src/index.ts +++ b/src/index.ts @@ -63,6 +63,7 @@ export { SERVE_METHODS, SERVE_NOTIFICATION_EVENT, SERVE_ERROR_CODES, + SERVE_PROTOCOL_VERSION, } from "./serve/protocol.js"; export type { HealthParams, @@ -78,6 +79,7 @@ export type { PromptAbortResult, PromptSubmitParams, PromptSubmitResult, + ServeCapabilities, ServeEventNotification, ServeEventParams, ServeMethod, diff --git a/src/serve/protocol.ts b/src/serve/protocol.ts index f2d1f66..2f77615 100644 --- a/src/serve/protocol.ts +++ b/src/serve/protocol.ts @@ -24,6 +24,12 @@ export const SERVE_NOTIFICATION_EVENT = "event" as const; export type ServeNotificationMethod = typeof SERVE_NOTIFICATION_EVENT; +/** + * Wire protocol version reported by `health`. Bumped only on breaking changes; + * additive features show up in `capabilities` instead. + */ +export const SERVE_PROTOCOL_VERSION = 1; + export type JsonRpcId = string | number; export interface JsonRpcRequest { @@ -75,7 +81,19 @@ export type HealthParams = Record; export interface HealthResult { status: "ok"; + /** Lich release (`LICH_VERSION`). */ version: string; + /** `SERVE_PROTOCOL_VERSION` of this server. */ + protocol_version: number; + capabilities: ServeCapabilities; +} + +/** What this server answers, so clients can check before calling. */ +export interface ServeCapabilities { + /** Methods served; `prompt.*` only when the server has an Agent. */ + methods: ServeMethod[]; + /** Server → client notification methods. */ + notifications: ServeNotificationMethod[]; } export interface SessionCreateParams { diff --git a/src/serve/rpc.ts b/src/serve/rpc.ts index e78c102..2435051 100644 --- a/src/serve/rpc.ts +++ b/src/serve/rpc.ts @@ -10,7 +10,12 @@ import type { JsonRpcSuccess, ServeMethod, } from "./protocol.js"; -import { SERVE_ERROR_CODES, SERVE_METHODS } from "./protocol.js"; +import { + SERVE_ERROR_CODES, + SERVE_METHODS, + SERVE_NOTIFICATION_EVENT, + SERVE_PROTOCOL_VERSION, +} from "./protocol.js"; import type { ServeEventNotify, ServePromptService } from "./prompts.js"; import type { ServeSessionStore } from "./sessions.js"; @@ -78,7 +83,17 @@ async function dispatch_method( if (params !== undefined && is_empty_params(params) !== true) { return error_response(id, SERVE_ERROR_CODES.INVALID_PARAMS, "Invalid params"); } - const result: HealthResult = { status: "ok", version: context.version }; + const result: HealthResult = { + status: "ok", + version: context.version, + protocol_version: SERVE_PROTOCOL_VERSION, + capabilities: { + methods: SERVE_METHODS.filter( + (name) => context.prompts !== undefined || name.startsWith("prompt.") !== true, + ), + notifications: [SERVE_NOTIFICATION_EVENT], + }, + }; return { jsonrpc: "2.0", id: id as JsonRpcId, result }; } if (method === "session.create") { diff --git a/test/serve_cli.test.ts b/test/serve_cli.test.ts index b7281c5..0300f83 100644 --- a/test/serve_cli.test.ts +++ b/test/serve_cli.test.ts @@ -165,7 +165,7 @@ describe("bun src/cli.ts serve boot", () => { expect(result).toEqual({ jsonrpc: "2.0", id: 1, - result: { status: "ok", version: LICH_VERSION }, + result: expect.objectContaining({ status: "ok", version: LICH_VERSION, protocol_version: 1 }), }); child.kill("SIGTERM"); await wait_exit(child); diff --git a/test/serve_rpc.test.ts b/test/serve_rpc.test.ts index 25c48f9..47dc189 100644 --- a/test/serve_rpc.test.ts +++ b/test/serve_rpc.test.ts @@ -130,7 +130,7 @@ describe("serve rpc framing", () => { expect(null_id.body).toEqual({ jsonrpc: "2.0", id: null, - result: { status: "ok", version: "9.9.9" }, + result: expect.objectContaining({ status: "ok", version: "9.9.9" }), }); const null_params = await dispatch( @@ -140,7 +140,7 @@ describe("serve rpc framing", () => { expect(null_params.body).toEqual({ jsonrpc: "2.0", id: 3, - result: { status: "ok", version: "9.9.9" }, + result: expect.objectContaining({ status: "ok", version: "9.9.9" }), }); }); diff --git a/test/serve_transport.test.ts b/test/serve_transport.test.ts index 792b478..93a1365 100644 --- a/test/serve_transport.test.ts +++ b/test/serve_transport.test.ts @@ -9,6 +9,8 @@ import path from "node:path"; import { Writable } from "node:stream"; import WebSocket from "ws"; import { LICH_VERSION } from "../src/index.js"; +import type { ServePromptService } from "../src/serve/prompts.js"; +import { SERVE_METHODS, SERVE_PROTOCOL_VERSION, type HealthResult } from "../src/serve/protocol.js"; import { handle_serve_rpc_message } from "../src/serve/rpc.js"; import { create_serve_server, type ServeServer } from "../src/serve/server.js"; import { create_serve_session_store } from "../src/serve/sessions.js"; @@ -91,7 +93,7 @@ afterEach(async () => { }); describe("serve rpc health", () => { - it("returns status and version for health", async () => { + it("returns status, versions and capabilities for health", async () => { const sessions = create_serve_session_store(path.join(await make_temp_dir("serve-health"), "s")); const raw = await handle_serve_rpc_message( JSON.stringify({ jsonrpc: "2.0", id: 1, method: "health", params: {} }), @@ -100,10 +102,28 @@ describe("serve rpc health", () => { expect(JSON.parse(raw ?? "")).toEqual({ jsonrpc: "2.0", id: 1, - result: { status: "ok", version: "9.9.9" }, + result: { + status: "ok", + version: "9.9.9", + protocol_version: SERVE_PROTOCOL_VERSION, + capabilities: { + methods: ["health", "session.create", "session.list", "session.clear", "session.resume"], + notifications: ["event"], + }, + }, }); }); + it("lists prompt methods in health only when an agent is configured", async () => { + const sessions = create_serve_session_store(path.join(await make_temp_dir("serve-health-agent"), "s")); + const raw = await handle_serve_rpc_message( + JSON.stringify({ jsonrpc: "2.0", id: 1, method: "health", params: {} }), + { version: "9.9.9", sessions, prompts: {} as ServePromptService }, + ); + const body = JSON.parse(raw ?? "") as { result: HealthResult }; + expect(body.result.capabilities.methods).toEqual([...SERVE_METHODS]); + }); + it("returns agent-not-configured for prompt.submit without an agent", async () => { const sessions = create_serve_session_store(path.join(await make_temp_dir("serve-stub"), "s")); const raw = await handle_serve_rpc_message( @@ -146,7 +166,7 @@ describe("serve websocket transport", () => { expect(result).toEqual({ jsonrpc: "2.0", id: 1, - result: { status: "ok", version: LICH_VERSION }, + result: expect.objectContaining({ status: "ok", version: LICH_VERSION }), }); });