Skip to content
25 changes: 24 additions & 1 deletion cmd/codeaf/carried.go
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ import (
"time"

"github.com/Agent-Field/agentfield/sdk/go/ai"
"github.com/Agent-Field/codeaf/internal/catalog"
"github.com/Agent-Field/codeaf/internal/config"
"github.com/Agent-Field/codeaf/internal/delegate"
"github.com/Agent-Field/codeaf/internal/delegate/builtin"
Expand Down Expand Up @@ -130,13 +131,21 @@ type carriedRoad struct {
seat string
defaultSeat func() (string, error)
signNamed bool
// warmCatalog waits, within ctx, for the profile's model catalog to be
// resolved and written down. Nil for a road with no catalog.
warmCatalog func(ctx context.Context) bool
}

// carriedModels resolves a shell run's road. It is the person's own profile,
// read the way `codeaf exec` reads it; a variable so a test can hand a
// scripted road instead of a profile and a key.
var carriedModels = profileRoad

// carriedCatalogWarm bounds how long a senior-dev shell run waits for the
// profile's model catalog to be written before it starts: the catalog's own
// fetch timeout, so the wait is never longer than one fetch.
const carriedCatalogWarm = catalog.FetchTimeout

// profileRoad is the road through the person's profile: config.Load's
// services and keys — so a machine with no key at all is answered with the
// one sentence every command gives it — the crew's work seat if this run
Expand Down Expand Up @@ -172,7 +181,8 @@ func profileRoad() (carriedRoad, error) {
}
return seats.Work.Model, nil
},
signNamed: config.AttributionModelAt(settings.ProfileDir),
signNamed: config.AttributionModelAt(settings.ProfileDir),
warmCatalog: settings.Models.Warmed,
}, nil
}

Expand Down Expand Up @@ -324,6 +334,19 @@ func runCarriedHost(ctx context.Context, inv *delegate.Invocation) error {
})
defer clock.Stop()
}
// SENIOR-DEV SIZES ITS MODELS FROM THE CATALOG THIS PROFILE KEEPS ON DISK
// whenever models.dev cannot answer (internal/config's
// CachedContextWindow), and it cannot fetch that catalog itself: it runs
// with no provider key. On a profile that has never listed its models the
// file is written by the catalog's first resolution, so the run waits for
// it, at most one catalog fetch's length. A profile that already holds the
// list answers at once, and a fetch that cannot land leaves the run to say
// it guessed.
if inv.Program.Name == "senior-dev" && road.warmCatalog != nil {
warmCtx, warmed := context.WithTimeout(runCtx, carriedCatalogWarm)
road.warmCatalog(warmCtx)
warmed()
}
ledger := session.UsageLedgerPath()
// A SHELL RUN'S ROWS NAME THE RUN. There is no conversation and no task
// behind them, and a row that named nothing was money the spending page
Expand Down
23 changes: 22 additions & 1 deletion cmd/codeaf/carried_seniordev_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ import (
"time"

"github.com/Agent-Field/agentfield/sdk/go/ai"
"github.com/Agent-Field/codeaf/internal/catalog"
"github.com/Agent-Field/codeaf/internal/delegate"
"github.com/Agent-Field/codeaf/internal/delegate/builtin"
"github.com/Agent-Field/codeaf/internal/provider"
Expand Down Expand Up @@ -95,7 +96,23 @@ func TestSeniorDevWorksATaskThroughTheShellHostsModelAPI(t *testing.T) {
t.Setenv("CODEAF_NO_UPDATE_CHECK", "1")
model := &seniorDevModel{}
previousRoad, previousOut, previousGrace := carriedModels, carriedStdout, carriedGrace
carriedModels = func() (carriedRoad, error) { return carriedRoad{completerFor: model.completerFor}, nil }
// THE PROFILE'S CATALOG IS WAITED FOR BEFORE THE PROGRAM STARTS, once and
// within one catalog fetch, so the window senior-dev reads from it is on
// disk before its first call (carried.go's carriedCatalogWarm).
var warms, callsAtWarm int
var warmBound time.Duration
carriedModels = func() (carriedRoad, error) {
return carriedRoad{completerFor: model.completerFor, warmCatalog: func(ctx context.Context) bool {
warms++
if deadline, ok := ctx.Deadline(); ok {
warmBound = time.Until(deadline)
}
model.mu.Lock()
callsAtWarm = model.calls
model.mu.Unlock()
return true
}}, nil
}
printed := &lockedBuffer{}
carriedStdout = printed
carriedGrace = 5 * time.Second
Expand All @@ -111,6 +128,10 @@ func TestSeniorDevWorksATaskThroughTheShellHostsModelAPI(t *testing.T) {
}
t.Fatalf("senior-dev's shell run left with %d:\n%s\nits stderr:\n%s", code, out, stderr)
}
if warms != 1 || callsAtWarm != 0 || warmBound <= 0 || warmBound > catalog.FetchTimeout {
t.Fatalf("catalog waited for %d times, after %d calls, bounded by %v; want once, before any call, within %v",
warms, callsAtWarm, warmBound, catalog.FetchTimeout)
}
for _, want := range []string{"senior-dev · working in " + workspace, "senior-dev finished", "senior-dev's model said: feature.txt now holds the feature", " · 300 in · 20 out · $0.0020"} {
if !strings.Contains(out, want) {
t.Fatalf("the shell run never printed %q:\n%s", want, out)
Expand Down
21 changes: 21 additions & 0 deletions docs/changes/unreleased/1791-senior-dev-small-windows.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
---
kind: fixed
title: senior-dev reads its window from codeaf offline, refuses tiny models, keeps progress on compaction
pr: 1791
surface: [engine, chat, docs]
invalidates:
- "senior-dev sized every model at 16,384 tokens when models.dev could not be reached. It now reads the window from the model catalog codeaf keeps for the profile, and only guesses 16,384 when neither knows the model, saying so on its page and in its log."
- "senior-dev started on any model it could size. A model the person asked for that is known to hold 32,768 tokens or fewer is now refused before its first call with `senior-dev cannot work with <model> (<n> tokens): …`, and nothing is spent; a crew seat that small is left out with a note instead."
- "When a compacted history did not fit, senior-dev replaced the summary with a stub that kept no progress. The stub now carries as much of the newest summary as fits, and the changed-files list."
- "The summary length budget assumed 2,048 tokens of fixed context. It now measures the system prompt, tools and pins, and never asks for more than the summarizer's output limit."
- "A context-overflow pin applied to one senior-dev session. It now applies to every session on the same model for the rest of the run."
---

A CyberGym run in a container that could not reach models.dev sized a 1M-token model at
16,384 tokens. It compacted 115 times in 25 minutes, and 113 of those compactions erased
its progress. The do stage delivered nothing. codeaf's own catalog in that container
already listed the real window.

A senior-dev shell run now waits up to one catalog fetch (`catalog.FetchTimeout`) for the
profile's model list before it starts, so a fresh profile writes the list senior-dev
reads.
6 changes: 5 additions & 1 deletion internal/catalog/catalog.go
Original file line number Diff line number Diff line change
Expand Up @@ -232,6 +232,10 @@ type cache struct {
Base string `json:"base,omitempty"`
}

// FetchTimeout bounds one listing request on the default HTTP client. A caller
// that waits for a catalog to resolve waits no longer than this.
const FetchTimeout = 15 * time.Second

// Options describes the one catalog fetch. Dir is the codeaf configuration
// directory (CODEAF_PROFILE_DIR when configured, ~/.codeaf otherwise).
type Options struct {
Expand Down Expand Up @@ -1197,7 +1201,7 @@ func fetch(ctx context.Context, options Options) ([]Model, error) {
}
client := options.HTTPClient
if client == nil {
client = &http.Client{Timeout: 15 * time.Second}
client = &http.Client{Timeout: FetchTimeout}
}
response, err := client.Do(request)
if err != nil {
Expand Down
56 changes: 56 additions & 0 deletions internal/config/window.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
package config

import (
"strings"

"github.com/Agent-Field/codeaf/internal/catalog"
"github.com/Agent-Field/codeaf/internal/env"
"github.com/Agent-Field/codeaf/internal/modelsource"
)

// CachedContextWindow is how many tokens a model accepts according to the
// model catalog this profile already keeps on disk, or zero when that catalog
// cannot say: an unknown model, a service whose list was never fetched, a
// profile that has never listed anything.
//
// IT NEVER REACHES THE NETWORK AND READS NO KEY. It exists for a program codeaf
// starts (senior-dev), which runs with every provider key taken out of its
// environment and may sit behind a network that reaches nothing but the run's
// own model API. That program still deserves the window codeaf itself would
// use, and codeaf wrote it down the last time it listed the models: the
// default service's listing in `model-catalog.json`, every other connected
// service's in its own compartment beside it ([CatalogOptionsFor]). The model
// is spelled the way a person writes it, and its service is found the way
// every other door finds one ([modelsource.Set.For]).
func CachedContextWindow(model string) int {
model = strings.TrimSpace(model)
if model == "" {
return 0
}
profileDir := ProfileDir()
values, _ := readProfileConfig(profileDir)
baseURL := firstNonEmpty(env.Get("CODEAF_BASE_URL"), DefaultBaseURL)
noKey := func(PersistedSource, modelsource.Source) string { return "" }
sources := sourceHomes(resolveSources("", baseURL, persistedSourcesFrom(values), noKey), profileDir)
service, bare := sources.For(model)
id := strings.TrimSpace(service.Source.ID)
switch {
case strings.EqualFold(id, modelsource.DefaultID):
// The default service's listing is the profile's one shared catalog,
// written under no source name (cmd/codeaf's sharedCatalog).
return catalog.Recall(catalog.Options{BaseURL: baseURL, Dir: profileDir}).ContextLength(bare)
case strings.EqualFold(id, "codex"):
// A Codex row remembered before its window was kept answers the
// window the fallback list names ([CodexRememberedModels]).
for _, row := range CodexRememberedModels(service, profileDir) {
if strings.EqualFold(strings.TrimSpace(row.ID), bare) {
return row.ContextLength
}
}
return 0
}
// The same compartment the connected service's listing is written to.
// Recall reads the file and nothing else, so the key the options carry
// (none here) and their client are never used.
return catalog.Recall(CatalogOptionsFor(service, profileDir)).ContextLength(bare)
}
96 changes: 96 additions & 0 deletions internal/config/window_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,96 @@
package config

import (
"net/http"
"net/http/httptest"
"sync/atomic"
"testing"

"github.com/Agent-Field/codeaf/internal/catalog"
"github.com/Agent-Field/codeaf/internal/modelsource"
)

// A PROGRAM'S WINDOW COMES FROM THE CATALOG CODEAF ALREADY KEPT. The listing
// the default service's catalog wrote answers a model in either spelling the
// person uses for it, a model it never listed answers zero, and nothing is
// asked of the network on the way.
func TestCachedContextWindowReadsTheProfilesOwnCatalog(t *testing.T) {
var calls atomic.Int32
server := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {
calls.Add(1)
}))
defer server.Close()
profile := t.TempDir()
t.Setenv(ProfileDirEnv, profile)
t.Setenv("CODEAF_HOME", profile)
// The service's address is the stub, so a lookup that fetched would be
// counted.
t.Setenv("CODEAF_BASE_URL", server.URL+"/api/v1")
if err := catalog.Remember(catalog.Options{BaseURL: server.URL + "/api/v1", Dir: profile}, []catalog.Model{
{ID: "deepseek/deepseek-v4.1-flash", ContextLength: 1_048_576},
{ID: "vendor/small", ContextLength: 16_384},
}); err != nil {
t.Fatal(err)
}
for _, tc := range []struct {
model string
want int
}{
{"deepseek/deepseek-v4.1-flash", 1_048_576},
{"openrouter/deepseek/deepseek-v4.1-flash", 1_048_576},
{"vendor/small", 16_384},
{"vendor/never-listed", 0},
} {
if got := CachedContextWindow(tc.model); got != tc.want {
t.Errorf("CachedContextWindow(%q) = %d, want %d", tc.model, got, tc.want)
}
}
if got := CachedContextWindow(" "); got != 0 {
t.Errorf("a blank model answered %d, want 0", got)
}
if calls.Load() != 0 {
t.Fatalf("the lookup reached the network %d times", calls.Load())
}
}

// A profile that never listed anything answers zero for every model, which
// is "nobody can say" and never a small window.
func TestCachedContextWindowOnAProfileWithNoCatalogIsZero(t *testing.T) {
profile := t.TempDir()
t.Setenv(ProfileDirEnv, profile)
t.Setenv("CODEAF_HOME", profile)
if got := CachedContextWindow("deepseek/deepseek-v4.1-flash"); got != 0 {
t.Fatalf("window = %d, want 0", got)
}
}

// A model on another connected service is answered from THAT service's own
// compartment, found by the prefix the person writes it with, and never from
// the default service's list.
func TestCachedContextWindowReadsAConnectedServicesOwnCompartment(t *testing.T) {
profile := t.TempDir()
t.Setenv(ProfileDirEnv, profile)
t.Setenv("CODEAF_HOME", profile)
t.Setenv("CODEAF_BASE_URL", "")
if err := writeProfileValue(profile, keyModelSources, []PersistedSource{{ID: "z-ai", Written: "z-ai", Key: "zai-key-0123456789", Order: 1}}); err != nil {
t.Fatal(err)
}
var zai modelsource.Connected
for _, service := range ResolveSources(profile, "", "").All() {
if service.Source.ID == "z-ai" {
zai = service
}
}
if zai.Source.ID == "" {
t.Fatal("the z-ai service did not resolve")
}
if err := catalog.Remember(CatalogOptionsFor(zai, profile), []catalog.Model{{ID: "glm-5.1", ContextLength: 200_000}}); err != nil {
t.Fatal(err)
}
if err := catalog.Remember(catalog.Options{Dir: profile}, []catalog.Model{{ID: "glm-5.1", ContextLength: 7}}); err != nil {
t.Fatal(err)
}
if got := CachedContextWindow("z-ai/glm-5.1"); got != 200_000 {
t.Fatalf("z-ai/glm-5.1 = %d, want its own service's 200000", got)
}
}
47 changes: 39 additions & 8 deletions internal/manual/chat/senior-dev.md
Original file line number Diff line number Diff line change
Expand Up @@ -419,8 +419,10 @@ dependencies). `SENIOR_DEV_NET=off` withholds `webfetch` and `websearch` for tha
drops the part of its instructions about installing anything, and makes its copy link
your ignored dependency folders instead of having its own.
Senior-dev also requests model sizes and capabilities from models.dev when its cached
catalog is absent or stale. If that site cannot be reached, the run uses conservative
model limits and still calls models through codeaf's loopback API.
catalog is absent or stale. If that site cannot be reached, or has never listed the model,
it reads the model's window from the model catalog codeaf keeps for your profile, which
needs no network. Only when neither knows the model does it assume 16,384 tokens, and it
says so. Every model call still goes through codeaf's loopback API.

## senior-dev and missing dependencies — pytest not installed, No module named, pip refuses, npm install, a virtual environment

Expand Down Expand Up @@ -965,10 +967,12 @@ with kimi-k2.6", or several: "with kimi-k2.6 and deepseek-v4-pro" — and senior
with exactly those, routing among them call by call when there are several; the card and
the task's first line name them. A name that fits more than one model is put to you to
settle. A model none of your connected services can serve is refused before the card, by
name, rather than swapped for another. When a catalog was loaded, a model it cannot size cannot be used: the
run ends before its first call with `senior-dev cannot work with <model>: …`, and nothing
is spent. If models.dev is unavailable and there is no cache, conservative limits let
the run start. The models are fixed when the run starts; changing the crew later does not move
name, rather than swapped for another. When a catalog was loaded, a model neither it nor
codeaf's own model catalog can size cannot be used: the run ends before its first call with
`senior-dev cannot work with <model>: …`, and nothing is spent. A model you asked for
that is known to hold 32,768 tokens or fewer is refused the same way (see the section on
a model too small for senior-dev). If models.dev is unavailable and codeaf's catalog does not know the model
either, the run starts on an assumed 16,384 tokens and says so. The models are fixed when the run starts; changing the crew later does not move
a run already working. `/senior-dev` typed with a brief uses your crew.

**Otherwise, from the chat it uses your crew.** codeaf hands senior-dev the worker
Expand All @@ -982,8 +986,35 @@ planner and checker routing and `/crew`'s per-task limit apply to codeaf's own
tasks, not senior-dev's run. Change the crew and the next run follows. The
mastermind (brain) model is not used: every call senior-dev makes is either its
work or a history summary.
A crew model senior-dev's model catalog cannot size is left out, and its log says so;
if that leaves no working model, it uses its own list instead.
A crew model senior-dev's model catalog cannot size is left out, and its log says so,
and so is one known to hold 32,768 tokens or fewer; if that leaves no working model, it
uses its own list instead.

## senior-dev keeps compacting, or refused a model as too small — how much its model can hold, a 32K model, could not learn how much its model holds

senior-dev keeps its brief, its tools, its checklist and its progress in its model's
window, and compacts its history once it fills 60% of what the model can take in. It learns the window from models.dev
first, then from the model catalog codeaf keeps for your profile, which it reads from
disk with no network. A shell run waits up to 15 seconds for codeaf to list its models
first on a profile that has never listed them.

**A model you asked for that is known to hold 32,768 tokens or fewer is refused before
its first call**, and nothing is spent: `senior-dev cannot work with <model> (16,384
tokens): a run needs a model that holds more than 32,768 tokens to keep its brief, its
tools and its progress in view, so nothing was started; ask for a model with a larger
window`. Ask for a larger model. **A crew model that small is left out instead**, with a
note in the run's log; if that leaves no model for its work, it uses its own list.

**When nothing knows the model, it assumes 16,384 tokens and runs.** Its page says
`could not learn how much its model holds; assumed 16,384 tokens`, and its log says
which model. A run on that guess compacts every few steps. A real window that small
is rare, so the guess is never refused. To give it the real figure, run `codeaf models`
once on that machine while it can reach your model service; codeaf keeps the list it
prints, and senior-dev reads the window from it.

When a compacted history still does not fit, senior-dev keeps as much of its newest
summary as fits beside the brief and the list of changed files. Only when none of the
summary fits does the list of changed files carry over on its own.

## Which model is my senior-dev run on — the models it was launched with, the foot of its task page

Expand Down
3 changes: 3 additions & 0 deletions internal/manual/chat_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -1101,6 +1101,9 @@ func TestTheChatManualAnswersTheQuestionsPeopleAsk(t *testing.T) {
{"who writes the commit message for senior-dev's work", "senior-dev"},
{"how much does a senior-dev run cost", "senior-dev"},
{"which model is my senior-dev run using", "senior-dev"},
{"why does senior-dev keep compacting every few steps", "senior-dev"},
{"senior-dev refused my model as too small", "senior-dev"},
{"senior-dev says it could not learn how much its model holds", "senior-dev"},
{"how do I read the whole brief senior-dev was given", "senior-dev"},
{"scroll senior-dev's brief", "senior-dev"},
{"senior-dev finished my new file but it is not on the branch", "senior-dev"},
Expand Down
9 changes: 9 additions & 0 deletions internal/seniordev/actions.go
Original file line number Diff line number Diff line change
Expand Up @@ -275,6 +275,15 @@ func presentStage(stage, status string, facts stageFacts) (delegate.Shown, bool)
text = fmt.Sprintf("learned its model holds %s tokens", thousands(limit))
}
return delegate.Shown{Text: text}, true
case "compaction-capacity/guessed":
// Neither catalog could size the model, so the run started on the
// guess (app/window_check.go); a person seeing it compact every few
// steps is told why on the first line it writes.
text := "could not learn how much its model holds"
if limit := facts.whole("limit_tokens"); limit > 0 {
text += fmt.Sprintf("; assumed %s tokens", thousands(limit))
}
return delegate.Shown{Text: text}, true
case "compaction/summarized", "compaction/fallback":
shown := delegate.Shown{Text: "compacted its memory", Memory: true}
if status == "fallback" {
Expand Down
Loading
Loading