From 7f9a157db46915c7fc192535e208411dd2ce1c61 Mon Sep 17 00:00:00 2001 From: Pete Cornish Date: Sun, 13 Sep 2026 00:43:29 +0100 Subject: [PATCH 1/4] feat: add the fleet gateway, a fleet's OpenAI-compatible front door --- .github/workflows/ci.yml | 16 + cmd/spinloop/commands.go | 1 + cmd/spinloop/fleet.go | 24 +- cmd/spinloop/gateway.go | 202 ++++ cmd/spinloop/gateway_test.go | 276 +++++ cmd/spinloop/main.go | 15 + cmd/spinloop/remote.go | 37 +- cmd/spinloop/route.go | 45 +- cmd/spinloop/route_test.go | 211 +++- cmd/spinloop/serve_daemon.go | 2 +- cmd/spinloop/serve_daemon_test.go | 71 ++ cmd/spinloop/status_render.go | 18 +- cmd/spinloop/status_render_test.go | 30 + docs/README.md | 6 +- docs/commands/fleet.md | 23 + docs/commands/gateway.md | 149 +++ docs/env-vars.md | 2 +- docs/openapi.yaml | 16 +- docs/spinloop-file.md | 24 +- examples/gateway-docker/.env.example | 12 + examples/gateway-docker/Dockerfile | 60 + examples/gateway-docker/README.md | 153 +++ examples/gateway-docker/client/Spinloop | 13 + examples/gateway-docker/compose.yaml | 63 + .../gateway-docker/engine/engine-config.yaml | 79 ++ examples/gateway-docker/engine/gate.js | 53 + examples/gateway-docker/fleet.yaml | 36 + .../gateway-docker/gateway/fleet-cold.yaml | 22 + examples/gateway-docker/gateway/fleet.yaml | 32 + examples/gateway-docker/node/Spinloop | 14 + examples/gateway-docker/run-tests.sh | 621 ++++++++++ examples/gateway-docker/shim/llama-server | 46 + internal/daemon/daemon.go | 68 +- internal/daemon/daemon_test.go | 12 +- internal/daemon/readiness_test.go | 2 +- internal/fleet/config.go | 39 + internal/fleet/config_test.go | 38 + internal/fleet/remote_node.go | 31 +- internal/fleet/remote_node_test.go | 53 +- internal/fleet/select.go | 107 +- internal/fleet/select_test.go | 98 ++ internal/fleet/wake.go | 116 +- internal/fleet/wake_test.go | 131 ++- internal/gateway/gateway.go | 502 ++++++++ internal/gateway/gateway_test.go | 1020 +++++++++++++++++ internal/remote/remote.go | 23 +- .../.openspec.yaml | 2 + .../2026-09-12-add-fleet-gateway/design.md | 248 ++++ .../2026-09-12-add-fleet-gateway/proposal.md | 124 ++ .../specs/daemon-api/spec.md | 24 + .../specs/fleet-config/spec.md | 41 + .../specs/fleet-gateway/spec.md | 389 +++++++ .../specs/fleet-routing/spec.md | 199 ++++ .../specs/remote-node/spec.md | 59 + .../2026-09-12-add-fleet-gateway/tasks.md | 43 + openspec/specs/daemon-api/spec.md | 25 + openspec/specs/fleet-config/spec.md | 42 + openspec/specs/fleet-gateway/spec.md | 390 +++++++ openspec/specs/fleet-routing/spec.md | 57 +- openspec/specs/remote-node/spec.md | 24 +- remote/lambda/shared/daemon.ts | 5 + remote/lambda/start/index.ts | 55 +- remote/test/start-status.test.ts | 73 ++ 63 files changed, 6281 insertions(+), 131 deletions(-) create mode 100644 cmd/spinloop/gateway.go create mode 100644 cmd/spinloop/gateway_test.go create mode 100644 docs/commands/gateway.md create mode 100644 examples/gateway-docker/.env.example create mode 100644 examples/gateway-docker/Dockerfile create mode 100644 examples/gateway-docker/README.md create mode 100644 examples/gateway-docker/client/Spinloop create mode 100644 examples/gateway-docker/compose.yaml create mode 100644 examples/gateway-docker/engine/engine-config.yaml create mode 100644 examples/gateway-docker/engine/gate.js create mode 100644 examples/gateway-docker/fleet.yaml create mode 100644 examples/gateway-docker/gateway/fleet-cold.yaml create mode 100644 examples/gateway-docker/gateway/fleet.yaml create mode 100644 examples/gateway-docker/node/Spinloop create mode 100755 examples/gateway-docker/run-tests.sh create mode 100755 examples/gateway-docker/shim/llama-server create mode 100644 internal/gateway/gateway.go create mode 100644 internal/gateway/gateway_test.go create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/.openspec.yaml create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/design.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/proposal.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/daemon-api/spec.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-config/spec.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-gateway/spec.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-routing/spec.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/remote-node/spec.md create mode 100644 openspec/changes/archive/2026-09-12-add-fleet-gateway/tasks.md create mode 100644 openspec/specs/fleet-gateway/spec.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d87f3d5f..4a1c25f4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -90,3 +90,19 @@ jobs: - name: Fleet integration test run: examples/fleet-docker/run-tests.sh + + # The dockerised gateway example, driven end to end: real daemons, a real + # gateway process selecting, waking and keying over the network. Same + # double duty as the fleet job — coverage and example at once. + gateway-integration: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + + - uses: actions/setup-go@v7 + with: + go-version-file: go.mod + cache: true + + - name: Gateway integration test + run: examples/gateway-docker/run-tests.sh diff --git a/cmd/spinloop/commands.go b/cmd/spinloop/commands.go index 19b643d4..5ee4ff7c 100644 --- a/cmd/spinloop/commands.go +++ b/cmd/spinloop/commands.go @@ -66,6 +66,7 @@ harness could be configured with, spinloop show what it has been.`, serveCmd(), upCmd(), daemonCmd(), + gatewayCmd(), exportCmd(), hfCmd(), initProvidersCmd(), diff --git a/cmd/spinloop/fleet.go b/cmd/spinloop/fleet.go index 67d9a865..d437c3e2 100644 --- a/cmd/spinloop/fleet.go +++ b/cmd/spinloop/fleet.go @@ -98,6 +98,7 @@ func fleetRow(r fleet.NodeResult) (state, serving string) { UptimeSeconds: r.Status.UptimeSeconds, LastActiveAt: r.Status.LastActiveAt, IdleSeconds: r.Status.IdleSeconds, + Ready: r.Status.Ready, } return f.State, f.servingText() } @@ -777,9 +778,17 @@ func runFleetRoute(path, node, prefer string, args []string) error { resolvedPath) } if isEndpoint(target) { - return fmt.Errorf( - "FLEET %s names an endpoint, and gateway routing is not implemented yet: "+ - "name a fleet file to choose a node from", target) + // The endpoint has already done the choosing: no fleet file to read, + // no node to query, nothing to start. Say where a launch would point + // the agent. + if sel.BaseURL != "" { + fmt.Printf("This Spinloop pins BASEURL %s, so a launch would not route at all.\n", sel.BaseURL) + return nil + } + fmt.Printf("Spinloop: %s\nFleet: %s (an endpoint, not a fleet file)\n\n", resolvedPath, target) + fmt.Printf("The endpoint has already chosen: a launch would point the agent at %s.\n", endpointBaseURL(target)) + fmt.Println("No node is queried, and nothing is started.") + return nil } cfg, err := fleet.Resolve(resolveFleetPath(target, fromFlag, resolvedPath)) if err != nil { @@ -820,8 +829,13 @@ func runFleetRoute(path, node, prefer string, args []string) error { fmt.Printf("\nA launch could not start one either: %v\n", dcErr) return nil } - if wake, ok := cfg.WouldWake(none.Results, dc); ok { - fmt.Printf("\nA launch would wake %s and wait for its engine. Nothing has been started.\n", wake.Name) + if wake, ok := cfg.WouldWake(none.Results, fleet.ConstantConfig(dc, dcErr)); ok { + if cfg.Wakes() { + fmt.Printf("\nA launch would wake %s and wait for its engine. Nothing has been started.\n", wake.Name) + } else { + fmt.Printf("\nA launch would refuse: wake is off in %s. Start %s with `spinloop fleet start %s`. Nothing has been started.\n", + cfg.Path, wake.Name, wake.Name) + } return nil } fmt.Println("\nNo node could be woken for it either. Nothing has been started.") diff --git a/cmd/spinloop/gateway.go b/cmd/spinloop/gateway.go new file mode 100644 index 00000000..8a1b551d --- /dev/null +++ b/cmd/spinloop/gateway.go @@ -0,0 +1,202 @@ +// spinloop gateway: the fleet's OpenAI-compatible front door. It is the fleet +// client wearing a server — selection, waking, endpoint resolution, and key +// handling all come from internal/fleet, and the HTTP surface lives in +// internal/gateway, so this command resolves the fleet file, resolves its own +// token the way the daemon's is resolved, and serves. + +package main + +import ( + "context" + "fmt" + "net" + "net/http" + "os" + "os/signal" + "syscall" + "time" + + "github.com/spinloop-ai/spinloop/internal/fleet" + "github.com/spinloop-ai/spinloop/internal/gateway" + "github.com/spinloop-ai/spinloop/internal/remote" + + "github.com/spf13/cobra" + "github.com/spf13/pflag" +) + +func gatewayCmd() *cobra.Command { + var fleetPath, listen, apiToken, apiTokenFile string + var wakeTimeout time.Duration + var loopback bool + c := &cobra.Command{ + Use: "gateway", + Short: "serve the fleet under one OpenAI-compatible endpoint", + Long: `runs in the foreground, the way spinloop serve does: it holds the +fleet file it serves (a --fleet path, or ./fleet.yaml), answers +/v1/models and completion requests by choosing a node with the fleet's +own selector, and wakes a node when nothing is serving what a request +asks for, holding the request until the engine answers. It needs the +same environment a machine running spinloop fleet start would: the +tokens the fleet file names, set here or in the .env beside it. + +A Spinloop points an agent at it with a FLEET that names its address: + + FLEET http://gateway.internal:4000 + +The agent then needs only the gateway's token, as OPENAI_API_KEY.`, + Args: cobra.NoArgs, + SilenceErrors: true, + SilenceUsage: true, + RunE: func(c *cobra.Command, args []string) error { + resolve(c) + return runGatewayCommand(fleetPath, listen, apiToken, apiTokenFile, wakeTimeout, loopback, c.Flags()) + }, + } + fs := c.Flags() + fs.StringVarP(&fleetPath, "fleet", "f", "", fleetFileUsage) + fs.StringVar(&listen, "listen", gateway.DefaultListen, "the address to listen on") + fs.BoolVarP(&loopback, "loopback", "l", false, "bind the gateway to loopback on the default port ("+gateway.LoopbackListen+"); needs no token") + fs.StringVar(&apiTokenFile, "api-token-file", "", "read the gateway's bearer token from this file") + fs.StringVar(&apiToken, "api-token", "", "the gateway's bearer token") + fs.DurationVar(&wakeTimeout, "wake-timeout", 0, "how long to wait for a woken engine to answer") + compRegister(c, "fleet", compFiles) + return c +} + +// cmdGateway runs the command through the tree — the seam the suite calls. +func cmdGateway(args []string) error { return execCmd(gatewayCmd(), args) } + +// runGatewayCommand is the body of `spinloop gateway`: the server, and the +// signal handling that shuts it down cleanly. +func runGatewayCommand(fleetPath, listen, apiToken, apiTokenFile string, wakeTimeout time.Duration, loopback bool, flags *pflag.FlagSet) error { + // Whether --listen was typed at all, not whether it differs from the + // default: --listen :4000 --loopback is still a conflict, and a + // compare-against-default check would let it pass. + listenExplicit := flags.Changed("listen") + listen, err := gatewayListenAddr(listen, listenExplicit, loopback) + if err != nil { + return err + } + // The override lives as long as the server serves, not as long as the + // setup takes. + var restore func() + if wakeTimeout > 0 { + prev := fleet.WakeTimeout + fleet.WakeTimeout = wakeTimeout + restore = func() { fleet.WakeTimeout = prev } + defer restore() + } + srv, ln, err := newGatewayServer(fleetPath, listen, apiToken, apiTokenFile) + if err != nil { + return err + } + defer ln.Close() + + // The handler goes in before a signal can arrive, so a signal at any point + // from here on shuts the server down rather than killing the process. + sigCh := make(chan os.Signal, 1) + signal.Notify(sigCh, syscall.SIGINT, syscall.SIGTERM) + defer signal.Stop(sigCh) + go srv.Serve(ln) + + // Foreground until signalled, the way the daemon waits: the signal is the + // only exit, so the command returns nil when a clean shutdown ends Serve's + // http.ErrServerClosed with it. + <-sigCh + ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second) + srv.Shutdown(ctx) + cancel() + return nil +} + +// gatewayListenAddr resolves the address the gateway listens on: --loopback +// takes the default, and an explicit address and --loopback are two answers +// to one question. +func gatewayListenAddr(listen string, listenExplicit, loopback bool) (string, error) { + if loopback && listenExplicit { + return "", fmt.Errorf("--loopback and --listen both given: pass one") + } + if loopback { + return gateway.LoopbackListen, nil + } + return listen, nil +} + +// newGatewayServer resolves the fleet file and the gateway's token, checks the +// file's token references the way a startup must, opens the listener, and +// prints the address a Spinloop names in its FLEET. Everything that can fail +// without serving fails here, before a listener exists. +func newGatewayServer(fleetPath, listen, apiToken, apiTokenFile string) (*http.Server, net.Listener, error) { + cfg, err := fleet.Resolve(fleetPath) + if err != nil { + return nil, nil, err + } + token, err := daemonToken(apiToken, apiTokenFile) + if err != nil { + return nil, nil, err + } + logger, err := commandLogger("") + if err != nil { + return nil, nil, err + } + + // The gateway's own startup failures are the fleet file's: a token variable + // it names that is set nowhere fails here, naming the node, rather than + // surfacing later as a per-request authentication failure. + for _, entry := range cfg.Nodes { + if _, err := cfg.Token(entry); err != nil { + return nil, nil, err + } + if entry.Kind == fleet.KindRemote { + if _, err := cfg.RemoteEngineToken(entry); err != nil { + return nil, nil, err + } + } else if _, err := cfg.EngineToken(entry); err != nil { + return nil, nil, err + } + } + + // The gateway is the client that wakes a node, so it resolves what each + // node runs the way `spinloop fleet start` does: the node's own source, + // never a config invented for the request. + cfgFor := func(entry fleet.NodeConfig) (remote.DeployConfig, error) { + arg, _, err := resolveNodeSpinloop(entry, cfg.Dir) + if err != nil { + return remote.DeployConfig{}, err + } + sel, path, err := readSpinloop("the Spinloop of node "+entry.Name, arg) + if err != nil { + return remote.DeployConfig{}, err + } + if err := applySpinloopEnv(sel, path); err != nil { + return remote.DeployConfig{}, err + } + return deployConfigForNode(sel, path) + } + + h := gateway.New(cfg, token, gateway.Options{ConfigFor: cfgFor, Log: logger}) + ln, err := gateway.Listen(listen, token) + if err != nil { + return nil, nil, err + } + + fmt.Printf("Gateway for %s is listening on %s\n", cfg.Path, ln.Addr().String()) + fmt.Printf("Name %s in a Spinloop's FLEET\n\n", fleetURL(ln.Addr().String())) + return &http.Server{Handler: h}, ln, nil +} + +// fleetURL turns the address the gateway listens on into the value a Spinloop +// names in its FLEET: an http URL the agent's machine can reach. The host it +// can know is the one it was told to bind; for a wildcard bind the host is +// whatever this machine is called from the other side, which only the operator +// knows. +func fleetURL(listenAddr string) string { + host, port, err := net.SplitHostPort(listenAddr) + if err != nil { + return listenAddr + } + if host == "" || host == "0.0.0.0" || host == "::" || host == "[::]" { + host = "" + } + return "http://" + net.JoinHostPort(host, port) +} diff --git a/cmd/spinloop/gateway_test.go b/cmd/spinloop/gateway_test.go new file mode 100644 index 00000000..8bf82891 --- /dev/null +++ b/cmd/spinloop/gateway_test.go @@ -0,0 +1,276 @@ +package main + +import ( + "fmt" + "io" + "net" + "net/http" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/spinloop-ai/spinloop/internal/gateway" +) + +// The gateway starts with the fleet file it serves and answers, and its +// banner names the address a Spinloop puts in its FLEET. +func TestGatewayStartsAndAnswers(t *testing.T) { + isolateConfig(t) + t.Setenv("SPINLOOP_API_TOKEN", "") + node := newRoutableNode(t, "qwen3-27b", true, 300) + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + t.Chdir(dir) + + var srv *http.Server + var ln net.Listener + out := captureStdout(t, func() { + var err error + srv, ln, err = newGatewayServer("", "127.0.0.1:0", "", "") + if err != nil { + t.Fatal(err) + } + }) + defer ln.Close() + go srv.Serve(ln) + + port := ln.Addr().(*net.TCPAddr).Port + if !strings.Contains(out, "http://127.0.0.1:"+fmt.Sprint(port)) { + t.Errorf("the banner should name the address to put in a Spinloop's FLEET, got:\n%s", out) + } + + client := &http.Client{Timeout: 2 * time.Second} + resp, err := client.Get(fmt.Sprintf("http://127.0.0.1:%d/health", port)) + if err != nil { + t.Fatalf("health: %v", err) + } + if resp.StatusCode != http.StatusOK { + t.Errorf("health answered %d", resp.StatusCode) + } + + // The printed address is the one a Spinloop's FLEET names: asking it for + // the fleet's models gets the running node's. + resp, err = client.Get(fmt.Sprintf("http://127.0.0.1:%d/v1/models", port)) + if err != nil { + t.Fatalf("models: %v", err) + } + if resp.StatusCode != http.StatusOK { + t.Errorf("models answered %d", resp.StatusCode) + } + body, _ := io.ReadAll(resp.Body) + if !strings.Contains(string(body), "qwen3-27b") { + t.Errorf("the running node's model should be listed, got: %s", body) + } +} + +// A missing fleet file fails naming the expected path, and nothing listens. +func TestGatewayFailsWithoutAFleetFile(t *testing.T) { + t.Chdir(t.TempDir()) + _, ln, err := newGatewayServer("", "127.0.0.1:0", "", "") + if err == nil { + t.Fatal("a gateway with no fleet file should fail") + } + if ln != nil { + t.Error("a failing gateway opened a listener") + } + if !strings.Contains(err.Error(), "fleet.yaml") { + t.Errorf("the failure should name the expected path, got: %v", err) + } +} + +// A fleet file naming a token variable set nowhere fails at startup, naming +// the node and the variable, rather than listening and failing per request. +func TestGatewayFailsOnAnUnsetTokenVariable(t *testing.T) { + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n - name: gated\n host: 127.0.0.1\n port: 14242\n tokenEnv: GW_NODE_TOKEN_UNSET\n") + t.Chdir(dir) + + _, ln, err := newGatewayServer("", "127.0.0.1:0", "", "") + if err == nil { + t.Fatal("an unset token variable should fail the gateway at startup") + } + if ln != nil { + t.Error("a failing gateway opened a listener") + } + for _, want := range []string{"gated", "GW_NODE_TOKEN_UNSET"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the failure should name %q, got: %v", want, err) + } + } +} + +// Two token sources at once is a conflict, resolved the way the daemon's is. +func TestGatewayTokenSourcesConflict(t *testing.T) { + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n - name: n\n host: 127.0.0.1\n port: 14242\n") + t.Chdir(dir) + + tokenFile := filepath.Join(t.TempDir(), "token") + mustWrite(t, tokenFile, "from-file\n") + _, _, err := newGatewayServer("", "127.0.0.1:0", "literal", tokenFile) + if err == nil { + t.Fatal("two token sources should be a conflict") + } + if !strings.Contains(err.Error(), "both given") { + t.Errorf("the conflict should name both sources, got: %v", err) + } +} + +// A tokenless non-loopback listen is refused at startup, naming every way a +// token can be supplied. +func TestGatewayRefusesTokenlessNonLoopback(t *testing.T) { + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n - name: n\n host: 127.0.0.1\n port: 14242\n") + t.Chdir(dir) + t.Setenv("SPINLOOP_API_TOKEN", "") + + _, ln, err := newGatewayServer("", "0.0.0.0:0", "", "") + if err == nil { + t.Fatal("a tokenless non-loopback gateway should refuse to start") + } + if ln != nil { + t.Error("a refusing gateway opened a listener") + } + for _, want := range []string{"--api-token-file", "SPINLOOP_API_TOKEN", "--api-token"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal should name %q, got: %v", want, err) + } + } +} + +func TestGatewayListenAddr(t *testing.T) { + tests := []struct { + name string + listen string + explicit bool + loopback bool + want string + wantErr bool + }{ + {name: "neither", listen: gateway.DefaultListen, want: gateway.DefaultListen}, + {name: "typed address alone", listen: "10.0.0.5:9999", explicit: true, want: "10.0.0.5:9999"}, + {name: "loopback replaces the default", listen: gateway.DefaultListen, loopback: true, want: gateway.LoopbackListen}, + {name: "loopback and a typed address conflict", listen: "10.0.0.5:9999", explicit: true, loopback: true, wantErr: true}, + {name: "loopback and the repeated default conflict", listen: gateway.DefaultListen, explicit: true, loopback: true, wantErr: true}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, err := gatewayListenAddr(tt.listen, tt.explicit, tt.loopback) + if (err != nil) != tt.wantErr { + t.Fatalf("gatewayListenAddr(%q, %v, %v) error = %v, wantErr %v", tt.listen, tt.explicit, tt.loopback, err, tt.wantErr) + } + if !tt.wantErr && got != tt.want { + t.Fatalf("gatewayListenAddr = %q, want %q", got, tt.want) + } + }) + } +} + +// TestCmdGateway_LoopbackConflictsWithExplicitListen covers the rule through +// the real flag parsing, so the -l spelling and a --listen typed to the +// default's own value are both counted as explicit. The conflict is detected +// by fs.Changed, which is order-independent — the last case types the address +// first, pinning that a sequential rewrite can't make the rule depend on flag +// position. +func TestCmdGateway_LoopbackConflictsWithExplicitListen(t *testing.T) { + for _, args := range [][]string{ + {"--loopback", "--listen", "127.0.0.1:0"}, + {"--loopback", "--listen", gateway.DefaultListen}, + {"-l", "--listen", gateway.DefaultListen}, + {"--listen", "127.0.0.1:0", "--loopback"}, + } { + isolateConfig(t) + t.Setenv("SPINLOOP_API_TOKEN", "") + t.Chdir(t.TempDir()) + err := cmdGateway(args) + if err == nil || !strings.Contains(err.Error(), "--loopback") || !strings.Contains(err.Error(), "--listen") { + t.Fatalf("cmdGateway(%v) = %v, want a conflict naming both flags", args, err) + } + } +} + +// TestCmdGateway_LoopbackBindsLoopback checks the shorthand end to end: the +// gateway binds gateway.LoopbackListen and answers unauthenticated, because a +// loopback listen needs no token. The port is fixed, so the test declines +// rather than fights one — a developer in this repo often has a real gateway +// on it, and the rest of the suite never binds a fixed port for the same +// reason. +func TestCmdGateway_LoopbackBindsLoopback(t *testing.T) { + probe, err := net.DialTimeout("tcp", gateway.LoopbackListen, 500*time.Millisecond) + if err == nil { + probe.Close() + t.Skipf("%s is taken", gateway.LoopbackListen) + } + isolateConfig(t) + t.Setenv("SPINLOOP_API_TOKEN", "") + node := newRoutableNode(t, "qwen3-27b", true, 300) + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + t.Chdir(dir) + + // The banner goes to stdout; wait for it the way the daemon test waits + // for its stderr record. + out := filepath.Join(t.TempDir(), "stdout") + f, err := os.Create(out) + if err != nil { + t.Fatal(err) + } + old := os.Stdout + os.Stdout = f + t.Cleanup(func() { + os.Stdout = old + f.Close() + }) + + done := make(chan error, 1) + go func() { done <- cmdGateway([]string{"--loopback"}) }() + deadline := time.Now().Add(10 * time.Second) + for time.Now().Before(deadline) { + data, _ := os.ReadFile(out) + if strings.Contains(string(data), "listening on "+gateway.LoopbackListen) { + break + } + time.Sleep(20 * time.Millisecond) + } + data, _ := os.ReadFile(out) + if !strings.Contains(string(data), "listening on "+gateway.LoopbackListen) { + t.Fatalf("gateway --loopback did not bind %s; stdout so far:\n%s", gateway.LoopbackListen, data) + } + + // No token was configured anywhere: an unauthenticated health answers. + client := &http.Client{Timeout: 2 * time.Second} + resp, err := client.Get("http://" + gateway.LoopbackListen + "/health") + if err != nil { + t.Fatalf("health: %v", err) + } + if resp.StatusCode != http.StatusOK { + t.Fatalf("unauthenticated health answered %d, want 200", resp.StatusCode) + } + + interruptSelf(t) + select { + case err := <-done: + if err != nil { + t.Fatalf("gateway exited with %v", err) + } + case <-time.After(10 * time.Second): + t.Fatal("gateway did not exit on SIGINT") + } +} + +func TestFleetURL(t *testing.T) { + cases := map[string]string{ + "127.0.0.1:4000": "http://127.0.0.1:4000", + "gw.internal:4000": "http://gw.internal:4000", + ":4000": "http://:4000", + "0.0.0.0:4000": "http://:4000", + "[::]:4000": "http://:4000", + } + for got, want := range cases { + if url := fleetURL(got); url != want { + t.Errorf("fleetURL(%q) = %q, want %q", got, url, want) + } + } +} diff --git a/cmd/spinloop/main.go b/cmd/spinloop/main.go index 420618e7..a03e576c 100644 --- a/cmd/spinloop/main.go +++ b/cmd/spinloop/main.go @@ -1199,6 +1199,21 @@ func applyBeforeLaunch(f spinloopPathFlag, providers string, h harness.Harness, if choice != nil && choice.APIKey != "" { resolve = fleetLaunchResolver(resolve, choice.APIKey) } + if choice != nil && choice.Gateway { + // A FLEET naming an endpoint authenticates with a token the client + // holds itself, resolved the way a key is resolved elsewhere: an ENV + // instruction, else the process environment, else the .env beside the + // Spinloop. Set nowhere, the launch cannot authenticate, and an + // endpoint that refuses every request is not one to point an agent at. + key := localKey(sel, localResolve) + if key == "" { + return spinloop.Selection{}, "", nil, nil, fmt.Errorf( + "no token to reach the FLEET endpoint %s: export %s, or set it in the .env beside %s", + choice.BaseURL, remoteAPIKeyEnv, path) + } + choice.APIKey = key + resolve = fleetLaunchResolver(resolve, key) + } if err := applySelection(sel, h, path, resolve); err != nil { return spinloop.Selection{}, "", nil, nil, err } diff --git a/cmd/spinloop/remote.go b/cmd/spinloop/remote.go index 9ecfbb17..dcab37c7 100644 --- a/cmd/spinloop/remote.go +++ b/cmd/spinloop/remote.go @@ -1242,27 +1242,34 @@ func deployConfigFor(sel spinloop.Selection, spinloopPath string) (remote.Deploy } // deployConfigForNode derives the same config for a machine that already -// exists — a fleet node being woken. Two things differ from a cloud -// deployment, both because the machine is the operator's rather than the +// exists — a fleet node being woken. Three things differ from a cloud +// deployment, all because the machine is the operator's rather than the // deployment's: sizing falls back to the engine's own default, so a Spinloop -// that `spinloop serve` runs happily needs no CONTEXT added merely to be routed; -// and the preset's bind survives, so an engine told to listen on 0.0.0.0 does. +// that `spinloop serve` runs happily needs no CONTEXT added merely to be +// routed; the preset's bind survives, so an engine told to listen on 0.0.0.0 +// does; and the Spinloop's own BASEURL is carried too, so a node with no +// preset does not wake onto the engine's default — llama.cpp's loopback, +// reachable from nobody but the node itself. func deployConfigForNode(sel spinloop.Selection, spinloopPath string) (remote.DeployConfig, error) { return deployConfig(sel, spinloopPath, deployTarget{ - runner: nodeRunnerFor, - owns: isNodeOwned, + runner: nodeRunnerFor, + owns: isNodeOwned, + carriesBaseURL: true, }) } // deployTarget is what the derivation cannot decide for itself: which runners // it accepts, whether a context size is required, which preset flags the -// destination assigns, and whether it fetches the weights itself (and so needs -// companions named). +// destination assigns, whether it fetches the weights itself (and so needs +// companions named), and whether the Spinloop's BASEURL is rendered into the +// engine's bind — a machine the operator owns listens wherever the Spinloop +// says; the cloud assigns its own port, so it keeps the field empty. type deployTarget struct { runner func(provider string) (string, error) requireContext bool seedsWeights bool owns func(key string) bool + carriesBaseURL bool } func deployConfig(sel spinloop.Selection, spinloopPath string, target deployTarget) (remote.DeployConfig, error) { @@ -1371,7 +1378,19 @@ func deployConfig(sel spinloop.Selection, spinloopPath string, target deployTarg dc.Companions = companionsFrom(global, params) } - dc.ServeArgs = preset.Flags(dropOwned(target.owns, global), dropOwned(target.owns, params)) + // The Spinloop's own bind is the final layer, where it travels at all: it + // wins over the preset's host and port exactly as it does for a local + // serve, so a woken engine listens wherever the Spinloop says rather than + // on the engine's default. + layers := [][]preset.Param{dropOwned(target.owns, global), dropOwned(target.owns, params)} + if target.carriesBaseURL { + bind, err := bindAddressParams(sel) + if err != nil { + return dc, err + } + layers = append(layers, bind) + } + dc.ServeArgs = preset.Flags(layers...) if dc.ServeArgs == nil { dc.ServeArgs = []string{} } diff --git a/cmd/spinloop/route.go b/cmd/spinloop/route.go index 280c3ca7..25c5daca 100644 --- a/cmd/spinloop/route.go +++ b/cmd/spinloop/route.go @@ -10,6 +10,7 @@ import ( "context" "errors" "fmt" + "net/url" "os" "path/filepath" "strings" @@ -62,12 +63,16 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp return nil, nil } // A FLEET naming a URL is the gateway shape: it has already done the - // choosing. Parsing accepts it so the eventual gateway needs no new - // keyword; nothing here can act on it yet. + // choosing, so there is no fleet file to read and no node to contact. The + // token is not resolved here: applyBeforeLaunch resolves it through the + // same chain the launch uses, so a missing value fails before anything is + // written. if isEndpoint(target) { - return nil, fmt.Errorf( - "FLEET %s names an endpoint, and gateway routing is not implemented yet: "+ - "name a fleet file to choose a node from", target) + return &fleet.Choice{ + Gateway: true, + BaseURL: endpointBaseURL(target), + Reason: "FLEET names an endpoint", + }, nil } cfg, err := fleet.Resolve(resolveFleetPath(target, opts.fleetPath != "", spinloopPath)) @@ -115,10 +120,22 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp if opts.noWake { return nil, fmt.Errorf("%w\nStart one with `spinloop fleet start `, or drop --no-wake to have spinloop do it", err) } + if !cfg.Wakes() { + // The fleet file says the machines are not to be started on demand. The + // refusal still names the node that would have been woken, the way a + // --no-wake refusal does: the setting decides whether to wake, not what + // would be woken. + if wake, ok := cfg.WouldWake(none.Results, fleet.ConstantConfig(dc, dcErr)); ok { + return nil, fmt.Errorf( + "%w\nwake is off in %s: start %s with `spinloop fleet start %s`", + err, cfg.Path, wake.Name, wake.Name) + } + return nil, fmt.Errorf("%w\nwake is off in %s", err, cfg.Path) + } if dcErr != nil { return nil, fmt.Errorf("%w\nand this Spinloop cannot be turned into something to start: %v", err, dcErr) } - choice, err = cfg.Wake(ctx, want, dc, none.Results, func(format string, args ...any) { + choice, err = cfg.Wake(ctx, want, fleet.ConstantConfig(dc, dcErr), none.Results, func(format string, args ...any) { fmt.Fprintf(os.Stderr, format, args...) }) if err != nil { @@ -131,9 +148,25 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp // announceChoice names the node a launch landed on before the agent starts, so // an unexpected route says so at the time rather than at the first request. func announceChoice(c *fleet.Choice) { + if c.Gateway { + fmt.Fprintf(os.Stderr, "Routing at the FLEET endpoint %s\n", c.BaseURL) + return + } fmt.Fprintf(os.Stderr, "Using %s at %s — %s\n", c.Node.Name, c.BaseURL, c.Reason) } +// endpointBaseURL is the address a launch gives an agent for a FLEET that names +// an endpoint: the value as given when it carries a path, and the OpenAI- +// compatible prefix added when it does not, so FLEET http://gw:4000 points the +// agent at http://gw:4000/v1. +func endpointBaseURL(target string) string { + u, err := url.Parse(target) + if err != nil || (u.Path != "" && u.Path != "/") { + return target + } + return strings.TrimRight(target, "/") + "/v1" +} + // resolveFleetPath resolves a fleet file's path. A relative FLEET is resolved // against the Spinloop that names it, the same rule PRESET and REMOTE follow — an // Spinloop and the fleet beside it travel together, and resolving against the diff --git a/cmd/spinloop/route_test.go b/cmd/spinloop/route_test.go index cd3134c4..057ffc28 100644 --- a/cmd/spinloop/route_test.go +++ b/cmd/spinloop/route_test.go @@ -314,20 +314,170 @@ func TestRoutingToARunningNodeToleratesAnUnusableParallel(t *testing.T) { }) } -// A FLEET naming a URL is the gateway shape: it parses, and says plainly that -// it is not implemented rather than being treated as a filename. -func TestFleetURLIsRefusedAsUnimplemented(t *testing.T) { +// A FLEET naming a URL is the gateway shape: the endpoint has already done the +// choosing, so no fleet file is read, no node is contacted, and the node- +// steering flags are inert. A value with no path gets the OpenAI-compatible +// prefix. +func TestFleetURLYieldsTheEndpoint(t *testing.T) { spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } - _, err = routeThroughFleet(sel, path, routeOptions{}) - if err == nil { - t.Fatal("a gateway URL should fail for now") + captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{node: "nobody", prefer: "sideways", noWake: true}) + if err != nil { + t.Fatalf("an endpoint FLEET should not consult any node: %v", err) + } + if !c.Gateway { + t.Fatalf("the choice should mark itself as an endpoint, got %+v", c) + } + if c.BaseURL != "http://gateway.internal:4000/v1" { + t.Errorf("an endpoint without a path gets the prefix, got %s", c.BaseURL) + } + }) +} + +func TestFleetURLWithAPathIsUsedAsGiven(t *testing.T) { + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000/proxy/v1") + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{}) + if err != nil { + t.Fatal(err) + } + if !c.Gateway || c.BaseURL != "http://gateway.internal:4000/proxy/v1" { + t.Errorf("an endpoint carrying a path is used as given, got %+v", c) + } + }) +} + +// A pinned BASEURL wins over an endpoint FLEET, as it wins over a fleet file. +func TestPinnedBaseURLBeatsAnEndpointFleet(t *testing.T) { + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), + "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\nFLEET http://gateway.internal:4000\n") + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + stderr := captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{}) + if err != nil { + t.Fatal(err) + } + if c != nil { + t.Errorf("a pinned BASEURL is not routed, got %+v", c) + } + }) + if !strings.Contains(stderr, "Not routing") { + t.Errorf("spinloop should say it is not routing, got:\n%s", stderr) } - if !strings.Contains(err.Error(), "not implemented yet") { - t.Errorf("error should say it is not implemented, got: %v", err) +} + +// stubHarnessBinaryWithEnv is stubHarnessBinary plus a dump of the two +// variables a routed launch injects, for asserting what the agent actually +// got. +func stubHarnessBinaryWithEnv(t *testing.T, argsFile, envFile string) { + t.Helper() + dir := t.TempDir() + body := "#!/bin/sh\n" + + "printf '%s\\n' \"$@\" > " + argsFile + "\n" + + "printf 'BASE=%s\\nKEY=%s\\n' \"$OPENAI_BASE_URL\" \"$OPENAI_API_KEY\" > " + envFile + "\n" + if err := os.WriteFile(filepath.Join(dir, "opencode"), []byte(body), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) +} + +// A FLEET naming an endpoint points the agent at it: the address gets the +// OpenAI-compatible prefix, and the token is resolved from the client's +// environment the way a key is resolved elsewhere. +func TestLaunchWithEndpointFleetPointsTheAgentAtTheGateway(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "gw-token") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") + captureStdout(t, func() { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("harness was not launched: %v", err) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + out := string(data) + if !strings.Contains(out, "BASE=http://gateway.internal:4000/v1") { + t.Errorf("the agent's base URL should be the endpoint with the prefix, got:\n%s", out) + } + if !strings.Contains(out, "KEY=gw-token") { + t.Errorf("the agent should carry the gateway's token as its key, got:\n%s", out) + } +} + +// A FLEET naming an endpoint with no token anywhere fails before the agent +// launches and before the harness config is written, naming the variable. +func TestLaunchWithEndpointFleetFailsWithoutAToken(t *testing.T) { + home := isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "") + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") + captureStdout(t, func() { + err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}) + if err == nil { + t.Fatal("a launch that cannot authenticate the endpoint should fail") + } + if !strings.Contains(err.Error(), "OPENAI_API_KEY") { + t.Errorf("the failure should name the variable to set, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the agent launched without a token to reach the endpoint") + } + if _, err := os.Stat(filepath.Join(home, ".config", "opencode", "opencode.json")); err == nil { + t.Error("the harness config was written for a launch that could not authenticate") + } +} + +// A fleet that declares wake: off refuses to start anything when nothing is +// serving, and names the node that would have been woken with the command that +// would start it. +func TestWakeOffRefusesNamingTheNode(t *testing.T) { + node := newRoutableNode(t, "", false, 0) + dir := t.TempDir() + fleetPath := fleetFileIn(t, dir, "wake: off\nnodes:\n"+node.entry("idle-box")) + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + captureStderr(t, func() { + _, err := routeThroughFleet(sel, path, routeOptions{}) + if err == nil { + t.Fatal("wake: off with nothing serving should fail") + } + for _, want := range []string{"wake is off", "idle-box", "spinloop fleet start idle-box"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("message should mention %q, got:\n%s", want, err) + } + } + }) + if node.started { + t.Error("wake: off started an engine") } } @@ -456,6 +606,27 @@ func TestCmdFleetRouteExplainsTheChoice(t *testing.T) { } } +// A FLEET naming an endpoint has already chosen: the route says where a launch +// would point the agent, and queries nothing. +func TestCmdFleetRouteAgainstAnEndpoint(t *testing.T) { + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gw.internal:4000") + + out := captureStdout(t, func() { + if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + t.Fatal(err) + } + }) + for _, want := range []string{ + "gw.internal:4000 (an endpoint, not a fleet file)", + "would point the agent at http://gw.internal:4000/v1", + "nothing is started", + } { + if !strings.Contains(out, want) { + t.Errorf("output should mention %q, got:\n%s", want, out) + } + } +} + // The flag lets the two preferences be compared on a live fleet without // editing the file. func TestCmdFleetRoutePreferenceFlagBeatsTheFile(t *testing.T) { @@ -524,6 +695,30 @@ func TestCmdFleetRouteStartsNothing(t *testing.T) { } } +// A fleet that declares wake: off still reports, when nothing is running, the +// node whose source describes the model and the command that would start it — +// but says a launch would refuse. +func TestCmdFleetRouteWakeOffRefusal(t *testing.T) { + node := newRoutableNode(t, "", false, 0) + dir := t.TempDir() + fleetPath := fleetFileIn(t, dir, "wake: off\nnodes:\n"+node.entry("idle-box")) + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + + out := captureStdout(t, func() { + if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + t.Fatal(err) + } + }) + for _, want := range []string{"wake is off", "idle-box", "spinloop fleet start idle-box", "Nothing has been started"} { + if !strings.Contains(out, want) { + t.Errorf("output should mention %q, got:\n%s", want, out) + } + } + if node.started { + t.Error("fleet route started an engine") + } +} + // A Spinloop naming no fleet, with no --fleet, has nothing to report. func TestCmdFleetRouteNeedsAFleet(t *testing.T) { spinloopDir := routedSpinloop(t, "qwen3-27b", "") diff --git a/cmd/spinloop/serve_daemon.go b/cmd/spinloop/serve_daemon.go index b1aaa690..1dbd5034 100644 --- a/cmd/spinloop/serve_daemon.go +++ b/cmd/spinloop/serve_daemon.go @@ -267,7 +267,7 @@ func startSupervisedForeground(sel spinloop.Selection, spinloopPath string, engi if model == "" { model = sel.Alias } - d.SetServed(sel.Provider, model) + d.SetServed(sel.Provider, model, sel.Alias) // A supervised engine gets its metrics endpoint switched on, exactly as // the cloud path does for a deployed one; an engine with no metrics // dialect gets no added switch and the host's series. diff --git a/cmd/spinloop/serve_daemon_test.go b/cmd/spinloop/serve_daemon_test.go index 53ba1ff3..3fde2056 100644 --- a/cmd/spinloop/serve_daemon_test.go +++ b/cmd/spinloop/serve_daemon_test.go @@ -1366,6 +1366,77 @@ ctx-size = 4096 } } +// The Spinloop's own BASEURL travels where the preset's bind does: a woken +// node with no preset must not land on the engine's default bind, which for +// llama.cpp is loopback — reachable from nobody but the node itself. +func TestNodeDeployConfigCarriesTheSpinloopBind(t *testing.T) { + spinloopPath := writeDeploySpinloop(t, + "PROVIDER llamacpp\nMODEL org/model:Q4_K_M\nCONTEXT 4096\nBASEURL http://0.0.0.0:8080/v1\n", "") + sel, _, err := readSpinloop("test", spinloopPath) + if err != nil { + t.Fatal(err) + } + + node, err := deployConfigForNode(sel, spinloopPath) + if err != nil { + t.Fatal(err) + } + args := strings.Join(node.ServeArgs, " ") + for _, want := range []string{"--host 0.0.0.0", "--port 8080"} { + if !strings.Contains(args, want) { + t.Errorf("a node's serve args should carry the Spinloop's bind %q, got: %s", want, args) + } + } + + // The cloud assigns its own bind, so the Spinloop's is not carried there. + cloud, err := deployConfigFor(sel, spinloopPath) + if err != nil { + t.Fatal(err) + } + cloudArgs := strings.Join(cloud.ServeArgs, " ") + for _, unwanted := range []string{"--host", "--port"} { + if strings.Contains(cloudArgs, unwanted) { + t.Errorf("the cloud sets its own bind, so %q should not be carried, got: %s", unwanted, cloudArgs) + } + } +} + +// Where the Spinloop states a bind and the preset states one too, the +// Spinloop's wins — the same precedence a local serve gives its own BASEURL +// over the preset — and the engine is told once, not twice. +func TestNodeDeployConfigBindBeatsThePreset(t *testing.T) { + spinloopPath := writeDeploySpinloop(t, + "PROVIDER llamacpp\nALIAS qwen\nPRESET ./preset.ini\nBASEURL http://0.0.0.0:8080/v1\n", + `[*] +host = 127.0.0.1 +port = 9090 + +[qwen] +hf = org/model:Q4_K_M +ctx-size = 4096 +`) + sel, _, err := readSpinloop("test", spinloopPath) + if err != nil { + t.Fatal(err) + } + + node, err := deployConfigForNode(sel, spinloopPath) + if err != nil { + t.Fatal(err) + } + args := strings.Join(node.ServeArgs, " ") + for _, want := range []string{"--host 0.0.0.0", "--port 8080"} { + if !strings.Contains(args, want) { + t.Errorf("the Spinloop's bind should win, got: %s", args) + } + } + for _, unwanted := range []string{"127.0.0.1", "9090"} { + if strings.Contains(args, unwanted) { + t.Errorf("the preset's bind should be overridden, got: %s", args) + } + } +} + // mtplxNodePreset is an MTPLX-vocabulary preset: long-form keys, the model under // `model`, the window under `context-window`, the cap under // `max-active-requests`, a served name under `model-id`, and a scheduling mode diff --git a/cmd/spinloop/status_render.go b/cmd/spinloop/status_render.go index 04864805..727ba427 100644 --- a/cmd/spinloop/status_render.go +++ b/cmd/spinloop/status_render.go @@ -9,7 +9,11 @@ package main -import "fmt" +import ( + "fmt" + + "github.com/spinloop-ai/spinloop/internal/daemon" +) // statusFact is the shared status view: the facts both status commands have in // common. A command fills it from its own native reply and reads the shared text @@ -22,6 +26,11 @@ type statusFact struct { UptimeSeconds int LastActiveAt string IdleSeconds int + // Ready is the daemon's readiness reading, daemon.ReadyYes or + // daemon.ReadyNo, empty when none applies. Only ReadyNo is rendered: a + // running engine that has answered is the ordinary case and needs no mark, + // and an absent reading is not evidence of anything to report. + Ready string } // servingText is the "what it serves" text: runner and model, then the uptime and @@ -40,6 +49,13 @@ func (f statusFact) servingText() string { serving = f.Runner + " " + serving } } + // A node whose engine process is up but has not answered its health check + // is not servable yet, however long its uptime says it has been running. + // Shown next to the model, before the timings, because it changes what the + // rest of the row means. + if f.Ready == daemon.ReadyNo { + serving += " (not ready)" + } if f.UptimeSeconds > 0 { serving += fmt.Sprintf(" (up %s)", formatDuration(f.UptimeSeconds)) } diff --git a/cmd/spinloop/status_render_test.go b/cmd/spinloop/status_render_test.go index 9ca27e8d..39e42a5e 100644 --- a/cmd/spinloop/status_render_test.go +++ b/cmd/spinloop/status_render_test.go @@ -3,6 +3,8 @@ package main import ( "strings" "testing" + + "github.com/spinloop-ai/spinloop/internal/daemon" ) // The shared status view is where `remote status` and `fleet status` agree on @@ -33,3 +35,31 @@ func TestStatusFactServingTextLeavesOutAbsentFacts(t *testing.T) { t.Errorf("an empty fact should have no serving text, got %q", got) } } + +// A node whose engine is up as a process but has not answered its health check +// is running by state and unusable in fact. Without the mark the row reads as +// serving the model, which is what made a failed request look unexplained. +func TestStatusFactMarksAnEngineThatHasNotAnswered(t *testing.T) { + f := statusFact{ + State: "running", Model: "qwen", Runner: "llamacpp", + UptimeSeconds: 90, Ready: daemon.ReadyNo, + } + got := f.servingText() + if !strings.Contains(got, "(not ready)") { + t.Errorf("servingText %q should mark the engine not ready", got) + } + if strings.Index(got, "(not ready)") > strings.Index(got, "(up 1m 30s)") { + t.Errorf("the mark belongs before the timings it qualifies, got %q", got) + } +} + +// An engine that has answered is the ordinary case and carries no mark, and a +// daemon reporting no reading is unknown rather than not ready. +func TestStatusFactMarksNothingWhenReadyOrUnknown(t *testing.T) { + for _, ready := range []string{daemon.ReadyYes, ""} { + f := statusFact{State: "running", Model: "qwen", Ready: ready} + if got := f.servingText(); got != "qwen" { + t.Errorf("Ready=%q: servingText = %q, want just the model", ready, got) + } + } +} diff --git a/docs/README.md b/docs/README.md index 2355ac9b..587c6f44 100644 --- a/docs/README.md +++ b/docs/README.md @@ -34,8 +34,9 @@ Four words carry the whole tool: - [The HTTP control API](http-api.md) — driving a supervised engine over HTTP, and the [OpenAPI contract](openapi.yaml) for writing a client - [Running a fleet](commands/fleet.md) — one spinloop watching every machine you - run, with a [containerised fleet](../examples/fleet-docker/) you can bring up - on a laptop + run, with a [containerised fleet](../examples/fleet-docker/) and a + [containerised gateway](../examples/gateway-docker/) you can bring up on a + laptop - [Environment variables](env-vars.md) — every variable spinloop reads - [Runnable examples](../examples/) — ready-to-apply Spinloops with walkthroughs - [Deploying your own cloud GPU endpoint](../remote/) — the AWS project behind @@ -62,6 +63,7 @@ Four words carry the whole tool: | [`spinloop up`](commands/up.md) | Start the engine this directory holds: the fleet, or the `Spinloop`'s server | | [`spinloop daemon`](commands/serve.md#the-control-api---api-and-spinloop-daemon) | Supervise an engine over the [control API](http-api.md) | | [`spinloop fleet`](commands/fleet.md) | Observe and drive the engines on every machine you run | +| [`spinloop gateway`](commands/gateway.md) | Serve the fleet under one OpenAI-compatible endpoint | | [`spinloop remote`](commands/remote.md) | Run the model on a cloud GPU that stops when you do | | [`spinloop export`](commands/export.md) | Capture the current setup as a `Spinloop` | | [`spinloop hf`](commands/hf.md) | Write a `Spinloop` for a Hugging Face model, from its page reference | diff --git a/docs/commands/fleet.md b/docs/commands/fleet.md index 2b992c45..939d0039 100644 --- a/docs/commands/fleet.md +++ b/docs/commands/fleet.md @@ -191,6 +191,29 @@ nodes: … override the file for one command, which is the cheap way to see what the other setting would do before committing to it. +### Waking + +`wake` decides whether routing may start an engine on a node that is not +running one: + +```yaml +wake: off # or: on +nodes: … +``` + +- **`on`** (the default, and the behaviour of a file that declares nothing) — + when nothing is serving, spinloop starts a node and waits for its engine to + answer before the agent launches or the request is answered. +- **`off`** — a request nothing is serving fails rather than starting + anything, naming the node that would have been woken and the `spinloop fleet + start ` command that would start it. Use it where the machines are not + to be started on demand — the models are loaded by hand, or someone else + drives the starts. + +An explicit `--no-wake` still refuses to start anything, whatever the file +says; an explicit `spinloop fleet start` does the opposite — it always starts, +because it was asked. + ### Tokens `tokenEnv` names an environment variable; the value is resolved from the diff --git a/docs/commands/gateway.md b/docs/commands/gateway.md new file mode 100644 index 00000000..c50eeb17 --- /dev/null +++ b/docs/commands/gateway.md @@ -0,0 +1,149 @@ +# spinloop gateway + +Serve a [fleet](fleet.md) under one OpenAI-compatible endpoint. The gateway is +the fleet client wearing a server: it holds a `fleet.yaml`, answers +`/v1/models` and completion requests by choosing a node with the fleet's own +selector, and wakes a node when nothing is serving what a request asks for — so +a machine running agents needs nothing but a URL and one token. + +```sh +spinloop gateway # serves ./fleet.yaml on :4000 +spinloop gateway --fleet ./fleet.yaml +spinloop gateway --listen 0.0.0.0:4000 +spinloop gateway --api-token-file /run/secrets/gateway +``` + +It runs in the foreground, the way [`spinloop serve`](serve.md) does: it holds +the fleet file it serves, and a signal shuts it down cleanly. On startup it +resolves the fleet file and its own token, and checks the token references the +file names — a `tokenEnv` or `engineTokenEnv` variable set nowhere fails here, +naming the node, rather than surfacing later as a per-request authentication +failure. It prints the address a Spinloop names in its `FLEET`: + +``` +Gateway for fleet.yaml is listening on [::]:4000 +Name http://:4000 in a Spinloop's FLEET +``` + +The host it can know is the one it was told to bind; for a wildcard bind the +machine's own name is only the operator's to know, so the printed address says +`` and you fill in whatever this machine is called from the other +side. + +## Pointing an agent at it + +A Spinloop names the gateway's address in its `FLEET` — a URL, not a file: + +```dockerfile +PROVIDER llamacpp +MODEL qwen3-27b +FLEET http://gateway.internal:4000 +``` + +The launch reads no fleet file and contacts no node — the endpoint has already +done the choosing — and the agent it launches authenticates with the +gateway's token, as `OPENAI_API_KEY`, resolved the way a key is resolved +elsewhere: an `ENV` instruction, then the process environment, then the `.env` +beside the Spinloop. An agent pointed at the gateway holds exactly that one +credential; the node tokens and engine keys live with the gateway, which +presents them to the nodes and the engines. See +[The `Spinloop` file](../spinloop-file.md#running-the-model-on-another-machine-you-own). + +## What it answers + +| Path | Meaning | +| ---- | ------- | +| `GET /health` | That the gateway is up. It touches no node on purpose — it is how you tell the gateway down from the fleet down. | +| `GET /v1/models` | The OpenAI list of what a request can reach: what the running nodes report (the served name when a node reports one, else the model id), and — when [wake](#waking-a-node) is on — what a stopped node's own source describes, the model a request would start it with. Duplicates once. Nothing reachable is an empty list, not an error. | +| `POST /v1/chat/completions` | Routed to the node serving the request's `model`, the way a launch routes. | +| `POST /v1/completions` | The same, for the completions endpoint. | + +A request naming no `model` is refused saying so, and a path the gateway does +not serve is refused with a `404` naming the ones it does. + +The list is what a request can reach, so it is bounded by what the gateway can +start: a running node contributes only what it reports — a running engine is +never displaced to make room — and a `kind: remote` environment contributes +nothing beyond what it runs, because a request never wakes one. With +`wake: off`, only what is running is listed. Each node's source is read at most +once in a short window, so a poll of the models list is cheap. + +### Routing a request + +A completion request is answered by the fleet's own selection: a node already +running the model wins, ranked by the fleet file's `prefer` with fleet-file +order breaking ties. A node whose engine is bound to loopback without an +[`engine` override](fleet.md#where-a-nodes-engine-answers) is never selected, +and when it is the only match the failure says so rather than holding the +request until the wake timeout. + +The request's body goes out unmodified and streamed replies are flushed as +they are produced, so a `stream: true` request streams through. The caller's +authorisation never travels past the gateway: the engine is reached with the +key its fleet entry names (`engineTokenEnv`, or the fleet-wide `apiKeyEnv` for +a `kind: remote` node), and an ungated engine is reached with none. The reply +the engine gives is the reply the caller gets — the gateway never retries +another node, and an upstream failure reaches the caller as an error naming +the node. + +### Waking a node + +When no running node serves the model and the fleet file's +[`wake`](fleet.md#waking) setting allows it, the gateway starts a node with +the config that node's own Spinloop source resolves to — only nodes whose +source describes the requested model are candidates, and a node whose stored +config already matches is tried first — and holds the request until the engine +answers, bounded by `--wake-timeout` (default 5m). A timeout fails the request +saying so and leaves the engine running, so a slow load is not thrown away. +Concurrent requests for the same model wake at most one engine: a request that +loses the start to the daemon's "already running" answer takes the node the +other one started. + +With `wake: off`, or when no node's source describes the model, a request +nothing is serving fails without starting anything, naming the node and the +`spinloop fleet start ` command that would start it. + +The gateway needs the same environment a machine running +`spinloop fleet start` would: the tokens the fleet file names, set in its +process environment or in a `.env` beside the fleet file. + +## The gateway's token + +Callers present the gateway's token as a bearer token on every request — a +wrong or missing one is a `401`. The token comes from one of three places, the +same rules the [daemon's](serve.md#the-control-api---api-and-spinloop-daemon) +token follows, and giving two at once is an error rather than a silent +precedence: + +| Source | Notes | +| ------ | ----- | +| `--api-token-file ` | The file's contents, trimmed. | +| `SPINLOOP_API_TOKEN` | The environment. | +| `--api-token ` | The token itself — readable by every local user through `ps`, like the daemon's. | + +A non-loopback listen with no token refuses to start, naming the three ways to +supply one; a loopback listen (`--loopback`, or `--listen 127.0.0.1:4000`) +needs none. The default, `:4000`, binds every interface — the shape a gateway +on a shared machine wants, and the reason the token is not optional there. + +## Flags + +| Flag | Meaning | +| ---- | ------- | +| `-f`, `--fleet` | The fleet file to serve (default `./fleet.yaml`) | +| `--listen` | The address to listen on (default `:4000`) | +| `-l`, `--loopback` | Bind to loopback on the default port (`127.0.0.1:4000`); needs no token | +| `--api-token-file` | Read the gateway's bearer token from this file | +| `--api-token` | The gateway's bearer token | +| `--wake-timeout` | How long to wait for a woken engine to answer (default 5m) | + +## See also + +- [`spinloop fleet`](fleet.md) — the file the gateway serves, and the nodes it + drives +- [`spinloop daemon`](serve.md#the-control-api---api-and-spinloop-daemon) — what + each node runs +- [`examples/gateway-docker/`](../../examples/gateway-docker/) — a gateway and + its fleet in containers, with the test suite that asserts all of this +- [The `Spinloop` file](../spinloop-file.md) — the `FLEET` that points an agent + here diff --git a/docs/env-vars.md b/docs/env-vars.md index fe3d3236..c5e4b01d 100644 --- a/docs/env-vars.md +++ b/docs/env-vars.md @@ -13,7 +13,7 @@ from the environment or a `.env` beside the Spinloop — never written into an | `SPINLOOP_ALIAS` | every command that takes a Spinloop path | A name registered with [`spinloop alias`](commands/alias.md), used when the command is given no path. Precedence: the path or alias argument > `SPINLOOP_ALIAS` > `./Spinloop`. It holds a registry name, never a path, and a same-named file in the working directory does not shadow it. It decides *which* Spinloop is the default, not *whether* one is applied — a bare `spinloop harness` still applies nothing, and `spinloop alias` ignores it. | | `SPINLOOP_PROVIDERS` | `list`, `add`, `apply`, … | Path to a `providers.yaml` that overrides the built-in catalogue. Precedence: `--providers` flag > `SPINLOOP_PROVIDERS` > embedded. | | `SPINLOOP_BASE_URL` | `add`, `apply` | Base-URL override for the provider being configured. Precedence: `--base-url`/`-u` > `SPINLOOP_BASE_URL` > the provider's own option var > the catalogue default. | -| `SPINLOOP_API_TOKEN` | `spinloop daemon`, `spinloop serve --api` | Bearer token for the daemon control API. One of three peer sources, alongside `--api-token-file` and `--api-token`; two at once is an error. From a service manager prefer the file form — see [serve](commands/serve.md). A non-loopback API listen without any of them refuses to start. | +| `SPINLOOP_API_TOKEN` | `spinloop daemon`, `spinloop serve --api`, `spinloop gateway` | Bearer token for the daemon control API — and the token a [gateway](commands/gateway.md)'s callers must present. One of three peer sources, alongside `--api-token-file` and `--api-token`; two at once is an error. From a service manager prefer the file form — see [serve](commands/serve.md). A non-loopback listen without any of them refuses to start. | | `SPINLOOP_REMOTE_KEYSTORE` | `spinloop remote auth` | Set to `file` to keep the stored control-plane credential in the owner-only file under the config directory, even where an OS keystore is reachable — the opt-out for a machine whose keystore is locked or unreachable. Unset, the OS keystore is used where available. See [credentials](commands/remote.md#credentials). | | `SPINLOOP_LOG_LEVEL` | `spinloop daemon`, `spinloop serve` | How much spinloop records about the control API and the supervised engine: `debug`, `info` (default), `warn` or `error`. Precedence: `--log-level` flag > `SPINLOOP_LOG_LEVEL` > `info`. An unrecognised value refuses to start rather than falling back to the default. Under `spinloop serve` the `.env` beside the Spinloop can set it; the daemon reads no Spinloop, so there it comes from the environment its service manager gives it. Records go to stderr; see [what gets logged](commands/serve.md#what-gets-logged). | | *(per-node, named by `tokenEnv`)* | `spinloop fleet` | A fleet node's bearer token. `fleet.yaml` names the variable rather than holding the value; it resolves from the environment, then the `.env` beside the fleet file. See [fleet](commands/fleet.md). | diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 2aca8c5c..2286b28c 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -298,6 +298,12 @@ components: model: type: string description: What it is serving, when known. + servedName: + type: string + description: | + The name the engine answers to — the served name its deploy config + or Spinloop set — reported beside `model` when set. An aliased + engine answers to both, and a caller may know either. uptimeSeconds: type: integer description: How long the engine has been running. Zero unless running. @@ -344,9 +350,17 @@ components: purpose: the daemon knows its engine binds `127.0.0.1:8080`, which is useless to anyone else, and it cannot know the name a client reaches this host by — a LAN name, a tailscale name, a published container - port. The caller composes these against the host it already has. + port. The caller composes these against the host it already has. A node + that does know that name — a remote environment, whose control plane + publishes the instance's address — reports it in `host`. required: [port] properties: + host: + type: string + description: | + The name or address a client reaches the engine by, when the node + knows it. A daemon leaves it absent; a remote environment's status + fills it with the instance's published address. port: type: integer description: | diff --git a/docs/spinloop-file.md b/docs/spinloop-file.md index 182109c8..e015a85b 100644 --- a/docs/spinloop-file.md +++ b/docs/spinloop-file.md @@ -145,12 +145,28 @@ picking one. As with `REMOTE`, note the missing `BASEURL` — the address is whichever node gets chosen. Writing one pins the address and turns routing off, and spinloop says so rather than choosing a node and discarding it. -A `FLEET` may also name a URL rather than a file, for a single endpoint that has -already done the choosing. That is the shape the spinloop gateway will take; it is -not implemented yet, and naming one today fails saying so. +A `FLEET` may also name a URL rather than a file: a single endpoint that has +already done the choosing, the shape +[`spinloop gateway`](commands/gateway.md) serves: + +```dockerfile +PROVIDER llamacpp +MODEL qwen3-27b +FLEET http://gateway.internal:4000 +``` + +Naming one reads no fleet file and contacts no node. The launch is pointed at +the address as given — with the OpenAI-compatible `/v1` prefix added when it +carries no path, and a value that already carries one used as given — and the +agent it launches authenticates with the endpoint's token, resolved the way a +key is resolved elsewhere: an `ENV` instruction, then the process environment, +then the `.env` beside the Spinloop. A variable already set wins, as on the +remote path. Set nowhere, the launch fails before it writes anything, naming +`OPENAI_API_KEY`. See [`spinloop fleet route`](commands/fleet.md#which-node-would-i-get) to check -which node you would get before launching anything. +which node you would get before launching anything — a route against an +endpoint just names it, without querying a node or starting one. ## Syntax diff --git a/examples/gateway-docker/.env.example b/examples/gateway-docker/.env.example new file mode 100644 index 00000000..2f69794a --- /dev/null +++ b/examples/gateway-docker/.env.example @@ -0,0 +1,12 @@ +# Copy to .env beside fleet.yaml, then `docker compose up -d --build`. +# +# Every value is a bearer secret for one box: the NODE_*_TOKENs gate each +# node's daemon control API, the NODE_*_ENGINE_KEYs gate each engine, and +# GATEWAY_TOKEN gates the gateway (both gateway services share it). Any +# non-empty values work — these are example secrets for a local stack, not +# secrets to reuse anywhere real. +NODE_A_TOKEN=node-a-dev-token +NODE_B_TOKEN=node-b-dev-token +NODE_A_ENGINE_KEY=node-a-engine-dev-key +NODE_B_ENGINE_KEY=node-b-engine-dev-key +GATEWAY_TOKEN=gateway-dev-token diff --git a/examples/gateway-docker/Dockerfile b/examples/gateway-docker/Dockerfile new file mode 100644 index 00000000..2ccaf869 --- /dev/null +++ b/examples/gateway-docker/Dockerfile @@ -0,0 +1,60 @@ +# A fleet node, and the gateway that fronts it: one image, two entry points. +# +# The node is a real `spinloop daemon` supervising a fake engine — Imposter's +# native engine standing in for llama-server, so the stack needs no GPU and no +# model. The gateway is the same binary running `spinloop gateway` over the +# fleet file baked in beside it; compose points each service at its entry. + +# Build spinloop from the working tree, so the stack tests THIS commit rather +# than a published artifact. +FROM golang:1.25-alpine AS build +WORKDIR /src +COPY go.mod go.sum ./ +RUN go mod download +# Only what the binary needs, so the image cannot depend on anything else in +# the tree and rebuilds stay cheap. +COPY cmd ./cmd +COPY internal ./internal +RUN CGO_ENABLED=0 go build -o /out/spinloop ./cmd/spinloop + +FROM alpine:3.20 +ARG TARGETARCH +# Pinned so a node image is reproducible; bump deliberately. +ARG IMPOSTER_VERSION=5.21.3 + +# procps gives the collector a vmstat whose output matches what it parses; +# BusyBox's differs, so CPU would otherwise be absent. +RUN apk add --no-cache ca-certificates curl bash procps + +# Imposter's native engine — a standalone binary, which is what lets the daemon +# supervise it as a child process the way it would a real engine. +RUN curl -fsSL -o /tmp/imposter.tar.gz \ + "https://github.com/imposter-project/imposter-go/releases/download/v${IMPOSTER_VERSION}/imposter-go_linux_${TARGETARCH}.tar.gz" \ + && curl -fsSL -o /tmp/checksums.txt \ + "https://github.com/imposter-project/imposter-go/releases/download/v${IMPOSTER_VERSION}/checksums.txt" \ + && (cd /tmp && grep " imposter-go_linux_${TARGETARCH}.tar.gz\$" checksums.txt | sed 's/imposter-go_linux_.*/imposter.tar.gz/' | sha256sum -c -) \ + && tar -xzf /tmp/imposter.tar.gz -C /usr/local/bin imposter-go \ + && chmod +x /usr/local/bin/imposter-go \ + && rm -f /tmp/imposter.tar.gz /tmp/checksums.txt + +COPY --from=build /out/spinloop /usr/local/bin/spinloop +COPY examples/gateway-docker/shim/llama-server /usr/local/bin/llama-server +COPY examples/gateway-docker/engine /opt/engine +# The gateway's own files: the two fleet files it can serve, and the nodes' +# Spinloop source beside them, so a wake resolves what a node runs from inside +# the container. The host-side fleet.yaml points at the same file where it +# lives in the tree, node/Spinloop. +COPY examples/gateway-docker/gateway /opt/gw +COPY examples/gateway-docker/node/Spinloop /opt/gw/Spinloop +RUN chmod +x /usr/local/bin/llama-server + +# A container has no useful $HOME, so pin spinloop's config directory — the +# same reason the cloud instance's daemon unit pins it. +ENV SPINLOOP_CONFIG_DIR=/var/lib/spinloop +WORKDIR /opt/node + +EXPOSE 4242 +# The default entry is the node; the gateway services override it in compose. +# Bind the control API on all interfaces so the gateway container can reach +# it; a token is required for that, supplied per-node by compose. +CMD ["spinloop", "daemon", "--api-addr", "0.0.0.0:4242"] diff --git a/examples/gateway-docker/README.md b/examples/gateway-docker/README.md new file mode 100644 index 00000000..5c279f87 --- /dev/null +++ b/examples/gateway-docker/README.md @@ -0,0 +1,153 @@ +# A gateway you can actually run + +Two `spinloop daemon` nodes and a `spinloop gateway` in front of them, on your +laptop, in containers — so you can see what +[`spinloop gateway`](../../docs/commands/gateway.md) does before pointing a real +fleet at it. No GPUs, no cloud, no model downloads. + +```sh +cp .env.example .env +docker compose up -d --build + +# from this directory, with the tokens exported +set -a && . ./.env && set +a + +# the gateway's own surface +curl -H "Authorization: Bearer $GATEWAY_TOKEN" http://127.0.0.1:4000/v1/models +curl -X POST -H "Authorization: Bearer $GATEWAY_TOKEN" \ + -H "Content-Type: application/json" \ + -d '{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' \ + http://127.0.0.1:4000/v1/chat/completions + +# and the fleet underneath it, the way spinloop fleet drives any fleet +spinloop fleet status --fleet ./fleet.yaml +spinloop fleet start node-b --fleet ./fleet.yaml +``` + +The stack brings up **two gateways** over the same two nodes: `gateway` on port +4000, which wakes a node when nothing is serving, and `gateway-cold` on port +4001, which refuses to — the `wake: off` policy from its own fleet file, so the +two ways of running a fleet are visible side by side. + +## What is real and what is not + +**Real**: each node runs the actual `spinloop daemon` from this repository, +serving its control API over the network with bearer-token auth, and supervises +its engine as a real child process. The gateway is the same binary running +`spinloop gateway`: it chooses a node with the fleet's own selector, wakes one +when nothing is serving, holds the request until the engine answers, and swaps +the caller's authorisation for the engine key its fleet entry names. + +**Not real**: the engine. Instead of `llama-server` there is a +[`llama-server` shim](shim/llama-server) that starts +[Imposter](https://imposter.sh)'s native engine, which serves a canned +`/health`, a `/metrics` in llama.cpp's Prometheus dialect, and OpenAI-shaped +completion replies — streamed, when asked, in the server-sent-events shape. So +a request through the gateway genuinely travels to a woken node and back, and +the streamed reply you see is the one the fake engine produced. Nothing is +inferring anything. + +That trade is deliberate: what is being demonstrated (and tested) is the +gateway's routing, waking and key handling, not inference. + +**Also real**: the keys. Each engine is gated with the key its fleet entry +names — the node reports a key is required and never what it is, and the key +reaches the engine as a file path, so `docker compose exec node-a ps ax` shows +`--api-key-file`, not the key. The caller of the gateway presents only the +gateway's token; the node tokens and engine keys live with the gateway, which +is the one place that holds all of them. + +There are three Spinloops here: + +- [`client/Spinloop`](client/Spinloop) — what an *agent's* machine wears. Its + `FLEET` is the gateway's address, a URL rather than a file: the gateway has + done the choosing, and the agent is only pointed at it, with the gateway's + token as its key. +- [`node/Spinloop`](node/Spinloop) — what a *node* runs when started. Its + `BASEURL` binds the engine to every interface, which is why the gateway — a + different container — can reach it at all. +- the nodes hold no Spinloop of their own in the container; the image bakes a + copy of `node/Spinloop` beside the gateway's fleet files so the gateway's own + wakes resolve what a node runs. + +## Things worth trying + +```sh +# Cold request: nothing is serving, so the gateway wakes a node, holds the +# request until the engine answers, and streams the reply back. +curl -X POST -H "Authorization: Bearer $GATEWAY_TOKEN" \ + -H "Content-Type: application/json" \ + -d '{"model":"fake-model","stream":true,"messages":[{"role":"user","content":"hi"}]}' \ + http://127.0.0.1:4000/v1/chat/completions + +# The same at the wake: off gateway: refused, naming the node and the command +# that would start it. +curl -i -X POST -H "Authorization: Bearer $GATEWAY_TOKEN" \ + -H "Content-Type: application/json" \ + -d '{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' \ + http://127.0.0.1:4001/v1/chat/completions + +# wake: off still routes to what is already running — it decides whether to +# start, not whether to answer. +spinloop fleet start node-a --fleet ./fleet.yaml +curl -X POST -H "Authorization: Bearer $GATEWAY_TOKEN" \ + -H "Content-Type: application/json" \ + -d '{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' \ + http://127.0.0.1:4001/v1/chat/completions + +# A wrong token is a 401, not a routing decision. +curl -i http://127.0.0.1:4000/v1/models + +# The engine, directly: gated, like any engine the fleet gates. +curl -i -X POST http://127.0.0.1:18080/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' + +# Launch an agent against the gateway: the agent gets the gateway's address +# and the gateway's token, and nothing else. +OPENAI_API_KEY="$GATEWAY_TOKEN" spinloop harness ./client/Spinloop +``` + +## It is also the integration test + +`./run-tests.sh` drives this same stack and asserts the behaviours above: a +model is listed once running, a cold request wakes a node and streams a reply, +the engine key is injected and never reaches the caller, a wrong token is 401, +and `wake: off` refuses. CI runs it on every pull request, which is the point: +an example that is exercised cannot quietly stop working. + +```sh +./run-tests.sh # up, assert, tear down +./run-tests.sh --keep # leave the stack running to poke at +``` + +## How it fits together + +| File | What it is | +| --- | --- | +| `compose.yaml` | Two nodes, a gateway that wakes, a gateway that refuses to. Every service that listens on a non-loopback address needs a token, and the gateways need the fleet file's node tokens and engine keys in their environment. | +| `fleet.yaml` | The *operator's* view: the two nodes over their published ports, with `engine:` blocks because the engines are published on ports the daemons cannot know. | +| `gateway/fleet.yaml` | What the `gateway` service serves: the same two nodes, addressed by compose service name — where the gateway can reach them, with no `engine:` override needed. | +| `gateway/fleet-cold.yaml` | The same fleet with `wake: off`, served by `gateway-cold`. | +| `Dockerfile` | Builds spinloop from this working tree, adds the Imposter engine and the shim, and bakes the gateway's files in. | +| `shim/llama-server` | Stands in for the engine binary. Reads the key file the daemon passes and hands the mock its gate as an environment variable, so the value never rides on a command line. | +| `engine/` | What the fake engine serves: `/health` and `/metrics` for the daemon, and the gated, stream-answering OpenAI routes. | +| `node/Spinloop` | What a node runs when started: a model, and a `BASEURL` that binds the engine to every interface. | +| `client/Spinloop` | What an *agent's* machine wears: a model, and a `FLEET` that is the gateway's address. | + +Two details that are easy to get wrong, and matter: + +- **The node's `BASEURL` binds the engine wide.** Without it llama-server + binds `127.0.0.1`, the daemon reports the engine loopback-only, and the + gateway is right to refuse routing to it — an engine that answers only on + its own machine is not a candidate for a gateway on another. +- **The shim execs the engine binary, not `imposter up`.** The CLI wrapper + exits 0 when its child dies, which the daemon would correctly record as a + clean stop — so a crash test would pass while testing nothing. + +## See also + +- [`examples/fleet-docker/`](../fleet-docker/) — a plain fleet, no gateway +- [`docs/commands/gateway.md`](../../docs/commands/gateway.md) +- [`docs/spinloop-file.md`](../../docs/spinloop-file.md) — the endpoint form of `FLEET` +- [HTTP Control API](../../docs/http-api.md) diff --git a/examples/gateway-docker/client/Spinloop b/examples/gateway-docker/client/Spinloop new file mode 100644 index 00000000..c56f8a21 --- /dev/null +++ b/examples/gateway-docker/client/Spinloop @@ -0,0 +1,13 @@ +# What an agent's machine wears to use this stack: a model, and a FLEET that +# names the gateway's address rather than a fleet file. The gateway has done +# the choosing; the agent is only pointed at it. +# +# The agent holds exactly one credential — the gateway's token, as +# OPENAI_API_KEY (from the environment, or a .env beside this file). The node +# tokens and engine keys live with the gateway, which presents them to the +# nodes and the engines; none of them ever reach the agent. +PROVIDER llamacpp +MODEL org/fake-model +ALIAS fake-model +CONTEXT 4096 +FLEET http://127.0.0.1:4000 diff --git a/examples/gateway-docker/compose.yaml b/examples/gateway-docker/compose.yaml new file mode 100644 index 00000000..4f590506 --- /dev/null +++ b/examples/gateway-docker/compose.yaml @@ -0,0 +1,63 @@ +# Two fleet nodes and two gateways in front of them: one that wakes a node +# when nothing is serving, one that refuses to, so the two wake policies are +# visible side by side. +# +# cp .env.example .env +# docker compose up -d --build +# ./run-tests.sh # or curl the gateway yourself — see README.md +name: spinloop-gateway-example + +services: + node-a: + build: + context: ../.. + dockerfile: examples/gateway-docker/Dockerfile + image: spinloop-gateway-node:example + environment: + SPINLOOP_API_TOKEN: ${NODE_A_TOKEN:?set NODE_A_TOKEN (copy .env.example to .env)} + ports: + - "14242:4242" + # The engine, published for direct curls from the host. The gateways + # need no published port: inside the compose network they reach the + # engine where the node's Spinloop binds it, 8080. + - "18080:8080" + + node-b: + image: spinloop-gateway-node:example + depends_on: [node-a] + environment: + SPINLOOP_API_TOKEN: ${NODE_B_TOKEN:?set NODE_B_TOKEN (copy .env.example to .env)} + ports: + - "14243:4242" + - "18081:8080" + + gateway: + image: spinloop-gateway-node:example + depends_on: [node-a, node-b] + command: ["spinloop", "gateway", "--fleet", "/opt/gw/fleet.yaml", "--listen", "0.0.0.0:4000"] + environment: + # The gateway's own token, resolved the way a daemon's is — from the + # environment here. Callers present it; the node tokens and engine keys + # the fleet file names are resolved from the same environment, at + # startup and at every wake. + SPINLOOP_API_TOKEN: ${GATEWAY_TOKEN:?set GATEWAY_TOKEN (copy .env.example to .env)} + NODE_A_TOKEN: ${NODE_A_TOKEN:?} + NODE_B_TOKEN: ${NODE_B_TOKEN:?} + NODE_A_ENGINE_KEY: ${NODE_A_ENGINE_KEY:?} + NODE_B_ENGINE_KEY: ${NODE_B_ENGINE_KEY:?} + ports: + - "4000:4000" + + # The wake: off policy over the same two nodes, on its own port. + gateway-cold: + image: spinloop-gateway-node:example + depends_on: [node-a, node-b] + command: ["spinloop", "gateway", "--fleet", "/opt/gw/fleet-cold.yaml", "--listen", "0.0.0.0:4001"] + environment: + SPINLOOP_API_TOKEN: ${GATEWAY_TOKEN:?set GATEWAY_TOKEN (copy .env.example to .env)} + NODE_A_TOKEN: ${NODE_A_TOKEN:?} + NODE_B_TOKEN: ${NODE_B_TOKEN:?} + NODE_A_ENGINE_KEY: ${NODE_A_ENGINE_KEY:?} + NODE_B_ENGINE_KEY: ${NODE_B_ENGINE_KEY:?} + ports: + - "4001:4001" diff --git a/examples/gateway-docker/engine/engine-config.yaml b/examples/gateway-docker/engine/engine-config.yaml new file mode 100644 index 00000000..b20a5be6 --- /dev/null +++ b/examples/gateway-docker/engine/engine-config.yaml @@ -0,0 +1,79 @@ +# What the fake engine serves. +# +# The two ungated routes are for the daemon itself: /health is its readiness +# probe and /metrics the scrape its collector parses in llama.cpp's dialect. +# Both run on the node's own machine, so neither needs the credential that +# stands between the gateway and the engine. +# +# The three OpenAI routes are gated: the gateway presents the key its fleet +# entry names, and the shim hands it to this process as the ENGINE_API_KEY +# environment variable. gate.js applies the check and, for the completion +# routes, answers streaming requests in the server-sent-events shape. +plugin: rest +resources: + - path: /health + method: GET + response: + statusCode: 200 + headers: + Content-Type: application/json + content: '{"status":"ok"}' + + # The llamacpp: counters internal/metrics parses. Static values keep the + # assertions deterministic; what is being tested is the collection path, not + # the arithmetic of a real engine. + - path: /metrics + method: GET + response: + statusCode: 200 + headers: + Content-Type: text/plain + content: | + # HELP llamacpp:prompt_tokens_total Number of prompt tokens processed. + llamacpp:prompt_tokens_total 4096 + # HELP llamacpp:tokens_predicted_total Number of tokens predicted. + llamacpp:tokens_predicted_total 1024 + # HELP llamacpp:n_decode_total Number of decode runs. + llamacpp:n_decode_total 900 + # HELP llamacpp:requests_processing Number of processing requests. + llamacpp:requests_processing 2 + # HELP llamacpp:requests_deferred Number of deferred requests. + llamacpp:requests_deferred 1 + # HELP llamacpp:request_success_total Number of successful requests. + llamacpp:request_success_total 17 + + - path: /v1/models + method: GET + steps: + - type: script + lang: javascript + file: gate.js + response: + statusCode: 200 + headers: + Content-Type: application/json + content: '{"object":"list","data":[{"id":"fake-model","object":"model"}]}' + + - path: /v1/chat/completions + method: POST + steps: + - type: script + lang: javascript + file: gate.js + response: + statusCode: 200 + headers: + Content-Type: application/json + content: '{"id":"chatcmpl-1","object":"chat.completion","created":1,"model":"fake-model","choices":[{"index":0,"message":{"role":"assistant","content":"Hello from the fake engine"},"finish_reason":"stop"}],"usage":{"prompt_tokens":8,"completion_tokens":5,"total_tokens":13}}' + + - path: /v1/completions + method: POST + steps: + - type: script + lang: javascript + file: gate.js + response: + statusCode: 200 + headers: + Content-Type: application/json + content: '{"id":"cmpl-1","object":"text_completion","created":1,"model":"fake-model","choices":[{"index":0,"text":"Hello from the fake engine","finish_reason":"stop"}],"usage":{"prompt_tokens":8,"completion_tokens":5,"total_tokens":13}}' diff --git a/examples/gateway-docker/engine/gate.js b/examples/gateway-docker/engine/gate.js new file mode 100644 index 00000000..3fde923c --- /dev/null +++ b/examples/gateway-docker/engine/gate.js @@ -0,0 +1,53 @@ +// The fake engine's gate and stream branch, shared by every gated route. +// +// The key arrives as the ENGINE_API_KEY environment variable: the shim read it +// from the --api-key-file the daemon pointed the engine at, so the value never +// rides on an argument any local user could read. A request presenting a +// different bearer — or none — is refused. With no key supplied the engine is +// ungated, which is right for one reached only over loopback. +// +// A completion request that asks for a stream gets the server-sent-events +// shape; the reply is one canned body, because what is under test is the +// gateway passing a streamed reply through, not an engine tokenising. + +var key = env["ENGINE_API_KEY"] || ""; + +if (key !== "" && (context.request.headers["Authorization"] || "") !== "Bearer " + key) { + respond() + .withStatusCode(401) + .withHeader("Content-Type", "application/json") + .withContent('{"error":{"message":"incorrect API key provided","type":"invalid_request_error"}}') + .skipDefaultBehaviour(); +} else if (context.request.method === "POST" && + /"stream"\s*:\s*true/.test(context.request.body || "")) { + respond() + .withStatusCode(200) + .withHeader("Content-Type", "text/event-stream") + .withContent(streamReply(context.request.path)) + .skipDefaultBehaviour(); +} +// Otherwise the resource's own canned response applies. + +function streamReply(path) { + if (path === "/v1/completions") { + return "data: " + JSON.stringify({ + id: "cmpl-1", object: "text_completion", created: 1, model: "fake-model", + choices: [{ index: 0, text: "Hello from the fake engine", finish_reason: "stop" }] + }) + "\n\ndata: [DONE]\n\n"; + } + var first = { + id: "chatcmpl-1", object: "chat.completion.chunk", created: 1, model: "fake-model", + choices: [{ + index: 0, + delta: { role: "assistant", content: "Hello from the fake engine" }, + finish_reason: null + }] + }; + var last = { + id: "chatcmpl-1", object: "chat.completion.chunk", created: 1, model: "fake-model", + choices: [{ index: 0, delta: {}, finish_reason: "stop" }] + }; + return "data: " + JSON.stringify(first) + + "\n\ndata: " + JSON.stringify(last) + + "\n\ndata: [DONE]\n\n"; +} diff --git a/examples/gateway-docker/fleet.yaml b/examples/gateway-docker/fleet.yaml new file mode 100644 index 00000000..76d5dd75 --- /dev/null +++ b/examples/gateway-docker/fleet.yaml @@ -0,0 +1,36 @@ +# The operator's view of this stack, from your machine: the two nodes over the +# ports compose publishes. +# +# The gateways are not nodes in it — they serve the fleet files baked into +# the image — and the client's Spinloop names the gateway's published address, +# not this file. This is how you watch and drive the fleet itself: status, +# start, stop, metrics, logs. +# +# The engine: blocks here are not optional: each engine binds 8080 inside its +# container and is published on 18080/18081 outside, which the daemons cannot +# know. (The gateways' own fleet files need no override — inside the compose +# network the engine is where it binds.) +# +# The nodes' Spinloop source is node/Spinloop, which is also why +# `spinloop fleet start` works from here: it resolves what to run on a node +# and pushes that. +prefer: idle + +nodes: + - name: node-a + host: 127.0.0.1 + port: 14242 + tokenEnv: NODE_A_TOKEN + engineTokenEnv: NODE_A_ENGINE_KEY + file: ./node/Spinloop + engine: + port: 18080 + + - name: node-b + host: 127.0.0.1 + port: 14243 + tokenEnv: NODE_B_TOKEN + engineTokenEnv: NODE_B_ENGINE_KEY + file: ./node/Spinloop + engine: + port: 18081 diff --git a/examples/gateway-docker/gateway/fleet-cold.yaml b/examples/gateway-docker/gateway/fleet-cold.yaml new file mode 100644 index 00000000..1245077c --- /dev/null +++ b/examples/gateway-docker/gateway/fleet-cold.yaml @@ -0,0 +1,22 @@ +# The same fleet as fleet.yaml, with waking refused: a request nothing is +# serving fails naming the node that would be started and the command that +# would start it, rather than starting it. It is the wake: off policy for the +# whole file — may work be started on these machines on demand, or only used +# where it is already running. +prefer: idle +wake: off + +nodes: + - name: node-a + host: node-a + port: 4242 + tokenEnv: NODE_A_TOKEN + engineTokenEnv: NODE_A_ENGINE_KEY + file: ./Spinloop + + - name: node-b + host: node-b + port: 4242 + tokenEnv: NODE_B_TOKEN + engineTokenEnv: NODE_B_ENGINE_KEY + file: ./Spinloop diff --git a/examples/gateway-docker/gateway/fleet.yaml b/examples/gateway-docker/gateway/fleet.yaml new file mode 100644 index 00000000..baf70f97 --- /dev/null +++ b/examples/gateway-docker/gateway/fleet.yaml @@ -0,0 +1,32 @@ +# What the gateway serves: this stack's two nodes, addressed the way the +# gateway's container reaches them — by compose service name, on the daemon's +# port. +# +# The gateway is the client that wakes a node, so it must resolve each node's +# Spinloop source from its own filesystem: the Dockerfile bakes node/Spinloop +# in beside this file as /opt/gw/Spinloop, which is what `file` names. +# +# No engine: blocks, for once: each engine binds 0.0.0.0:8080 — its Spinloop's +# BASEURL says so, the daemon reports it — and the gateway dials :8080 +# over the compose network. The client-side fleet.yaml needs the override, +# because the engine is published to the host on another port. +# +# Tokens are named by variable, as everywhere: the gateway container's +# environment supplies the values (compose.yaml), so nothing secret is in this +# file. +prefer: idle + +nodes: + - name: node-a + host: node-a + port: 4242 + tokenEnv: NODE_A_TOKEN + engineTokenEnv: NODE_A_ENGINE_KEY + file: ./Spinloop + + - name: node-b + host: node-b + port: 4242 + tokenEnv: NODE_B_TOKEN + engineTokenEnv: NODE_B_ENGINE_KEY + file: ./Spinloop diff --git a/examples/gateway-docker/node/Spinloop b/examples/gateway-docker/node/Spinloop new file mode 100644 index 00000000..ac359350 --- /dev/null +++ b/examples/gateway-docker/node/Spinloop @@ -0,0 +1,14 @@ +# What each node runs when a start request gives it work — the source both +# fleet files point at. The image bakes a copy beside the gateway's fleet +# files, so the gateway's own wakes resolve it from inside the container. +# +# The BASEURL is the point: it binds the engine to every interface. Without +# it llama-server binds 127.0.0.1, the daemon reports the engine loopback-only, +# and the gateway — a different container — is right to refuse routing to it. +# The client's Spinloop has no BASEURL at all: it names no engine, because it +# never talks to one. +PROVIDER llamacpp +MODEL org/fake-model +ALIAS fake-model +CONTEXT 4096 +BASEURL http://0.0.0.0:8080/v1 diff --git a/examples/gateway-docker/run-tests.sh b/examples/gateway-docker/run-tests.sh new file mode 100755 index 00000000..622efcab --- /dev/null +++ b/examples/gateway-docker/run-tests.sh @@ -0,0 +1,621 @@ +#!/usr/bin/env bash +# +# Drives the dockerised gateway stack and asserts the behaviours the gateway +# promises: a model is listed once running, a cold request wakes a node and +# streams a reply, the engine key is injected and never reaches the caller, a +# wrong token is 401, and wake: off refuses. This is both the CI integration +# test and something a maintainer can run locally — there is no CI-only path +# that can drift from what you run by hand. +# +# Usage: ./run-tests.sh [--keep] +# --keep leave the stack running afterwards, to poke at it yourself + +set -euo pipefail + +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +readonly HERE +REPO_ROOT="$(cd "${HERE}/../.." && pwd)" +readonly REPO_ROOT +# Where the built binary lands; the stack is driven by the spinloop built from +# this working tree, so the test covers this commit. +readonly SPINLOOP_BIN="${HERE}/.spinloop-test-bin" +readonly READY_TIMEOUT_SECS=90 +readonly GATEWAY_URL=http://127.0.0.1:4000 +readonly GATEWAY_COLD_URL=http://127.0.0.1:4001 +readonly COMPLETION_BODY='{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' + +keep_stack=0 +failures=0 + +####################################### +# Report a passing assertion. +# Arguments: +# Description of what passed. +# Outputs: +# Writes the result to stdout. +####################################### +pass() { + echo " ok - $1" +} + +####################################### +# Report a failing assertion and record it, without aborting the run — one +# failure should not hide the rest. +# Globals: +# failures +# Arguments: +# Description, expected, actual. +# Outputs: +# Writes the failure to stderr. +####################################### +fail() { + echo " FAIL - $1" >&2 + echo " expected: $2" >&2 + echo " actual: $3" >&2 + failures=$((failures + 1)) +} + +####################################### +# Assert that a string contains a substring. +# Arguments: +# Description, haystack, needle. +####################################### +assert_contains() { + local description="$1" haystack="$2" needle="$3" + if [[ "${haystack}" == *"${needle}"* ]]; then + pass "${description}" + else + fail "${description}" "to contain '${needle}'" "${haystack}" + fi +} + +####################################### +# Assert that a string does not contain a substring. +# Arguments: +# Description, haystack, needle. +####################################### +assert_not_contains() { + local description="$1" haystack="$2" needle="$3" + if [[ "${haystack}" != *"${needle}"* ]]; then + pass "${description}" + else + fail "${description}" "not to contain '${needle}'" "${haystack}" + fi +} + +####################################### +# Assert an exact string equality. +# Arguments: +# Description, actual, expected. +####################################### +assert_equals() { + local description="$1" actual="$2" expected="$3" + if [[ "${actual}" == "${expected}" ]]; then + pass "${description}" + else + fail "${description}" "${expected}" "${actual}" + fi +} + +####################################### +# Run a docker compose command, showing its output only when it fails. These +# commands are noisy on success and the test's own output is the point, but a +# silent failure is worse than noise: a `compose up` that cannot pull leaves +# nothing behind but "Tearing down..." and an exit code. +# Globals: +# HERE +# Arguments: +# Arguments to pass to docker compose. +# Returns: +# The command's exit status. +####################################### +compose() { + local out rc=0 + out="$(docker compose -f "${HERE}/compose.yaml" "$@" 2>&1)" || rc=$? + if (( rc != 0 )); then + echo "Error: docker compose $* failed (exit ${rc}):" >&2 + echo "${out}" >&2 + fi + return "${rc}" +} + +####################################### +# The container state for one service, or "" when docker cannot say. +# Globals: +# HERE +# Arguments: +# Service name. +# Outputs: +# Writes the state to stdout. +####################################### +container_state() { + docker compose -f "${HERE}/compose.yaml" ps --format '{{.State}}' "$1" 2>/dev/null +} + +####################################### +# Dump what the containers are doing, for a wait that timed out. +# Globals: +# HERE +# Outputs: +# Writes container state and recent logs to stderr. +####################################### +diagnose_fleet() { + docker compose -f "${HERE}/compose.yaml" ps >&2 2>&1 || true + docker compose -f "${HERE}/compose.yaml" logs --tail 20 >&2 2>&1 || true +} + +####################################### +# Run `spinloop fleet` against the example's fleet.yaml. +# Globals: +# SPINLOOP_BIN, HERE +# Arguments: +# Arguments to pass to `spinloop fleet`. +# Outputs: +# The command's stdout; stderr is discarded so assertions read cleanly. +####################################### +fleet() { + "${SPINLOOP_BIN}" fleet "$@" --fleet "${HERE}/fleet.yaml" 2>/dev/null +} + +####################################### +# As fleet(), but merging stderr — for assertions about error messages. +# Globals: +# SPINLOOP_BIN, HERE +####################################### +fleet_with_stderr() { + "${SPINLOOP_BIN}" fleet "$@" --fleet "${HERE}/fleet.yaml" 2>&1 +} + +####################################### +# The state column for one node, or "" when the node is absent. +# Arguments: +# Node name. +####################################### +node_state() { + local name="$1" + fleet status | awk -v n="${name}" '$1 == n {print $2}' +} + +####################################### +# The gateway container's logs, for assertions about what it did. +# Globals: +# HERE +####################################### +gateway_logs() { + docker compose -f "${HERE}/compose.yaml" logs --no-color gateway 2>/dev/null || true +} + +####################################### +# The process list inside node-a. +# Globals: +# HERE +####################################### +node_a_processes() { + docker compose -f "${HERE}/compose.yaml" exec -T node-a ps ax 2>/dev/null || true +} + +####################################### +# GET a gateway path with the gateway's token. +# Globals: +# GATEWAY_URL, GATEWAY_TOKEN +# Arguments: +# Port (4000 or 4001), path. +####################################### +gateway_get() { + local port="$1" path="$2" + curl -fsS -H "Authorization: Bearer ${GATEWAY_TOKEN}" \ + "http://127.0.0.1:${port}${path}" 2>/dev/null || true +} + +####################################### +# POST a completion body at a gateway. Returns the body; a non-2xx makes the +# function fail, which is what the positive-path assertions want. +# Globals: +# GATEWAY_TOKEN +# Arguments: +# Port, path, body, extra curl arguments (e.g. a header dump file flag). +####################################### +gateway_post() { + local port="$1" path="$2" body="$3" + shift 3 + curl -fsS -X POST -H "Authorization: Bearer ${GATEWAY_TOKEN}" \ + -H "Content-Type: application/json" -d "${body}" \ + "http://127.0.0.1:${port}${path}" "$@" 2>/dev/null || true +} + +####################################### +# Wait until both node daemons answer, so assertions do not race the +# containers' startup. +# Globals: +# READY_TIMEOUT_SECS +# Returns: +# 0 once both nodes report a state, 1 on timeout. +####################################### +wait_for_fleet() { + local deadline=$((SECONDS + READY_TIMEOUT_SECS)) + while (( SECONDS < deadline )); do + # Read the table into a variable rather than piping it: under `pipefail` + # a `grep -q` that matches and exits first can leave the pipeline + # reporting the writer's SIGPIPE, which reads here as "nothing + # unreachable" — the opposite of what was found. + if [[ "$(fleet status)" != *unreachable* ]]; then + return 0 + fi + sleep 2 + done + echo "Error: the fleet did not become reachable in ${READY_TIMEOUT_SECS}s" >&2 + fleet status >&2 || true + diagnose_fleet + return 1 +} + +####################################### +# Wait until both gateways answer /health. +# Globals: +# READY_TIMEOUT_SECS, GATEWAY_TOKEN +####################################### +wait_for_gateways() { + local deadline=$((SECONDS + READY_TIMEOUT_SECS)) + while (( SECONDS < deadline )); do + if [[ -n "$(gateway_get 4000 /health)" && -n "$(gateway_get 4001 /health)" ]]; then + return 0 + fi + sleep 2 + done + echo "Error: the gateways did not come up in ${READY_TIMEOUT_SECS}s" >&2 + diagnose_fleet + return 1 +} + +####################################### +# Wait for one node to reach a state. +# Arguments: +# Node name, expected state, timeout in seconds. +# Returns: +# 0 when the state is reached, 1 on timeout. +####################################### +wait_for_state() { + local name="$1" want="$2" timeout="$3" + local deadline=$((SECONDS + timeout)) + while (( SECONDS < deadline )); do + if [[ "$(node_state "${name}")" == "${want}" ]]; then + return 0 + fi + sleep 1 + done + return 1 +} + +####################################### +# Tear the stack down unless --keep was given. Registered as an EXIT trap so a +# failure part-way through still cleans up. +# Globals: +# keep_stack, HERE +####################################### +cleanup() { + if (( keep_stack )); then + echo + echo "Stack left running (--keep). Try:" + echo " cd ${HERE} && set -a && . ./.env && set +a" + echo " curl -H 'Authorization: Bearer \$GATEWAY_TOKEN' http://127.0.0.1:4000/v1/models" + echo "Tear down with: docker compose -f ${HERE}/compose.yaml down -v" + return + fi + echo + echo "Tearing down..." + docker compose -f "${HERE}/compose.yaml" down -v >/dev/null 2>&1 || true + rm -f "${SPINLOOP_BIN}" +} + +####################################### +# Assert the gateway's own door: its token in, 401 out, and a 404 that names +# what it serves. +####################################### +test_gateway_auth() { + echo "The gateway's own door" + local ok + ok="$(gateway_get 4000 /health)" + assert_contains "health answers with the token" "${ok}" '"ok":true' + assert_equals "no token is 401" \ + "$(curl -s -o /dev/null -w '%{http_code}' "${GATEWAY_URL}/health")" "401" + assert_equals "a wrong token is 401" \ + "$(curl -s -o /dev/null -w '%{http_code}' -H 'Authorization: Bearer not-the-token' "${GATEWAY_URL}/health")" "401" + + local out + out="$(curl -s -w '\n%{http_code}' -H "Authorization: Bearer ${GATEWAY_TOKEN}" \ + "${GATEWAY_URL}/v1/nope" 2>/dev/null || true)" + assert_contains "an unknown path names the paths served" "${out}" "the gateway serves" + assert_contains "it names /v1/models" "${out}" "/v1/models" + assert_contains "and it is a 404" "${out}" "404" +} + +####################################### +# Assert a cold fleet lists what a wake can start, and the wake: off gateway +# lists nothing — it decides whether to start, and nothing is running. +####################################### +test_cold_listing() { + echo "A cold fleet lists what a wake can start" + local models cold + models="$(gateway_get 4000 /v1/models)" + assert_contains "the waking gateway lists the source's served name" \ + "${models}" '"id":"fake-model"' + cold="$(gateway_get 4001 /v1/models)" + assert_contains "the wake: off gateway lists nothing, since nothing runs" \ + "${cold}" '"data":[]' +} + +####################################### +# Assert the wake: off gateway refuses a cold request, names the node and the +# command that would start it, and starts nothing. +####################################### +test_wake_off_refuses_cold() { + echo "wake: off refuses a cold request" + local out + out="$(curl -s -w '\n%{http_code}' -X POST \ + -H "Authorization: Bearer ${GATEWAY_TOKEN}" \ + -H "Content-Type: application/json" -d "${COMPLETION_BODY}" \ + "${GATEWAY_COLD_URL}/v1/chat/completions" 2>/dev/null || true)" + assert_contains "nothing is serving, so the request fails" "${out}" "503" + assert_contains "the failure names the policy" "${out}" "wake is off" + assert_contains "it names the node and the start command" \ + "${out}" "spinloop fleet start node-a" + assert_equals "and node-a was not started" "$(node_state node-a)" "idle" + assert_equals "nor was node-b" "$(node_state node-b)" "idle" +} + +####################################### +# Assert the command the refusal names actually starts the node. +####################################### +test_suggested_start_works() { + echo "The command the refusal names starts the node" + fleet start node-a >/dev/null + if wait_for_state node-a running 30; then + pass "fleet start node-a brings it up" + else + fail "fleet start node-a brings it up" "running" "$(node_state node-a)" + fi + assert_contains "status shows what it serves" "$(fleet status)" "fake-model" +} + +####################################### +# Assert wake: off still routes to what is already running: it decides +# whether to start, not whether to answer. +####################################### +test_wake_off_routes_running() { + echo "wake: off still routes what is already running" + # The cold gateway's last reading is from before the start; let it go stale + # rather than race the two-second cache. + sleep 3 + local out + out="$(gateway_post 4001 /v1/chat/completions "${COMPLETION_BODY}")" + assert_contains "the running node answers through the cold gateway" \ + "${out}" "Hello from the fake engine" +} + +####################################### +# Assert the model list is what the fleet is running. +####################################### +test_models_listing() { + echo "The model list is what the fleet is running" + local models + models="$(gateway_get 4000 /v1/models)" + assert_contains "the gateway lists the served name" "${models}" '"id":"fake-model"' +} + +####################################### +# Assert a running node answers through the waking gateway, and the gateway +# logged the route with its node and its wake state. +####################################### +test_gateway_serves_running_node() { + echo "A running node answers through the gateway" + local out + out="$(gateway_post 4000 /v1/chat/completions "${COMPLETION_BODY}")" + assert_contains "the reply is the engine's" "${out}" "Hello from the fake engine" + + local logs + logs="$(gateway_logs)" + assert_contains "the gateway logged the route" "${logs}" "msg=routed" + assert_contains "it named the node" "${logs}" "node=node-a" + assert_contains "it knew the node was already running" "${logs}" "woken=false" +} + +####################################### +# Assert the engine key is injected and never reaches the caller: the engine +# is gated (a direct call without the key is refused, with it answered), the +# key arrived as a file path, and no reply, log or process list carries it. +####################################### +test_engine_key_gating() { + echo "The engine key is injected, and never reaches the caller" + local url="http://127.0.0.1:18080/v1/chat/completions" + assert_equals "a direct call with no key is refused" \ + "$(curl -s -o /dev/null -w '%{http_code}' -X POST -H 'Content-Type: application/json' \ + -d "${COMPLETION_BODY}" "${url}")" "401" + assert_equals "a direct call with a wrong key is refused" \ + "$(curl -s -o /dev/null -w '%{http_code}' -X POST \ + -H "Authorization: Bearer not-the-key" -H 'Content-Type: application/json' \ + -d "${COMPLETION_BODY}" "${url}")" "401" + local reply + reply="$(curl -fsS -X POST -H "Authorization: Bearer ${NODE_A_ENGINE_KEY}" \ + -H 'Content-Type: application/json' -d "${COMPLETION_BODY}" "${url}" 2>/dev/null || true)" + assert_contains "the node's own key opens it" "${reply}" "Hello from the fake engine" + + local status + status="$(curl -fsS -H "Authorization: Bearer ${NODE_A_TOKEN}" \ + http://127.0.0.1:14242/v1/status 2>/dev/null || true)" + assert_contains "the node reports its engine needs a key" "${status}" '"requiresKey":true' + assert_not_contains "the node never discloses the key" "${status}" "${NODE_A_ENGINE_KEY}" + + local enginelog + enginelog="$(fleet logs node-a --limit 50 2>/dev/null || true)" + assert_contains "the engine was gated by file" "${enginelog}" "--api-key-file" + assert_not_contains "the key itself never reaches the command line" \ + "${enginelog}" "${NODE_A_ENGINE_KEY}" + assert_not_contains "the key is not in the node's process list" \ + "$(node_a_processes)" "${NODE_A_ENGINE_KEY}" + + local via_gateway + via_gateway="$(gateway_post 4000 /v1/chat/completions "${COMPLETION_BODY}")" + assert_not_contains "the key never appears in a reply through the gateway" \ + "${via_gateway}" "${NODE_A_ENGINE_KEY}" +} + +####################################### +# Assert a cold request to the waking gateway starts a node and the streamed +# reply passes through in the server-sent-events shape. +####################################### +test_cold_request_wakes_and_streams() { + echo "A cold request wakes a node and streams a reply" + fleet stop node-a >/dev/null + wait_for_state node-a stopped 30 || true + # Let the gateway's reading of the fleet go stale, so the request below + # sees the stop rather than a two-second-old reading of a running node. + sleep 3 + + local headers="${HERE}/.stream-headers" + local out + out="$(gateway_post 4000 /v1/chat/completions \ + '{"model":"fake-model","stream":true,"messages":[{"role":"user","content":"hi"}]}' \ + -D "${headers}" || true)" + assert_contains "a streamed reply is server-sent-events" \ + "$(cat "${headers}" 2>/dev/null)" "text/event-stream" + assert_contains "the chunks pass through" "${out}" "Hello from the fake engine" + assert_contains "the stream ends" "${out}" "data: [DONE]" + rm -f "${headers}" + + if wait_for_state node-a running 60; then + pass "the request left the node it woke running" + else + fail "the request left the node it woke running" "running" "$(node_state node-a)" + fi + assert_contains "the gateway logged the wake" \ + "$(gateway_logs)" "Waking node-a to serve fake-model" +} + +####################################### +# Assert a harness launch against the client's Spinloop points the agent at +# the gateway's address with the gateway's token as its key. +# Globals: +# HERE, SPINLOOP_BIN +####################################### +test_launch_points_agent_at_gateway() { + echo "A launch points the agent at the gateway" + local sandbox="${HERE}/.launch-sandbox" + rm -rf "${sandbox}" + mkdir -p "${sandbox}/bin" "${sandbox}/home" + cat > "${sandbox}/bin/opencode" <<'STUB' +#!/usr/bin/env bash +echo "HARNESS base_url=${OPENAI_BASE_URL:-} key=${OPENAI_API_KEY:-}" +STUB + chmod +x "${sandbox}/bin/opencode" + + local launch + launch="$(PATH="${sandbox}/bin:${PATH}" HOME="${sandbox}/home" \ + XDG_CONFIG_HOME="${sandbox}/home/.config" \ + OPENAI_BASE_URL="" \ + OPENAI_API_KEY="${GATEWAY_TOKEN}" \ + "${SPINLOOP_BIN}" harness -O="${HERE}/client/Spinloop" -H opencode 2>&1 || true)" + assert_contains "the agent is pointed at the gateway with its prefix" \ + "${launch}" "base_url=http://127.0.0.1:4000/v1" + assert_contains "the agent is given the gateway's token as its key" \ + "${launch}" "key=${GATEWAY_TOKEN}" + local config="${sandbox}/home/.config/opencode/opencode.json" + if [[ -f "${config}" ]]; then + pass "the harness config was written" + assert_contains "and it carries the gateway's address" \ + "$(cat "${config}")" "127.0.0.1:4000" + else + fail "the harness config was written" "${config}" "missing" + fi + rm -rf "${sandbox}" +} + +####################################### +# Assert a launch that cannot authenticate the gateway fails before the agent +# is started and before anything is written, naming the variable. +# Globals: +# HERE, SPINLOOP_BIN +####################################### +test_launch_fails_without_token() { + echo "A launch without the gateway's token fails, naming the variable" + local sandbox="${HERE}/.launch-sandbox" + rm -rf "${sandbox}" + mkdir -p "${sandbox}/bin" "${sandbox}/home" + cat > "${sandbox}/bin/opencode" <<'STUB' +#!/usr/bin/env bash +echo "HARNESS base_url=${OPENAI_BASE_URL:-} key=${OPENAI_API_KEY:-}" +STUB + chmod +x "${sandbox}/bin/opencode" + + local launch + launch="$(PATH="${sandbox}/bin:${PATH}" HOME="${sandbox}/home" \ + XDG_CONFIG_HOME="${sandbox}/home/.config" \ + OPENAI_BASE_URL="" \ + OPENAI_API_KEY="" \ + "${SPINLOOP_BIN}" harness -O="${HERE}/client/Spinloop" -H opencode 2>&1 || true)" + assert_contains "the failure names the variable to set" "${launch}" "OPENAI_API_KEY" + assert_not_contains "the agent was not started" "${launch}" "HARNESS" + if [[ ! -f "${sandbox}/home/.config/opencode/opencode.json" ]]; then + pass "no harness config was written" + else + fail "no harness config was written" "no opencode.json" "one was written" + fi + rm -rf "${sandbox}" +} + +main() { + if [[ "${1:-}" == "--keep" ]]; then + keep_stack=1 + fi + + cd "${HERE}" + if [[ ! -f .env ]]; then + echo "Using .env.example for tokens (no .env present)" + cp .env.example .env + fi + set -a + # shellcheck source=/dev/null + . ./.env + set +a + + trap cleanup EXIT + + echo "Building spinloop from the working tree..." + (cd "${REPO_ROOT}" && go build -o "${SPINLOOP_BIN}" ./cmd/spinloop) + + echo "Bringing the stack up..." + compose up -d --build + wait_for_fleet + wait_for_gateways + + echo + test_gateway_auth + echo + test_cold_listing + echo + test_wake_off_refuses_cold + echo + test_suggested_start_works + echo + test_wake_off_routes_running + echo + test_models_listing + echo + test_gateway_serves_running_node + echo + test_engine_key_gating + echo + test_cold_request_wakes_and_streams + echo + test_launch_points_agent_at_gateway + echo + test_launch_fails_without_token + + echo + if (( failures > 0 )); then + echo "${failures} assertion(s) failed" >&2 + return 1 + fi + echo "All assertions passed" +} + +main "$@" diff --git a/examples/gateway-docker/shim/llama-server b/examples/gateway-docker/shim/llama-server new file mode 100755 index 00000000..4c627d43 --- /dev/null +++ b/examples/gateway-docker/shim/llama-server @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Stands in for llama-server so `spinloop serve`/`spinloop daemon` can start an +# "engine" with no GPU and no model. The daemon execs this exactly as it would +# the real binary, so process supervision, log capture and crash detection are +# genuinely exercised. +# +# It execs the Imposter ENGINE BINARY directly, never `imposter up`. The CLI +# wrapper exits 0 when its child dies, which the daemon would correctly record +# as a clean stop — a crash assertion would then pass while testing nothing. +# Exec'ing the engine makes it the daemon's direct child, so an abnormal death +# is a real non-zero exit and reads as `crashed`. +set -euo pipefail + +# Echo the command line into the daemon's engine log before consuming it — the +# loop below shifts every argument away. The key arrives as +# --api-key-file and stays that way: the tests check this log for the +# flag and for the absence of the value, and the process list for neither. +echo "llama-server shim: argv: $*" + +port=8080 +keyfile="" +# llama-server takes many flags; only the port and the key file matter here. +# Everything else is ignored on purpose, so new engine flags can never break +# the shim. +while [ $# -gt 0 ]; do + case "$1" in + --port) port="${2:-8080}"; shift 2 ;; + --api-key-file) keyfile="${2:-}"; shift 2 ;; + *) shift ;; + esac +done + +# The mock reads its gate from the environment, so the key never appears on a +# command line inside the container either. An engine started with no key +# file is ungated, which is right for one reached only over loopback. +if [ -n "${keyfile}" ]; then + ENGINE_API_KEY="$(cat "${keyfile}")" + export ENGINE_API_KEY + echo "llama-server shim: engine key supplied by file (value not echoed)" +else + echo "llama-server shim: engine ungated (no key file supplied)" +fi + +echo "llama-server shim: starting the Imposter engine on port ${port}" +export IMPOSTER_PORT="${port}" +exec imposter-go /opt/engine diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 544ddaa3..69aeaad4 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -67,11 +67,12 @@ type Daemon struct { ready readiness hist systemHistory - mu sync.Mutex - runner string - model string - scrape metrics.ScrapeTarget - endpoint *EngineEndpoint + mu sync.Mutex + runner string + model string + servedName string + scrape metrics.ScrapeTarget + endpoint *EngineEndpoint } // log reads the daemon's logger, defaulting to discarding. @@ -114,17 +115,19 @@ func (d *Daemon) engineEndpoint() *EngineEndpoint { return d.endpoint } -// SetServed records what the daemon is serving, for status and metrics. -func (d *Daemon) SetServed(runner, model string) { +// SetServed records what the daemon is serving, for status and metrics. The +// served name is the name the engine answers to — an alias when one is set — +// and is empty when the engine answers only to the model. +func (d *Daemon) SetServed(runner, model, servedName string) { d.mu.Lock() - d.runner, d.model = runner, model + d.runner, d.model, d.servedName = runner, model, servedName d.mu.Unlock() } -func (d *Daemon) served() (string, string) { +func (d *Daemon) served() (string, string, string) { d.mu.Lock() defer d.mu.Unlock() - return d.runner, d.model + return d.runner, d.model, d.servedName } // configPath is the stored deploy config's location in the state directory. @@ -168,7 +171,7 @@ func (d *Daemon) Push(dc remote.DeployConfig) error { if err := os.WriteFile(d.configPath(), append(data, '\n'), 0o600); err != nil { return err } - d.SetServed(dc.Runner, dc.ModelID) + d.SetServed(dc.Runner, dc.ModelID, dc.ServedModelName) return nil } @@ -248,9 +251,9 @@ func (d *Daemon) StartEngine() error { argv = append(argv, keyArgs...) } if dc != nil { - d.SetServed(dc.Runner, dc.ModelID) + d.SetServed(dc.Runner, dc.ModelID, dc.ServedModelName) } - runner, model := d.served() + runner, model, _ := d.served() d.log().Info("starting engine", slog.String("source", source), slog.String("runner", runner), @@ -275,9 +278,13 @@ func (d *Daemon) StartEngine() error { // StatusResponse is the control API's status reply. type StatusResponse struct { - State string `json:"state"` - Runner string `json:"runner,omitempty"` - Model string `json:"model,omitempty"` + State string `json:"state"` + Runner string `json:"runner,omitempty"` + Model string `json:"model,omitempty"` + // ServedName is the name the running engine answers to — the served name + // its deploy config or Spinloop set — reported beside the model id when + // set: an aliased engine answers to both, and a caller may know either. + ServedName string `json:"servedName,omitempty"` UptimeSeconds int `json:"uptimeSeconds,omitempty"` LogPath string `json:"logPath,omitempty"` // LastActiveAt is when the engine last did any work, RFC 3339. Empty @@ -310,8 +317,15 @@ type StatusResponse struct { // binds 127.0.0.1:8080, which is useless to anyone else, and it cannot know // the name a client reaches this host by — a LAN name, a tailscale name, a // published container port. The caller composes these against the host it -// already has. +// already has. A node that does know that name — a remote environment, whose +// control plane publishes the instance's address — reports it in Host, and the +// caller uses it in place of the host it would otherwise supply. type EngineEndpoint struct { + // Host is the name or address a client reaches the engine by, when the + // node knows it. A daemon leaves it empty — it cannot know a client-facing + // name — but a remote environment's status fills it with the instance's + // published address, which is all a caller needs. + Host string `json:"host,omitempty"` // Port is the port the engine listens on — the engine's, never the // control API's. Port int `json:"port"` @@ -333,11 +347,12 @@ type EngineEndpoint struct { // engine's log lives, and how long the engine has been idle. func (d *Daemon) Status() StatusResponse { state, _, uptime := d.Sup.Status() - runner, model := d.served() + runner, model, servedName := d.served() resp := StatusResponse{ State: string(state), Runner: runner, Model: model, + ServedName: servedName, UptimeSeconds: uptime, LogPath: d.Sup.LogPath, Version: d.Version, @@ -362,8 +377,17 @@ func (d *Daemon) Status() StatusResponse { return resp } +// ReadyYes and ReadyNo are the two values StatusResponse.Ready and +// metrics.Stats.Ready take when a readiness reading applies. An empty Ready is +// neither of them: no reading has landed, so readiness is unknown rather than +// false, and a caller that gates on ReadyNo passes an unknown through. +const ( + ReadyYes = "ready" + ReadyNo = "not-ready" +) + // readinessField renders the shared readiness record as the string -// /v1/status and /v1/metrics both report: "ready", "not-ready", or "" when +// /v1/status and /v1/metrics both report: ReadyYes, ReadyNo, or "" when // no reading has landed — before the first check, or for a runner with no // known health-check convention. Callers gate this on the engine running; // it does not check that itself, since both callers already have. @@ -373,9 +397,9 @@ func (d *Daemon) readinessField() string { return "" } if ready { - return "ready" + return ReadyYes } - return "not-ready" + return ReadyNo } // activity renders the activity record as the pair both /v1/status and @@ -400,7 +424,7 @@ func (d *Daemon) activity() (lastActiveAt string, idleSeconds int) { // Errors; an absent source is simply omitted, per the engine-metrics spec. func (d *Daemon) Metrics(ctx context.Context) metrics.Stats { state, _, uptime := d.Sup.Status() - runner, model := d.served() + runner, model, _ := d.served() stats := metrics.Stats{ State: string(state), Runner: runner, diff --git a/internal/daemon/daemon_test.go b/internal/daemon/daemon_test.go index a5105351..826c6183 100644 --- a/internal/daemon/daemon_test.go +++ b/internal/daemon/daemon_test.go @@ -339,13 +339,19 @@ while true; do sleep 0.05; done`) } // A start carrying its config pushes and starts in one call. - dc := `{"runner":"llamacpp","modelId":"org/model","serveArgs":[]}` + dc := `{"runner":"llamacpp","modelId":"org/model","serveArgs":[],"servedModelName":"org/model-alias"}` if resp, body := do("POST", "/v1/start", "sekrit", dc); resp.StatusCode != 200 || body["state"] != "running" { t.Fatalf("start with body = %d %v", resp.StatusCode, body) } if stored, _ := d.StoredConfig(); stored == nil || stored.ModelID != "org/model" { t.Fatalf("start body not persisted: %+v", stored) } + // The served name crosses the wire beside the model id: an aliased engine + // is addressed by either, and a caller may know only one of them. + if resp, body := do("GET", "/v1/status", "sekrit", ""); resp.StatusCode != 200 || + body["model"] != "org/model" || body["servedName"] != "org/model-alias" { + t.Fatalf("status after aliased start = %d %v", resp.StatusCode, body) + } // A start body while running is a 409 that stores nothing. if resp, body := do("POST", "/v1/start", "sekrit", @@ -407,8 +413,8 @@ while true; do sleep 0.05; done`) t.Fatal(err) } waitForState(t, crash.Sup, StateCrashed) - if got := crash.Status(); got.State != "crashed" { - t.Fatalf("status after crash = %+v", got) + if got := crash.Status(); got.State != "crashed" || got.ServedName != "" { + t.Fatalf("status after crash = %+v, want no served name without an alias", got) } } diff --git a/internal/daemon/readiness_test.go b/internal/daemon/readiness_test.go index 50cd45a1..7d1e7c0f 100644 --- a/internal/daemon/readiness_test.go +++ b/internal/daemon/readiness_test.go @@ -46,7 +46,7 @@ while true; do sleep 0.05; done`) t.Fatal(err) } waitForState(t, d.Sup, StateRunning) - d.SetServed(runner, "model") + d.SetServed(runner, "model", "") return d } diff --git a/internal/fleet/config.go b/internal/fleet/config.go index 13841ebe..1083c7cd 100644 --- a/internal/fleet/config.go +++ b/internal/fleet/config.go @@ -62,6 +62,36 @@ func ParsePrefer(s string) (Prefer, error) { return "", fmt.Errorf("unknown preference %q: use %q or %q", s, PreferIdle, PreferActive) } +// WakePolicy is whether routing may start an engine on a node that is not +// running one when no running node serves what is wanted. It sits in the fleet +// file beside prefer for the reason prefer does: it describes how this +// cluster is to be used — may work be started on its machines on demand, or +// only used where it is already running. +type WakePolicy string + +const ( + // WakeOn starts an engine on an idle node when nothing is serving. + WakeOn WakePolicy = "on" + // WakeOff never starts one: a request nothing is serving fails, naming + // the node that would have been woken and the command that would start it. + WakeOff WakePolicy = "off" +) + +// ParseWakePolicy validates a wake policy from a file. +func ParseWakePolicy(s string) (WakePolicy, error) { + switch WakePolicy(s) { + case WakeOn, WakeOff: + return WakePolicy(s), nil + } + return "", fmt.Errorf("unknown wake policy %q: use %q or %q", s, WakeOn, WakeOff) +} + +// Wakes reports whether routing may start an engine on a node that is not +// running one. A file that declares nothing wakes, as routing has always done. +func (c *Config) Wakes() bool { + return c.WakePolicy != WakeOff +} + // Config is a parsed fleet.yaml: the nodes, plus where the file was read from // (the directory whose .env supplies token values). type Config struct { @@ -71,6 +101,10 @@ type Config struct { // should be used — spread the work, or consolidate it. Empty means // PreferIdle. Prefer Prefer `yaml:"prefer"` + // WakePolicy is the fleet-wide wake policy: whether routing may start an + // engine on a node that is not running one. Empty means WakeOn, as + // routing has always done when the setting is absent. + WakePolicy WakePolicy `yaml:"wake"` // APIKeyEnv names the environment variable holding the key this fleet's // remote nodes require, shared by every one of them: a remote's engine is // always gated by its key, so a fleet of remotes can name the variable @@ -202,6 +236,11 @@ func (c *Config) validate() error { return err } } + if c.WakePolicy != "" { + if _, err := ParseWakePolicy(string(c.WakePolicy)); err != nil { + return err + } + } seen := map[string]bool{} for i := range c.Nodes { n := &c.Nodes[i] diff --git a/internal/fleet/config_test.go b/internal/fleet/config_test.go index c37b5198..c99c5862 100644 --- a/internal/fleet/config_test.go +++ b/internal/fleet/config_test.go @@ -625,3 +625,41 @@ func TestPreferRejectsUnknownValue(t *testing.T) { } } } + +func TestWakeSetting(t *testing.T) { + cases := []struct { + decl string + wake bool + }{ + {"wake: on\n", true}, + {"wake: off\n", false}, + {"", true}, // absent: routing wakes, as it always has + } + for _, c := range cases { + name := strings.TrimSpace(c.decl) + if name == "" { + name = "absent" + } + t.Run(name, func(t *testing.T) { + cfg, err := Load(writeFleet(t, c.decl+"nodes:\n - name: a\n host: a.local\n", "")) + if err != nil { + t.Fatal(err) + } + if got := cfg.Wakes(); got != c.wake { + t.Errorf("Wakes() = %v, want %v", got, c.wake) + } + }) + } +} + +func TestWakeRejectsUnknownValue(t *testing.T) { + _, err := Load(writeFleet(t, "wake: sometimes\nnodes:\n - name: a\n host: a.local\n", "")) + if err == nil { + t.Fatal("an unknown wake value should fail to parse") + } + for _, want := range []string{"on", "off"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("error should name %q, got %q", want, err) + } + } +} diff --git a/internal/fleet/remote_node.go b/internal/fleet/remote_node.go index 5425fbc2..c698388b 100644 --- a/internal/fleet/remote_node.go +++ b/internal/fleet/remote_node.go @@ -3,6 +3,8 @@ package fleet import ( "context" "fmt" + "net/url" + "strconv" "strings" "time" @@ -157,15 +159,36 @@ func (n *remoteNode) Logs(ctx context.Context, offset int64, limit int) (daemon. } // statusFromRemote maps the control plane's status reply onto the node's status. -// It carries only what a control-plane reply can honestly be mapped across: the -// state and its last-active record. The version is not in this reply — the stats -// reply carries it — so it is empty here, and runner/model are likewise absent. +// It carries what a status reply can honestly be mapped across: the state, what +// the engine is serving (runner, model, served name) and its last-active record. +// The version is not in this reply — the stats reply carries it — so it is empty +// here. The serving fields are empty when the daemon reports none: an engine that +// is not running one, or a daemon the control plane could not reach. +// +// A running environment's reply also names where its engine answers — the +// instance's published address, which a daemon on the instance cannot know for +// itself but the control plane can. It is carried as the engine's host, so +// routing resolves a remote node's address the way it resolves any node's. +// Absent (a stopped or undeployed environment reports none) means no engine +// address, exactly as the parts would be. func statusFromRemote(resp remote.Response) daemon.StatusResponse { - return daemon.StatusResponse{ + s := daemon.StatusResponse{ State: resp.State, + Runner: resp.Runner, + Model: resp.ModelID, + ServedName: resp.ServedName, LastActiveAt: resp.LastActiveAt, IdleSeconds: resp.IdleSeconds, } + if u, err := url.Parse(resp.BaseURL); resp.BaseURL != "" && err == nil && u.Host != "" { + port, _ := strconv.Atoi(u.Port()) + s.Engine = &daemon.EngineEndpoint{ + Host: u.Hostname(), + Port: port, + Path: u.Path, + } + } + return s } // statsFromRemote maps the stats Lambda's reply onto the shared stats shape. The diff --git a/internal/fleet/remote_node_test.go b/internal/fleet/remote_node_test.go index fae18901..73603c91 100644 --- a/internal/fleet/remote_node_test.go +++ b/internal/fleet/remote_node_test.go @@ -279,7 +279,8 @@ func TestRemoteNodeStartWithIsRefused(t *testing.T) { func TestRemoteNodeStatusOverTheControlPlane(t *testing.T) { stubAWSCreds(t) srv := remoteControlServer(t, - `{"state":"running","healthy":true,"lastActiveAt":"2026-01-02T00:00:00Z","idleSeconds":30}`, http.StatusOK) + `{"state":"running","healthy":true,"runner":"llamacpp","modelId":"org/m","servedName":"m",`+ + `"base_url":"http://1.2.3.4:8000/v1","lastActiveAt":"2026-01-02T00:00:00Z","idleSeconds":30}`, http.StatusOK) node, err := NewRemoteNode("env", remote.Config{StartURL: srv.URL, StopURL: srv.URL, Region: "us-east-1"}) if err != nil { t.Fatal(err) @@ -288,6 +289,16 @@ func TestRemoteNodeStatusOverTheControlPlane(t *testing.T) { if !r.OK() || r.Status.State != "running" || r.Status.IdleSeconds != 30 { t.Errorf("status result = %+v", r) } + // What the engine is serving rides the status, so a router can match a + // request to this node: the model id and the served name beside it. + if r.Status.Runner != "llamacpp" || r.Status.Model != "org/m" || r.Status.ServedName != "m" { + t.Errorf("serving facts not mapped: %+v", r.Status) + } + // And where the engine answers: the control plane's published address, so + // the router can reach the node rather than only name it. + if r.Status.Engine == nil || r.Status.Engine.Host != "1.2.3.4" || r.Status.Engine.Port != 8000 { + t.Errorf("engine address not mapped: %+v", r.Status.Engine) + } if r.Name != "env" { t.Errorf("name = %q", r.Name) } @@ -364,6 +375,46 @@ func TestKeeperIsRemoteOnly(t *testing.T) { } } +// statusFromRemote carries the serving facts when the daemon reports them, and +// leaves them empty when it does not — a running-but-unreachable daemon, or an +// engine that is not running a model, must not be invented into serving one. +func TestStatusFromRemoteServingFacts(t *testing.T) { + with := statusFromRemote(remote.Response{ + State: "running", + Runner: "llamacpp", + ModelID: "org/m", + ServedName: "m", + }) + if with.Runner != "llamacpp" || with.Model != "org/m" || with.ServedName != "m" { + t.Errorf("serving facts should map across, got %+v", with) + } + without := statusFromRemote(remote.Response{State: "running"}) + if without.Runner != "" || without.Model != "" || without.ServedName != "" { + t.Errorf("an absent serving fact must stay empty, got %+v", without) + } +} + +// statusFromRemote carries a running environment's engine address — the +// control plane's published base url — as the engine's host, so routing can +// reach it the way it reaches any node. A stopped or undeployed environment +// reports none, so its status carries no engine address. +func TestStatusFromRemoteCarriesTheEngineAddress(t *testing.T) { + got := statusFromRemote(remote.Response{ + State: "running", + BaseURL: "http://1.2.3.4:8000/v1", + }) + if got.Engine == nil { + t.Fatal("a running environment's engine address was not carried") + } + if got.Engine.Host != "1.2.3.4" || got.Engine.Port != 8000 || got.Engine.Path != "/v1" { + t.Errorf("engine endpoint = %+v, want host 1.2.3.4 port 8000 path /v1", got.Engine) + } + // No base url — a stopped or undeployed environment — means no address. + if got := statusFromRemote(remote.Response{State: "stopped"}); got.Engine != nil { + t.Errorf("a stopped environment should carry no engine address: %+v", got.Engine) + } +} + // A remote node drives start, stop and metrics over its control plane exactly // like a node would, mapping each reply onto the node's types. func TestRemoteNodeStartStopMetricsOverTheControlPlane(t *testing.T) { diff --git a/internal/fleet/select.go b/internal/fleet/select.go index bf0c5f2f..3cc62d0c 100644 --- a/internal/fleet/select.go +++ b/internal/fleet/select.go @@ -48,16 +48,33 @@ func (w Want) prefer() Prefer { return w.Prefer } -// matches reports whether a node serving `serving` is serving what is wanted. -// A launch that names no model wants any running engine. -func (w Want) matches(serving string) bool { +// matches reports whether a node reporting any of the given served names is +// serving what is wanted. A launch that names no model wants any running +// engine. +func (w Want) matches(serving ...string) bool { if w.Model == "" && w.Alias == "" && w.ModelID == "" { return true } - if serving == "" { - return false + for _, s := range serving { + if s != "" && (s == w.Model || s == w.Alias || s == w.ModelID) { + return true + } } - return serving == w.Model || serving == w.Alias || serving == w.ModelID + return false +} + +// servingNames is every name a node reports itself serving: the model id, and +// the served name it was started under when it reports one. An aliased engine +// answers to both, so either matching is a match. +func servingNames(s daemon.StatusResponse) []string { + var names []string + if s.Model != "" { + names = append(names, s.Model) + } + if s.ServedName != "" && s.ServedName != s.Model { + names = append(names, s.ServedName) + } + return names } // wanted names the model for a message, preferring the Spinloop's own MODEL. @@ -101,6 +118,10 @@ type Choice struct { Reason string // Woken records that this node was started to satisfy the launch. Woken bool + // Gateway records that FLEET named an endpoint rather than a fleet file: + // the endpoint has already done the choosing, Node is empty, and BaseURL + // is the endpoint's address rather than a node's engine. + Gateway bool } // candidate pairs a node's file entry with what it answered, keeping the @@ -127,19 +148,59 @@ func (c *Config) candidates(results []NodeResult) []candidate { // running keeps the nodes that answered and are serving what is wanted. A node // that did not answer is skipped rather than fatal, exactly as it is a row // rather than a failure in `spinloop fleet status`. +// +// A node reporting ReadyNo is skipped too: the state reaches running when the +// engine process exists, which is before llama.cpp has fetched and loaded +// weights, and during that window nothing is listening on the engine's port. +// Selecting it would hand the caller an address that refuses connections. func running(cands []candidate, w Want) []candidate { var out []candidate for _, c := range cands { if !c.result.OK() || c.result.Status.State != string(daemon.StateRunning) { continue } - if w.matches(c.result.Status.Model) { + if c.result.Status.Ready == daemon.ReadyNo { + continue + } + if w.matches(servingNames(c.result.Status)...) { out = append(out, c) } } return out } +// loading keeps the nodes that are running what is wanted but whose engines +// have not answered yet — the ones running() skips on readiness. They are the +// nodes worth waiting for: the weights are already being loaded, so they reach +// serving sooner than anything a wake would start from cold. +func loading(cands []candidate, w Want) []candidate { + var out []candidate + for _, c := range cands { + if !c.result.OK() || c.result.Status.State != string(daemon.StateRunning) { + continue + } + if c.result.Status.Ready != daemon.ReadyNo { + continue + } + if w.matches(servingNames(c.result.Status)...) { + out = append(out, c) + } + } + return out +} + +// Loading names the nodes running the wanted model whose engines have not +// answered yet, in fleet-file order. A caller holding an ErrNoneServing uses it +// to tell "nothing is serving this" from "something is about to". +func (c *Config) Loading(results []NodeResult, w Want) []NodeConfig { + cands := loading(c.candidates(results), w) + out := make([]NodeConfig, 0, len(cands)) + for _, cand := range cands { + out = append(out, cand.entry) + } + return out +} + // rank orders matching nodes best-first under the preference in force. Ties // break by fleet-file order, so the same fleet in the same state chooses the // same node every time. @@ -184,7 +245,15 @@ func (c *Config) Select(ctx context.Context, w Want) (*Choice, error) { } } results := scope.FanOut(ctx, StatusCall) - return scope.choose(results, w) + return scope.Choose(results, w) +} + +// Choose applies the ranking to a fan-out's results the caller already holds, +// and resolves the winner's endpoint. It is Select without the query, for a +// caller that keeps its own reading of the fleet and reuses it across calls — +// the gateway does, because a burst of requests must not pay a fan-out each. +func (c *Config) Choose(results []NodeResult, w Want) (*Choice, error) { + return c.choose(results, w) } // ErrNoneServing reports that no node is serving what was wanted. It carries @@ -220,6 +289,11 @@ func describe(r NodeResult) string { if r.Status.Model != "" { s += " " + r.Status.Model } + // Without this a node skipped on readiness reads as running the model the + // request asked for, which makes the refusal look wrong. + if r.Status.Ready == daemon.ReadyNo { + s += " (not ready)" + } return s } @@ -233,11 +307,20 @@ func (c *Config) choose(results []NodeResult, w Want) (*Choice, error) { if !only.result.OK() { return nil, fmt.Errorf("node %q: %s", only.entry.Name, describe(only.result)) } - if only.result.Status.State == string(daemon.StateRunning) && !w.matches(only.result.Status.Model) { + if only.result.Status.State == string(daemon.StateRunning) && !w.matches(servingNames(only.result.Status)...) { return nil, fmt.Errorf( "node %q is serving %s, not %s: it will not be restarted — pick another node, or stop it yourself", only.entry.Name, only.result.Status.Model, w.wanted()) } + // Pinned to a node whose engine is up as a process but has not + // answered yet: naming the state is more use than reporting that + // nothing serves the model, since this node is about to. + if only.result.Status.State == string(daemon.StateRunning) && only.result.Status.Ready == daemon.ReadyNo { + return nil, fmt.Errorf( + "node %q is still starting %s: its engine is running but has not answered yet "+ + "(it may be fetching or loading weights) — check its log with `spinloop fleet logs %s`", + only.entry.Name, w.wanted(), only.entry.Name) + } } matching := running(cands, w) if len(matching) == 0 { @@ -302,6 +385,12 @@ func (c *Config) EngineBaseURL(n NodeConfig, status daemon.StatusResponse) (stri host, port, path := n.Host, 0, "" if ep := status.Engine; ep != nil { port, path = ep.Port, ep.Path + // A node that reports its engine's host — a remote environment, whose + // control plane knows the instance's published address — is reached + // there, in place of the host the fleet file supplies. + if ep.Host != "" { + host = ep.Host + } } if o := n.Engine; o != nil { if o.Host != "" { diff --git a/internal/fleet/select_test.go b/internal/fleet/select_test.go index 3190a733..412d101e 100644 --- a/internal/fleet/select_test.go +++ b/internal/fleet/select_test.go @@ -299,6 +299,22 @@ func TestEngineBaseURL(t *testing.T) { status: reported, want: "https://engine.example:8080/v1", }, + { + name: "a reported engine host is used in place of the fleet file's", + node: NodeConfig{Name: "env", Kind: "remote"}, + status: daemon.StatusResponse{ + Engine: &daemon.EngineEndpoint{Host: "1.2.3.4", Port: 8000, Path: "/v1"}, + }, + want: "http://1.2.3.4:8000/v1", + }, + { + name: "an override still beats a reported engine host", + node: NodeConfig{Name: "env", Kind: "remote", Engine: &EngineOverride{Host: "proxy"}}, + status: daemon.StatusResponse{ + Engine: &daemon.EngineEndpoint{Host: "1.2.3.4", Port: 8000, Path: "/v1"}, + }, + want: "http://proxy:8000/v1", + }, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { @@ -467,3 +483,85 @@ func TestWokenNodeIsRecognisedOnASecondLaunch(t *testing.T) { t.Error("an unrelated model should not match") } } + +// notReadyNode is a node whose engine process is up but has not answered its +// health check — llama.cpp fetching or loading weights, which is minutes on a +// large model and the whole time its port refuses connections. +func notReadyNode(name, model string) NodeResult { + r := runningNode(name, model, -1) + r.Status.Ready = daemon.ReadyNo + return r +} + +func TestNotReadyNodeIsNotSelected(t *testing.T) { + cfg := testConfig(t, "loading", "up") + results := []NodeResult{ + notReadyNode("loading", "qwen"), + runningNode("up", "qwen", 10), + } + got, err := cfg.choose(results, Want{Model: "qwen"}) + if err != nil { + t.Fatal(err) + } + if got.Node.Name != "up" { + t.Errorf("chose %q, want the node whose engine has answered", got.Node.Name) + } +} + +// The only node running the model has not answered yet: routing there would +// fail the request, so nothing is serving it. +func TestNotReadyIsTheSameAsNothingServing(t *testing.T) { + cfg := testConfig(t, "loading") + results := []NodeResult{notReadyNode("loading", "qwen")} + _, err := cfg.choose(results, Want{Model: "qwen"}) + var none *ErrNoneServing + if !errors.As(err, &none) { + t.Fatalf("err = %v, want ErrNoneServing", err) + } + if !strings.Contains(none.Error(), "not ready") { + t.Errorf("the fleet state should mark the node not ready, got:\n%s", none.Error()) + } +} + +// A readiness reading that never landed — an older daemon, or a runner with no +// known health check — is unknown, not false, so it must not exclude a node. +func TestUnknownReadinessStillSelects(t *testing.T) { + cfg := testConfig(t, "old") + results := []NodeResult{runningNode("old", "qwen", 10)} + if results[0].Status.Ready != "" { + t.Fatal("this fixture should report no readiness reading") + } + if _, err := cfg.choose(results, Want{Model: "qwen"}); err != nil { + t.Errorf("a node reporting no readiness should still be selected: %v", err) + } +} + +func TestPinnedNodeStillStartingSaysSo(t *testing.T) { + cfg := testConfig(t, "gpu") + results := []NodeResult{notReadyNode("gpu", "qwen")} + _, err := cfg.choose(results, Want{Model: "qwen", Node: "gpu"}) + if err == nil { + t.Fatal("expected a failure") + } + for _, want := range []string{"gpu", "qwen", "has not answered"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("message should mention %q, got: %v", want, err) + } + } +} + +// Loading is what lets a caller tell "nothing is serving this" from "something +// is about to", so it names the still-starting nodes and nothing else. +func TestLoadingNamesTheStartingNodes(t *testing.T) { + cfg := testConfig(t, "loading", "up", "idle", "other") + results := []NodeResult{ + notReadyNode("loading", "qwen"), + runningNode("up", "qwen", 10), + idleNode("idle"), + notReadyNode("other", "different-model"), + } + got := cfg.Loading(results, Want{Model: "qwen"}) + if len(got) != 1 || got[0].Name != "loading" { + t.Fatalf("Loading() = %v, want just the node starting qwen", got) + } +} diff --git a/internal/fleet/wake.go b/internal/fleet/wake.go index c3c3cee3..b0d3fa65 100644 --- a/internal/fleet/wake.go +++ b/internal/fleet/wake.go @@ -31,6 +31,44 @@ var wakePoll = 2 * time.Second // reads as a hang, so the caller is given something to print. type Waker func(format string, args ...any) +// ConfigFor resolves the deploy config a candidate node would be started +// with. The launch path resolves the same config for every candidate — one +// Spinloop describes what any node should start — while the gateway resolves +// each node's own Spinloop source, so different nodes may describe different +// engines. +type ConfigFor func(entry NodeConfig) (remote.DeployConfig, error) + +// ConstantConfig adapts one deploy config, already resolved, to a per-candidate +// resolver. The launch path uses it: its Spinloop describes what any node would +// start. +func ConstantConfig(dc remote.DeployConfig, err error) ConfigFor { + return func(NodeConfig) (remote.DeployConfig, error) { return dc, err } +} + +// configResolver resolves each candidate's deploy config at most once per wake, +// by node name. Resolving can cost more than a status read — the gateway +// parses each node's Spinloop source — and a wake asks for the same config +// twice per candidate: once to order them, once to start them. +type configResolver struct { + fn ConfigFor + dcs map[string]remote.DeployConfig + errs map[string]error +} + +func newConfigResolver(fn ConfigFor) *configResolver { + return &configResolver{fn: fn, dcs: map[string]remote.DeployConfig{}, errs: map[string]error{}} +} + +func (r *configResolver) config(entry NodeConfig) (remote.DeployConfig, error) { + if err, ok := r.errs[entry.Name]; ok { + return r.dcs[entry.Name], err + } + dc, err := r.fn(entry) + r.dcs[entry.Name] = dc + r.errs[entry.Name] = err + return dc, err +} + // Wake starts an engine for want on a node that is not running one, and waits // until its engine answers. Candidates are tried in fleet-file order, and a // node whose stored config already matches is preferred: it has the weights. @@ -38,17 +76,23 @@ type Waker func(format string, args ...any) // A node that refuses the config — a runner or model it cannot serve — is not // fatal while other candidates remain. When none succeeds, every refusal is // reported together. -func (c *Config) Wake(ctx context.Context, w Want, dc remote.DeployConfig, results []NodeResult, log Waker) (*Choice, error) { +func (c *Config) Wake(ctx context.Context, w Want, cfgFor ConfigFor, results []NodeResult, log Waker) (*Choice, error) { if log == nil { log = func(string, ...any) {} } - cands := wakeable(c.candidates(results), dc) + resolver := newConfigResolver(cfgFor) + cands := wakeable(c.candidates(results), resolver) if len(cands) == 0 { return nil, &ErrNoneServing{Results: results, Want: w, Path: c.Path} } var refused []string for _, cand := range cands { + dc, err := resolver.config(cand.entry) + if err != nil { + refused = append(refused, fmt.Sprintf("%s: %v", cand.entry.Name, err)) + continue + } node, err := c.NewNode(cand.entry) if err != nil { refused = append(refused, fmt.Sprintf("%s: %v", cand.entry.Name, err)) @@ -63,15 +107,22 @@ func (c *Config) Wake(ctx context.Context, w Want, dc remote.DeployConfig, resul return nil, err } log("Waking %s to serve %s...\n", cand.entry.Name, w.wanted()) - status, err := node.StartWith(ctx, &dc, engineKey) + _, err = node.StartWith(ctx, &dc, engineKey) if err != nil { // Another client may have woken this node first. That is // another route to the same place, not a failure — re-read // its state and take it if it is now serving what we want. + // The state alone is not the answer, though: the other start may + // still be loading, so the same readiness wait applies to a node + // we did not start ourselves. if isAlreadyRunning(err) { - if status, err = node.Status(ctx); err == nil && w.matches(status.Model) { - log("%s was already started by someone else; using it.\n", cand.entry.Name) - cand.result = NodeResult{Name: cand.entry.Name, Outcome: OutcomeOK, Status: status} + if status, err := node.Status(ctx); err == nil && w.matches(servingNames(status)...) { + log("%s was already started by someone else; waiting for its engine to answer...\n", cand.entry.Name) + ready, err := c.waitReady(ctx, node, cand.entry, w, log) + if err != nil { + return nil, err + } + cand.result = NodeResult{Name: cand.entry.Name, Outcome: OutcomeOK, Status: ready} return c.choiceFor(cand, w, true, engineKey) } } @@ -93,14 +144,15 @@ func (c *Config) Wake(ctx context.Context, w Want, dc remote.DeployConfig, resul // wakeable keeps the nodes that could be started, in the order to try them: a // node whose stored config already names the wanted model first, since it has // the weights and starts sooner. -func wakeable(cands []candidate, dc remote.DeployConfig) []candidate { +func wakeable(cands []candidate, resolver *configResolver) []candidate { var warm, cold []candidate for _, c := range cands { if !c.result.OK() || c.result.Status.State == string(daemon.StateRunning) { // A running engine is never displaced to make room. continue } - if c.result.Status.Model != "" && c.result.Status.Model == dc.ModelID { + dc, err := resolver.config(c.entry) + if err == nil && c.result.Status.Model != "" && c.result.Status.Model == dc.ModelID { warm = append(warm, c) continue } @@ -109,11 +161,50 @@ func wakeable(cands []candidate, dc remote.DeployConfig) []candidate { return append(warm, cold...) } +// WaitLoading holds a request for a node that is already running the wanted +// model but has not answered yet, and returns it once its engine does. It is +// the counterpart to Wake for a node nothing needs to start: the weights are +// already being fetched or loaded, so waiting reaches a served request sooner +// than starting a second engine somewhere else — and on a fleet whose other +// nodes cost money to run, it avoids starting one at all. +// +// Candidates are taken in fleet-file order. Only the first is waited for: +// they are all loading the same model, so a second wait would just be the +// first one's timeout twice over. +func (c *Config) WaitLoading(ctx context.Context, w Want, results []NodeResult, log Waker) (*Choice, error) { + if log == nil { + log = func(string, ...any) {} + } + cands := loading(c.candidates(results), w) + if len(cands) == 0 { + return nil, &ErrNoneServing{Results: results, Want: w, Path: c.Path} + } + cand := cands[0] + node, err := c.NewNode(cand.entry) + if err != nil { + return nil, err + } + log("%s is starting %s; waiting for its engine to answer...\n", cand.entry.Name, w.wanted()) + ready, err := c.waitReady(ctx, node, cand.entry, w, log) + if err != nil { + return nil, err + } + cand.result = NodeResult{Name: cand.entry.Name, Outcome: OutcomeOK, Status: ready} + // Not woken: this engine was started by whoever started it, so its key is + // looked up rather than being the one a wake just gated it with. + return c.choiceFor(cand, w, false, "") +} + // waitReady blocks until the node reports running *and* its engine answers. // The supervisor reports running when the process exists; llama.cpp then loads // weights, so a launch that trusted the state alone would hand the agent an // endpoint that refuses connections. // +// A daemon that reports its own readiness reading — the engine has answered +// its health check — is taken on that word; it checked from the same machine +// the engine runs on. A daemon that reports none (older builds, or a runner +// with no known health-check convention) falls back to the TCP probe. +// // On timeout the started engine is deliberately left running: it is probably // still loading, and stopping it throws away the only expensive part. func (c *Config) waitReady(ctx context.Context, node Node, entry NodeConfig, w Want, log Waker) (daemon.StatusResponse, error) { @@ -125,6 +216,9 @@ func (c *Config) waitReady(ctx context.Context, node Node, entry NodeConfig, w W if err == nil { last = status if status.State == string(daemon.StateRunning) { + if status.Ready == daemon.ReadyYes { + return status, nil + } baseURL, urlErr := c.EngineBaseURL(entry, status) if urlErr != nil { return status, urlErr @@ -145,7 +239,7 @@ func (c *Config) waitReady(ctx context.Context, node Node, entry NodeConfig, w W } if time.Now().After(deadline) { return last, fmt.Errorf( - "node %q did not answer within %s of being started: its engine may still be loading, "+ + "node %q did not answer within %s: its engine may still be loading, "+ "so it has been left running — check `spinloop fleet status`", entry.Name, WakeTimeout) } @@ -191,8 +285,8 @@ func isAlreadyRunning(err error) bool { // is what lets a routing decision be explained before an agent depends on it — // and the reason the ordering lives in one place rather than being described // twice. -func (c *Config) WouldWake(results []NodeResult, dc remote.DeployConfig) (NodeConfig, bool) { - cands := wakeable(c.candidates(results), dc) +func (c *Config) WouldWake(results []NodeResult, cfgFor ConfigFor) (NodeConfig, bool) { + cands := wakeable(c.candidates(results), newConfigResolver(cfgFor)) if len(cands) == 0 { return NodeConfig{}, false } diff --git a/internal/fleet/wake_test.go b/internal/fleet/wake_test.go index 25ecc3df..e10f81f2 100644 --- a/internal/fleet/wake_test.go +++ b/internal/fleet/wake_test.go @@ -3,6 +3,7 @@ package fleet import ( "context" "encoding/json" + "fmt" "net" "net/http" "net/http/httptest" @@ -29,6 +30,11 @@ type fakeNode struct { startStatus int // engineDelay is how long after starting before the engine listens. engineDelay time.Duration + // ready, when set, is what /v1/status reports for `ready`. + ready bool + // noEngine keeps the engine's listener down even after an accepted start: + // readiness can only come from the daemon's own reading. + noEngine bool // started records whether a start was accepted. started bool // pushed is the deploy config the start carried. @@ -62,6 +68,9 @@ func newFakeNode(t *testing.T, state, model string) *fakeNode { resp := daemon.StatusResponse{State: f.state, Model: f.model} if f.state == string(daemon.StateRunning) { resp.Engine = &daemon.EngineEndpoint{Port: f.enginePort} + if f.ready { + resp.Ready = "ready" + } } json.NewEncoder(w).Encode(resp) }) @@ -87,11 +96,13 @@ func newFakeNode(t *testing.T, state, model string) *fakeNode { if dc.ModelID != "" { f.model = dc.ModelID } - delay := f.engineDelay - go func() { - time.Sleep(delay) - f.listenAsEngine() - }() + if !f.noEngine { + delay := f.engineDelay + go func() { + time.Sleep(delay) + f.listenAsEngine() + }() + } json.NewEncoder(w).Encode(daemon.StatusResponse{State: f.state, Model: f.model}) }) f.srv = httptest.NewServer(mux) @@ -153,7 +164,7 @@ func TestWakeStartsAnIdleNode(t *testing.T) { cfg := fleetOf(t, []string{"box"}, node) dc := remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"} - choice, err := cfg.Wake(context.Background(), Want{Model: "qwen3-27b"}, dc, statusOf(t, cfg), nil) + choice, err := cfg.Wake(context.Background(), Want{Model: "qwen3-27b"}, ConstantConfig(dc, nil), statusOf(t, cfg), nil) if err != nil { t.Fatal(err) } @@ -179,7 +190,7 @@ func TestWakeSkipsANodeThatRefusesTheConfig(t *testing.T) { cfg := fleetOf(t, []string{"wrong-box", "right-box"}, refuses, accepts) choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err != nil { t.Fatal(err) } @@ -197,7 +208,7 @@ func TestWakeReportsEveryRefusal(t *testing.T) { cfg := fleetOf(t, []string{"a", "b"}, a, b) _, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "vllm", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "vllm", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err == nil { t.Fatal("expected a failure when every node refuses") } @@ -220,7 +231,7 @@ func TestWakeWaitsForTheEngineToAnswer(t *testing.T) { start := time.Now() choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), log) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), log) if err != nil { t.Fatal(err) } @@ -245,7 +256,7 @@ func TestWakeTimesOutWithoutStopping(t *testing.T) { cfg := fleetOf(t, []string{"stuck"}, node) _, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err == nil { t.Fatal("expected a timeout") } @@ -279,7 +290,7 @@ func TestWakeLosingTheRaceUsesTheNode(t *testing.T) { Status: daemon.StatusResponse{State: string(daemon.StateIdle)}, }} choice, err := cfg.Wake(context.Background(), Want{Model: "qwen3-27b"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"}, stale, nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"}, nil), stale, nil) if err != nil { t.Fatalf("losing the race should not fail the launch: %v", err) } @@ -295,7 +306,7 @@ func TestWakeNeverDisplacesARunningEngine(t *testing.T) { cfg := fleetOf(t, []string{"busy"}, busy) _, err := cfg.Wake(context.Background(), Want{Model: "mine"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "mine"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "mine"}, nil), statusOf(t, cfg), nil) if err == nil { t.Fatal("expected a failure rather than a restart") } @@ -318,7 +329,7 @@ func TestWakePrefersANodeThatAlreadyHasTheModel(t *testing.T) { cfg := fleetOf(t, []string{"cold", "warm"}, cold, warm) choice, err := cfg.Wake(context.Background(), Want{Model: "qwen3-27b"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"}, nil), statusOf(t, cfg), nil) if err != nil { t.Fatal(err) } @@ -343,7 +354,7 @@ func TestWakeGatesTheEngineWithTheClientsKey(t *testing.T) { cfg.Nodes[0].EngineTokenEnv = "BOX_ENGINE_KEY" choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err != nil { t.Fatal(err) } @@ -365,7 +376,7 @@ func TestWakeWithoutAKeyIsUngated(t *testing.T) { cfg := fleetOf(t, []string{"box"}, node) choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err != nil { t.Fatal(err) } @@ -393,6 +404,94 @@ func TestRemoteRefusesToBeWoken(t *testing.T) { } } +// engineUpAfter makes the "engine" of a node started by someone else come +// listening after d: the node already reports running, so the engine's delay +// is not the start's. +func (f *fakeNode) engineUpAfter(d time.Duration) { + go func() { + time.Sleep(d) + f.listenAsEngine() + }() +} + +// The per-candidate resolver lets different nodes be started with different +// configs — the gateway shape, where each node's own Spinloop source decides +// what it would run. +func TestWakeTakesPerCandidateConfigs(t *testing.T) { + shortWake(t) + a := newFakeNode(t, string(daemon.StateIdle), "") + b := newFakeNode(t, string(daemon.StateIdle), "") + cfg := fleetOf(t, []string{"a", "b"}, a, b) + + cfgFor := func(entry NodeConfig) (remote.DeployConfig, error) { + if entry.Name == "a" { + return remote.DeployConfig{}, fmt.Errorf("node %q names no Spinloop source", entry.Name) + } + return remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil + } + choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, cfgFor, statusOf(t, cfg), nil) + if err != nil { + t.Fatal(err) + } + if choice.Node.Name != "b" { + t.Errorf("chose %q, want the node whose source resolved", choice.Node.Name) + } + b.mu.Lock() + defer b.mu.Unlock() + if b.pushed == nil || b.pushed.ModelID != "m" { + t.Errorf("the node's own config did not reach it: %+v", b.pushed) + } +} + +// A node woken by someone else is not taken on state alone: its engine may +// still be loading, so the same readiness wait applies to a node we did not +// start ourselves. +func TestWakeWaitsForARacedNodeToAnswer(t *testing.T) { + shortWake(t) + node := newFakeNode(t, string(daemon.StateRunning), "qwen3-27b") + node.startErr = "an engine is already running" + node.startStatus = http.StatusConflict + node.engineUpAfter(150 * time.Millisecond) + cfg := fleetOf(t, []string{"contested"}, node) + + stale := []NodeResult{{ + Name: "contested", + Outcome: OutcomeOK, + Status: daemon.StatusResponse{State: string(daemon.StateIdle)}, + }} + start := time.Now() + choice, err := cfg.Wake(context.Background(), Want{Model: "qwen3-27b"}, + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "qwen3-27b"}, nil), stale, nil) + if err != nil { + t.Fatalf("losing the race should not fail the launch: %v", err) + } + if time.Since(start) < 150*time.Millisecond { + t.Error("returned before the raced node's engine was listening") + } + if choice.Node.Name != "contested" { + t.Errorf("chose %q", choice.Node.Name) + } +} + +// A daemon that reports its own readiness reading is trusted without a probe: +// it checked from the same machine the engine runs on. +func TestWakeTrustsTheDaemonReadinessReading(t *testing.T) { + shortWake(t) + node := newFakeNode(t, string(daemon.StateIdle), "") + node.ready = true + node.noEngine = true // the probe could never succeed; only the reading could + cfg := fleetOf(t, []string{"box"}, node) + + choice, err := cfg.Wake(context.Background(), Want{Model: "m"}, + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) + if err != nil { + t.Fatalf("the daemon's own readiness reading should have been enough: %v", err) + } + if choice.Node.Name != "box" { + t.Errorf("chose %q", choice.Node.Name) + } +} + // A variable that resolves to nothing fails before any engine is started. func TestWakeFailsOnAnUnresolvableKey(t *testing.T) { shortWake(t) @@ -401,7 +500,7 @@ func TestWakeFailsOnAnUnresolvableKey(t *testing.T) { cfg.Nodes[0].EngineTokenEnv = "NOWHERE_ENGINE_KEY" _, err := cfg.Wake(context.Background(), Want{Model: "m"}, - remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, statusOf(t, cfg), nil) + ConstantConfig(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}, nil), statusOf(t, cfg), nil) if err == nil { t.Fatal("expected a failure") } diff --git a/internal/gateway/gateway.go b/internal/gateway/gateway.go new file mode 100644 index 00000000..1020d387 --- /dev/null +++ b/internal/gateway/gateway.go @@ -0,0 +1,502 @@ +// The fleet gateway: one OpenAI-compatible endpoint in front of a fleet. It +// answers agent requests with the fleet's own selector, holds each node's +// engine key, and wakes a node when nothing is serving what a request asks +// for — so a machine running an agent needs nothing but a URL and one token. +// +// It is a foreground process, the way `spinloop serve` is: it holds the fleet +// file it serves, and a machine that hosts agents points its Spinloop's FLEET +// at the address it prints. + +package gateway + +import ( + "bytes" + "context" + "crypto/subtle" + "encoding/json" + "errors" + "fmt" + "io" + "log/slog" + "net" + "net/http" + "net/http/httputil" + "net/url" + "strings" + "sync" + "time" + + "github.com/spinloop-ai/spinloop/internal/daemon" + "github.com/spinloop-ai/spinloop/internal/fleet" + "github.com/spinloop-ai/spinloop/internal/remote" +) + +// DefaultListen is where the gateway answers when --listen is not given: a +// fixed port on every interface, so a Spinloop's FLEET can name one address +// without knowing the machine it lands on. +const DefaultListen = ":4000" + +// LoopbackListen is where `--loopback` binds the gateway: the default port on +// loopback, the safe bind a local-only gateway wants — one that Listen's token +// check accepts without a token. +const LoopbackListen = "127.0.0.1:4000" + +// cacheTTL is how long a fan-out reading stays fresh for routing. A burst of +// requests pays one fan-out, not one each; a request older than this sees the +// fleet as it is now. A variable so tests do not wait. +var cacheTTL = 2 * time.Second + +// sourcesTTL is how long the reading of what each node's own source describes +// stays fresh for the models list. A source changes on a deploy, not on a +// request, so this is far longer than the status reading's: the cost is +// reading each node's Spinloop and its preset, which a burst of models +// requests must not pay per request. A variable so tests do not wait. +var sourcesTTL = 30 * time.Second + +// pathsServed is the surface the gateway answers, for the 404 that names it. +var pathsServed = []string{"/v1/models", "/v1/chat/completions", "/v1/completions", "/health"} + +// Options shapes a handler beyond the fleet it serves. +type Options struct { + // ConfigFor resolves the deploy config a candidate node would be started + // with — each node's own Spinloop source. nil disables waking: nothing can + // be started, and a request nothing is serving fails saying so. + ConfigFor fleet.ConfigFor + // Log receives the gateway's log lines; nil discards them. + Log *slog.Logger + // Now is the clock the reading cache ages against; nil uses time.Now. + Now func() time.Time +} + +// Handler is the gateway: the fleet it serves, the token its callers present, +// and the state a burst of requests shares — the last reading of the fleet, +// and the reading's age. +type Handler struct { + cfg *fleet.Config + token string + cfgFor fleet.ConfigFor + log *slog.Logger + now func() time.Time + + mu sync.Mutex + results []fleet.NodeResult + at time.Time + wakeable map[string]string + wakeableAt time.Time +} + +// New builds a gateway handler over a resolved fleet file. The token is the +// one callers must present — empty on loopback, where none is needed, as the +// daemon's control API allows. +func New(cfg *fleet.Config, token string, opts Options) *Handler { + log := opts.Log + if log == nil { + log = slog.New(slog.DiscardHandler) + } + now := opts.Now + if now == nil { + now = time.Now + } + return &Handler{cfg: cfg, token: token, cfgFor: opts.ConfigFor, log: log, now: now} +} + +// Listen opens the gateway's listener, applying the daemon's exposure rule: +// a non-loopback address is refused without a token, because it would put an +// engine's full output on the network for anyone to read. +func Listen(addr, token string) (net.Listener, error) { + if token == "" && !loopbackAddr(addr) { + return nil, fmt.Errorf( + "refusing to listen on non-loopback %q without a token: "+ + "pass --api-token-file , set %s, or pass --api-token — "+ + "or bind loopback, e.g. --listen 127.0.0.1:4000, which needs none", + addr, daemon.TokenEnvVar) + } + return net.Listen("tcp", addr) +} + +// loopbackAddr reports whether a listen address binds only loopback. An empty +// or wildcard host binds every interface, so it is not loopback. +func loopbackAddr(addr string) bool { + host, _, err := net.SplitHostPort(addr) + if err != nil { + host = addr + } + if host == "" { + return false + } + if strings.EqualFold(host, "localhost") { + return true + } + ip := net.ParseIP(host) + return ip != nil && ip.IsLoopback() +} + +// ServeHTTP is the gateway's whole surface: the three paths it serves, a +// health check that touches no node, and a 404 that names the rest. +func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { + var handler http.HandlerFunc + switch { + case r.Method == http.MethodGet && r.URL.Path == "/health": + handler = h.handleHealth + case r.Method == http.MethodGet && r.URL.Path == "/v1/models": + handler = h.handleModels + case r.Method == http.MethodPost && (r.URL.Path == "/v1/chat/completions" || r.URL.Path == "/v1/completions"): + handler = h.handleCompletion + default: + handler = func(w http.ResponseWriter, r *http.Request) { + writeError(w, http.StatusNotFound, fmt.Errorf( + "the gateway serves %s, not %s %s", strings.Join(pathsServed, ", "), r.Method, r.URL.Path)) + } + } + h.authenticate(handler)(w, r) +} + +// authenticate gates every request behind the bearer token, on the daemon's +// terms: an empty token means no auth, which Listen permits on loopback only. +func (h *Handler) authenticate(next http.HandlerFunc) http.HandlerFunc { + if h.token == "" { + return next + } + want := []byte("Bearer " + h.token) + return func(w http.ResponseWriter, r *http.Request) { + got := []byte(r.Header.Get("Authorization")) + if subtle.ConstantTimeCompare(got, want) != 1 { + writeError(w, http.StatusUnauthorized, fmt.Errorf("missing or invalid bearer token")) + return + } + next(w, r) + } +} + +// handleHealth answers that the gateway is up. It contacts no node on +// purpose: it is how an operator tells the gateway down from the fleet down. +func (h *Handler) handleHealth(w http.ResponseWriter, _ *http.Request) { + writeJSON(w, http.StatusOK, map[string]any{"ok": true}) +} + +// handleModels lists what a request can reach, in the OpenAI list shape: what +// the running nodes report, and — when the fleet allows waking — what a +// stopped node's own source describes, so a client sees the models it can ask +// for. Served name first, duplicates once. Nothing reachable is an empty +// list, not an error. +func (h *Handler) handleModels(w http.ResponseWriter, r *http.Request) { + results := h.reading(r.Context()) + wakeable := h.wakeableModels() + seen := map[string]bool{} + data := []map[string]any{} + add := func(name string) { + if name == "" || seen[name] { + return + } + seen[name] = true + data = append(data, map[string]any{"id": name, "object": "model"}) + } + for _, res := range results { + if !res.OK() || res.Status.State != string(daemon.StateRunning) { + continue + } + name := res.Status.ServedName + if name == "" { + name = res.Status.Model + } + add(name) + } + for _, res := range results { + if !res.OK() || res.Status.State == string(daemon.StateRunning) { + // A running engine is never displaced to make room, so a running + // node's source's model is not one this gateway would answer; a + // node that does not answer its status cannot be started either. + continue + } + add(wakeable[res.Name]) + } + writeJSON(w, http.StatusOK, map[string]any{"object": "list", "data": data}) +} + +// wakeableModels is what a request could start: for each node, the model its +// own source describes, under the served-name-first naming a running node +// reports. It is resolved at most once per sourcesTTL, shared by every models +// request. The gateway cannot start a remote environment — one the fleet +// names by environment and a request never wakes — so a remote node +// contributes nothing here, and neither does any node when the fleet's wake is +// off or the gateway holds no way to resolve a source. +func (h *Handler) wakeableModels() map[string]string { + if h.cfgFor == nil || !h.cfg.Wakes() { + return nil + } + h.mu.Lock() + defer h.mu.Unlock() + if h.wakeable != nil && h.now().Sub(h.wakeableAt) < sourcesTTL { + return h.wakeable + } + m := map[string]string{} + for _, entry := range h.cfg.Nodes { + if entry.Kind != fleet.KindDaemon { + continue + } + dc, err := h.cfgFor(entry) + if err != nil { + continue + } + name := dc.ServedModelName + if name == "" { + name = dc.ModelID + } + if name != "" { + m[entry.Name] = name + } + } + h.wakeable = m + h.wakeableAt = h.now() + return m +} + +// reading returns the fleet's last fan-out, taking one when the last is stale. +// It is the whole freshness story: a burst of requests shares one reading, +// and a request older than cacheTTL sees the fleet as it is now. The mutex is +// held for the fan-out, so requests that arrive mid-fan-out wait for it and +// take its result rather than fanning out again. +func (h *Handler) reading(ctx context.Context) []fleet.NodeResult { + h.mu.Lock() + defer h.mu.Unlock() + if h.results != nil && h.now().Sub(h.at) < cacheTTL { + return h.results + } + h.results = h.cfg.FanOut(ctx, fleet.StatusCall) + h.at = h.now() + return h.results +} + +// handleCompletion routes a completion request to the node serving its model, +// waking one when nothing is and the fleet file allows it. +func (h *Handler) handleCompletion(w http.ResponseWriter, r *http.Request) { + model, body, err := requestModel(r) + if err != nil { + writeError(w, http.StatusBadRequest, err) + return + } + if model == "" { + writeError(w, http.StatusBadRequest, fmt.Errorf( + "the request names no model: completion requests need a `model` field")) + return + } + ctx := r.Context() + prefer, err := h.cfg.Preference("") + if err != nil { + writeError(w, http.StatusBadRequest, err) + return + } + want := fleet.Want{Model: model, Prefer: prefer} + + results := h.reading(ctx) + choice, err := h.cfg.Choose(h.reachable(results), want) + if err != nil { + var none *fleet.ErrNoneServing + if !errors.As(err, &none) { + writeError(w, http.StatusBadGateway, err) + return + } + choice, err = h.routeNothingServing(ctx, want, none) + if err != nil { + writeError(w, http.StatusServiceUnavailable, err) + return + } + } + + h.log.Info("routed", + slog.String("model", model), + slog.String("node", choice.Node.Name), + slog.Bool("woken", choice.Woken)) + h.proxy(w, r, body, choice) +} + +// requestModel pulls the model field out of a completion request and returns +// it with the full body, which the proxy must forward unmodified. A body that +// is not a JSON object fails saying so, rather than being routed at a guess. +func requestModel(r *http.Request) (string, []byte, error) { + body, err := io.ReadAll(io.LimitReader(r.Body, 1<<20)) + if err != nil { + return "", nil, fmt.Errorf("reading the request: %w", err) + } + var req struct { + Model string `json:"model"` + } + if len(bytes.TrimSpace(body)) > 0 { + if err := json.Unmarshal(body, &req); err != nil { + return "", nil, fmt.Errorf("the request is not a JSON body: %v", err) + } + } + return req.Model, body, nil +} + +// reachable marks the running engines the gateway cannot reach as not +// candidates: a node whose engine is bound to loopback, without an override +// taking responsibility for reachability, answers only on its own machine, so +// selecting it would hold a request for the wake timeout at best. The mark +// carries the daemon's own explanation, which names the bind and the fix, so +// a fleet where that is the only match fails saying so. +func (h *Handler) reachable(results []fleet.NodeResult) []fleet.NodeResult { + out := make([]fleet.NodeResult, len(results)) + for i, res := range results { + out[i] = res + if !res.OK() || res.Status.State != string(daemon.StateRunning) { + continue + } + entry, ok := h.cfg.Node(res.Name) + if !ok { + continue + } + if _, err := h.cfg.EngineBaseURL(entry, res.Status); err != nil { + out[i] = fleet.NodeResult{ + Name: res.Name, + Outcome: fleet.OutcomeUnreachable, + Err: err, + Status: res.Status, + At: res.At, + } + } + } + return out +} + +// routeNothingServing answers a request that nothing in the fleet is serving. +// A node already running the wanted model whose engine has not answered yet is +// waited for rather than passed over: it is loading the weights this request +// needs, so it serves sooner than a node started from cold, and waking a +// second node would leave two engines up for one request. +func (h *Handler) routeNothingServing(ctx context.Context, want fleet.Want, none *fleet.ErrNoneServing) (*fleet.Choice, error) { + if starting := h.cfg.Loading(none.Results, want); len(starting) > 0 { + h.log.Info("waiting for a starting engine", + slog.String("model", want.Model), + slog.String("node", starting[0].Name)) + return h.cfg.WaitLoading(ctx, want, none.Results, h.progress()) + } + return h.wakeFor(ctx, want, none.Results) +} + +// progress adapts the fleet's progress reporting to the gateway's log, which +// is where a held request's waiting has to be visible: the caller is given +// nothing until the request completes. +func (h *Handler) progress() fleet.Waker { + return func(format string, args ...any) { + h.log.Info(strings.TrimSuffix(fmt.Sprintf(format, args...), "\n")) + } +} + +// wakeFor starts a node for a request nothing is serving, when the fleet file +// allows it, and holds the request until the engine answers. A concurrent +// request waking the same node loses its start to the daemon's 409 and takes +// the node the other one started — the same engine, the same wait. +func (h *Handler) wakeFor(ctx context.Context, want fleet.Want, results []fleet.NodeResult) (*fleet.Choice, error) { + none := &fleet.ErrNoneServing{Results: results, Want: want, Path: h.cfg.Path} + if h.cfgFor == nil { + return nil, fmt.Errorf("%s\nthis gateway can wake no node: it has no way to resolve a node's Spinloop source", none) + } + if !h.cfg.Wakes() { + return nil, h.refuseWake(want, none) + } + return h.cfg.Wake(ctx, want, h.matchingConfigFor(want.Model), results, h.progress()) +} + +// refuseWake is the wake-off answer: nothing is started, and the failure names +// the node whose source describes the model and the command that would start +// it — or, when no source describes it, that there is nothing to start. +func (h *Handler) refuseWake(want fleet.Want, none error) error { + cfgFor := h.matchingConfigFor(want.Model) + for _, entry := range h.cfg.Nodes { + if _, err := cfgFor(entry); err == nil { + return fmt.Errorf("%s\nwake is off in %s: %q's source describes %s; start it with `spinloop fleet start %s`", + none, h.cfg.Path, entry.Name, want.Model, entry.Name) + } + } + return fmt.Errorf("%s\nwake is off in %s, and no node's source describes %s", none, h.cfg.Path, want.Model) +} + +// matchingConfigFor wraps the per-node source resolver with the one condition +// a wake has to meet: the source's config is the model the request asks for. +// A node whose source describes a different model is not a candidate — it +// would be started with the wrong engine — and its refusal says so. +func (h *Handler) matchingConfigFor(model string) fleet.ConfigFor { + base := h.cfgFor + return func(entry fleet.NodeConfig) (remote.DeployConfig, error) { + dc, err := base(entry) + if err != nil { + return dc, err + } + if dc.ModelID != model && dc.ServedModelName != model { + described := dc.ServedModelName + if described == "" { + described = dc.ModelID + } + return dc, fmt.Errorf("its source describes %s, not %s", described, model) + } + return dc, nil + } +} + +// proxy forwards the request to the chosen engine and the engine's reply back +// to the caller, unmodified: the body goes out as it came in, a streamed +// reply passes through as it is produced, and the reply the engine gives is +// the reply the caller gets — the gateway never retries another node. +func (h *Handler) proxy(w http.ResponseWriter, r *http.Request, body []byte, choice *fleet.Choice) { + target, err := url.Parse(choice.BaseURL) + if err != nil { + writeError(w, http.StatusBadGateway, fmt.Errorf("node %q's engine address is not a URL: %v", choice.Node.Name, err)) + return + } + p := &httputil.ReverseProxy{ + Rewrite: func(pr *httputil.ProxyRequest) { + pr.SetURL(target) + // The caller asks the gateway for /v1/; the engine's base + // URL already carries its own prefix, so /v1 comes off the + // request and the rest goes on the base. + pr.Out.URL.Path = joinPath(target.Path, strings.TrimPrefix(pr.In.URL.Path, "/v1")) + pr.Out.URL.RawQuery = pr.In.URL.RawQuery + }, + // -1 flushes after every write: a streamed reply reaches the caller + // as the engine produces it, not when a buffer fills. + FlushInterval: -1, + // The default transport, with its connection pooling and no overall + // request timeout: a completion may take as long as the model takes. + ErrorHandler: func(w http.ResponseWriter, r *http.Request, err error) { + h.log.Error("route failed", + slog.String("node", choice.Node.Name), + slog.String("error", err.Error())) + writeError(w, http.StatusBadGateway, fmt.Errorf( + "the engine on %s failed to answer: %v", choice.Node.Name, err)) + }, + } + + out := r.Clone(r.Context()) + out.Body = io.NopCloser(bytes.NewReader(body)) + out.ContentLength = int64(len(body)) + if choice.APIKey != "" { + // The caller's authoriser never travels past the gateway: the engine + // is reached with the key its fleet entry names, resolved the way + // every other fleet client resolves it. + out.Header.Set("Authorization", "Bearer "+choice.APIKey) + } else { + out.Header.Del("Authorization") + } + p.ServeHTTP(w, out) +} + +// joinPath joins a base path and a request path with at most one slash +// between them, either piece absent. +func joinPath(base, rest string) string { + base = strings.TrimRight(base, "/") + rest = "/" + strings.TrimLeft(rest, "/") + return base + rest +} + +// writeJSON sends a JSON reply. +func writeJSON(w http.ResponseWriter, status int, v any) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + json.NewEncoder(w).Encode(v) +} + +// writeError sends a JSON error in the shape the OpenAI surface uses. +func writeError(w http.ResponseWriter, status int, err error) { + writeJSON(w, status, map[string]any{"error": map[string]any{"message": err.Error(), "type": "gateway_error"}}) +} diff --git a/internal/gateway/gateway_test.go b/internal/gateway/gateway_test.go new file mode 100644 index 00000000..ce648f30 --- /dev/null +++ b/internal/gateway/gateway_test.go @@ -0,0 +1,1020 @@ +package gateway + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "log/slog" + "net" + "net/http" + "net/http/httptest" + "slices" + "strconv" + "strings" + "sync" + "testing" + "time" + + "github.com/spinloop-ai/spinloop/internal/daemon" + "github.com/spinloop-ai/spinloop/internal/fleet" + "github.com/spinloop-ai/spinloop/internal/remote" +) + +// fakeNode is one machine: its daemon's control API and, once running, its +// engine's real HTTP endpoint on the port the daemon reports — so the +// gateway's proxy path is exercised end to end against a listener, not a +// stub. The engine's port is reserved at construction and occupied only when +// the engine is meant to be up, which is what readiness means here. +type fakeNode struct { + mu sync.Mutex + // state and what the daemon reports serving. + state string + model string + servedName string + // ready is what the daemon reports for `ready`: daemon.ReadyYes, + // daemon.ReadyNo, or empty for a daemon whose reading has not landed — + // an older build, or a runner with no known health-check convention. + ready string + // engineAuth, when set, is what the engine requires as its key. + engineAuth string + // startErr and startStatus are a refused start's reply. + startErr string + startStatus int + // engineDelay is how long after a start the engine listens. + engineDelay time.Duration + // noEngine keeps the engine down even after an accepted start: + // readiness can only come from the daemon's own reading. + noEngine bool + // loopbackOnly is what the daemon reports for its engine's bind. + loopbackOnly bool + // started counts accepted starts; pushed is the last config and key. + started int + pushed *remote.DeployConfig + pushedKey string + // engineGotAuth is the last authorisation the engine itself saw. + engineGotAuth string + // statusHits counts status calls, so a burst's fan-out is countable. + statusHits int + + daemonSrv *httptest.Server + engine *http.Server + engineLn net.Listener + enginePort int +} + +func newFakeNode(t *testing.T, state, model string) *fakeNode { + t.Helper() + f := &fakeNode{state: state, model: model} + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + f.enginePort = ln.Addr().(*net.TCPAddr).Port + ln.Close() // free it until the engine is meant to be up + f.engine = &http.Server{Handler: f.engineHandler()} + // A node already running has its engine up: its port answers. + if state == string(daemon.StateRunning) { + f.upAsEngine() + } + f.daemonSrv = httptest.NewServer(f.daemonMux()) + t.Cleanup(func() { + f.daemonSrv.Close() + f.mu.Lock() + defer f.mu.Unlock() + if f.engineLn != nil { + f.engineLn.Close() + } + }) + return f +} + +// engineHandler is the engine itself: it checks its key, streams when asked, +// and fails when told to. +func (f *fakeNode) engineHandler() http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + f.engineGotAuth = r.Header.Get("Authorization") + auth := f.engineAuth + f.mu.Unlock() + if auth != "" && f.engineGotAuth != "Bearer "+auth { + w.WriteHeader(http.StatusUnauthorized) + fmt.Fprint(w, `{"error":"bad key"}`) + return + } + body, _ := io.ReadAll(r.Body) + switch { + case strings.Contains(string(body), `"stream":true`): + w.Header().Set("Content-Type", "text/event-stream") + fl := w.(http.Flusher) + for _, chunk := range []string{"data: one\n\n", "data: two\n\n", "data: [DONE]\n\n"} { + io.WriteString(w, chunk) + fl.Flush() + } + case strings.Contains(string(body), `"fail":true`): + w.WriteHeader(http.StatusInternalServerError) + fmt.Fprint(w, `{"error":"the engine is on fire"}`) + default: + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"id":"cmpl-1","object":"chat.completion","choices":[{"message":{"role":"assistant","content":"hello"}}]}`) + } + }) +} + +func (f *fakeNode) daemonMux() *http.ServeMux { + mux := http.NewServeMux() + mux.HandleFunc("/v1/status", func(w http.ResponseWriter, _ *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + f.statusHits++ + resp := daemon.StatusResponse{State: f.state, Model: f.model, ServedName: f.servedName} + if f.state == string(daemon.StateRunning) { + resp.Engine = &daemon.EngineEndpoint{ + Port: f.enginePort, + LoopbackOnly: f.loopbackOnly, + RequiresKey: f.engineAuth != "", + } + resp.Ready = f.ready + } + json.NewEncoder(w).Encode(resp) + }) + mux.HandleFunc("/v1/start", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + if f.started > 0 && f.state == string(daemon.StateRunning) && f.startErr == "" { + // The daemon's own conflict: another start while one runs. + w.WriteHeader(http.StatusConflict) + json.NewEncoder(w).Encode(daemon.Error{Error: "an engine is already running"}) + f.mu.Unlock() + return + } + if f.startErr != "" { + status := f.startStatus + if status == 0 { + status = http.StatusBadRequest + } + w.WriteHeader(status) + json.NewEncoder(w).Encode(daemon.Error{Error: f.startErr}) + f.mu.Unlock() + return + } + var req daemon.StartRequest + json.NewDecoder(r.Body).Decode(&req) + dc := req.DeployConfig + f.pushed = &dc + f.pushedKey = req.EngineAPIKey + f.started++ + f.state = string(daemon.StateRunning) + if dc.ModelID != "" { + f.model = dc.ModelID + } + f.servedName = dc.ServedModelName + delay := f.engineDelay + noEngine := f.noEngine + w.WriteHeader(http.StatusOK) + json.NewEncoder(w).Encode(daemon.StatusResponse{State: f.state, Model: f.model}) + f.mu.Unlock() + if !noEngine { + go func() { + time.Sleep(delay) + f.upAsEngine() + }() + } + }) + return mux +} + +// upAsEngine occupies the reserved port, which readiness probes and the proxy +// both dial. +func (f *fakeNode) upAsEngine() { + ln, err := net.Listen("tcp", net.JoinHostPort("127.0.0.1", strconv.Itoa(f.enginePort))) + if err != nil { + return + } + f.mu.Lock() + f.engineLn = ln + f.mu.Unlock() + go f.engine.Serve(ln) +} + +// nodeConfig is the fleet-file entry pointing at this fake. +func (f *fakeNode) nodeConfig(name string) fleet.NodeConfig { + host, port, _ := net.SplitHostPort(strings.TrimPrefix(f.daemonSrv.URL, "http://")) + p, _ := strconv.Atoi(port) + return fleet.NodeConfig{Name: name, Host: host, Port: p, Kind: fleet.KindDaemon} +} + +// fleetOf builds a Config over the fakes, in the order given. +func fleetOf(t *testing.T, names []string, nodes ...*fakeNode) *fleet.Config { + t.Helper() + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir()} + for i, n := range nodes { + cfg.Nodes = append(cfg.Nodes, n.nodeConfig(names[i])) + } + return cfg +} + +// cfgForOf is a per-node source resolver from a table: what each node's own +// Spinloop source would resolve to, or its refusal. +func cfgForOf(t *testing.T, table map[string]remote.DeployConfig, refused map[string]string) fleet.ConfigFor { + return func(entry fleet.NodeConfig) (remote.DeployConfig, error) { + if msg, ok := refused[entry.Name]; ok { + return remote.DeployConfig{}, fmt.Errorf("%s", msg) + } + dc, ok := table[entry.Name] + if !ok { + return remote.DeployConfig{}, fmt.Errorf("node %q names no Spinloop source: no `file` field, no alias, no subdirectory", entry.Name) + } + return dc, nil + } +} + +// post sends one completion request to a handler and returns the reply. +func post(t *testing.T, h http.Handler, token string, body string) (*http.Response, string) { + t.Helper() + req := httptest.NewRequest(http.MethodPost, "http://gw/v1/chat/completions", strings.NewReader(body)) + if token != "" { + req.Header.Set("Authorization", "Bearer "+token) + } + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + data, _ := io.ReadAll(rec.Result().Body) + return rec.Result(), string(data) +} + +// --- the surface ---------------------------------------------------------- + +func TestListenRefusesTokenlessNonLoopback(t *testing.T) { + _, err := Listen("0.0.0.0:0", "") + if err == nil { + t.Fatal("a tokenless non-loopback listen should be refused") + } + for _, want := range []string{"--api-token-file", "SPINLOOP_API_TOKEN", "--api-token"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("error should name %q, got: %v", want, err) + } + } + ln, err := Listen("127.0.0.1:0", "") + if err != nil { + t.Fatalf("a tokenless loopback listen is allowed: %v", err) + } + ln.Close() +} + +func TestCallerAuthentication(t *testing.T) { + cfg := fleetOf(t, []string{"box"}, newFakeNode(t, string(daemon.StateIdle), "")) + h := New(cfg, "secret", Options{}) + + // Missing and wrong tokens are 401, and no node is contacted. + for _, tok := range []string{"", "wrong"} { + req := httptest.NewRequest(http.MethodGet, "http://gw/health", nil) + if tok != "" { + req.Header.Set("Authorization", "Bearer "+tok) + } + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusUnauthorized { + t.Fatalf("token %q: HTTP %d, want 401", tok, rec.Code) + } + } + + // The right one is through. + req := httptest.NewRequest(http.MethodGet, "http://gw/health", nil) + req.Header.Set("Authorization", "Bearer secret") + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("right token: HTTP %d, want 200", rec.Code) + } +} + +func TestTokenlessLoopbackServesWithoutAuth(t *testing.T) { + cfg := fleetOf(t, []string{"box"}, newFakeNode(t, string(daemon.StateIdle), "")) + h := New(cfg, "", Options{}) + req := httptest.NewRequest(http.MethodGet, "http://gw/health", nil) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("HTTP %d, want 200 with no token", rec.Code) + } +} + +func TestHealthTouchesNoNode(t *testing.T) { + // A node on a dead port: unreachable, but health must not care. + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + port := ln.Addr().(*net.TCPAddr).Port + ln.Close() + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir(), Nodes: []fleet.NodeConfig{ + {Name: "down", Host: "127.0.0.1", Port: port, Kind: fleet.KindDaemon}, + }} + h := New(cfg, "", Options{}) + req := httptest.NewRequest(http.MethodGet, "http://gw/health", nil) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("health with every node down: HTTP %d, want 200", rec.Code) + } +} + +func TestUnknownPathNamesTheSurface(t *testing.T) { + cfg := fleetOf(t, []string{"box"}, newFakeNode(t, string(daemon.StateIdle), "")) + h := New(cfg, "", Options{}) + req := httptest.NewRequest(http.MethodGet, "http://gw/v1/embeddings", nil) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusNotFound { + t.Fatalf("HTTP %d, want 404", rec.Code) + } + body := rec.Body.String() + for _, want := range []string{"/v1/models", "/v1/chat/completions", "/v1/completions", "/health"} { + if !strings.Contains(body, want) { + t.Errorf("404 should name %s, got: %s", want, body) + } + } +} + +// --- models ---------------------------------------------------------------- + +// modelsList makes one models request and returns the ids it lists. +func modelsList(t *testing.T, h http.Handler) []string { + t.Helper() + req := httptest.NewRequest(http.MethodGet, "http://gw/v1/models", nil) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("HTTP %d, want 200", rec.Code) + } + var list struct { + Data []map[string]any `json:"data"` + } + json.NewDecoder(rec.Body).Decode(&list) + var ids []string + for _, m := range list.Data { + ids = append(ids, m["id"].(string)) + } + return ids +} + +func TestModelsListsWhatIsRunning(t *testing.T) { + aliased := newFakeNode(t, string(daemon.StateRunning), "org/model") + aliased.servedName = "the-alias" + plain := newFakeNode(t, string(daemon.StateRunning), "org/other") + stopped := newFakeNode(t, string(daemon.StateStopped), "org/stale") + cfg := fleetOf(t, []string{"aliased", "plain", "stopped"}, aliased, plain, stopped) + h := New(cfg, "", Options{}) + + ids := modelsList(t, h) + if !slices.Contains(ids, "the-alias") || !slices.Contains(ids, "org/other") { + t.Errorf("models should list the running names, got %v", ids) + } + if slices.Contains(ids, "org/stale") { + t.Error("a stopped node with no resolved source contributes nothing") + } +} + +func TestModelsEmptyWhenNothingIsReachable(t *testing.T) { + cfg := fleetOf(t, []string{"box"}, newFakeNode(t, string(daemon.StateIdle), "")) + h := New(cfg, "", Options{}) + if ids := modelsList(t, h); len(ids) != 0 { + t.Errorf("nothing reachable is an empty list, not %v", ids) + } +} + +// With waking allowed, a node that is not running contributes the model its +// own source describes — the served name when the source gives one — so a +// client sees the models it can ask for, not only the ones answering now. +func TestModelsListsWhatAWakeCanStart(t *testing.T) { + live := newFakeNode(t, string(daemon.StateRunning), "org/live") + cold := newFakeNode(t, string(daemon.StateStopped), "") + cfg := fleetOf(t, []string{"live", "cold"}, live, cold) + h := New(cfg, "", Options{ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{ + "live": {Runner: "llamacpp", ModelID: "org/live"}, + "cold": {Runner: "llamacpp", ModelID: "org/cold", ServedModelName: "cold"}, + }, + nil)}) + + ids := modelsList(t, h) + if !slices.Contains(ids, "org/live") || !slices.Contains(ids, "cold") { + t.Errorf("models should list what runs and what a wake can start, got %v", ids) + } + if slices.Contains(ids, "org/cold") { + t.Error("the source's served name, not its model id, is what a request names") + } +} + +// A running engine is never displaced to make room, so a running node lists +// what it reports and nothing its source describes. +func TestRunningNodeListsOnlyWhatItReports(t *testing.T) { + node := newFakeNode(t, string(daemon.StateRunning), "org/live") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/other"}}, + nil)}) + if ids := modelsList(t, h); len(ids) != 1 || ids[0] != "org/live" { + t.Errorf("a running node lists what it reports, got %v", ids) + } +} + +// With the fleet's wake off, nothing is started, so nothing is listed beyond +// what answers now. +func TestModelsListsOnlyWhatRunsWhenWakeIsOff(t *testing.T) { + cold := newFakeNode(t, string(daemon.StateStopped), "") + cfg := fleetOf(t, []string{"cold"}, cold) + cfg.WakePolicy = fleet.WakeOff + h := New(cfg, "", Options{ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"cold": {Runner: "llamacpp", ModelID: "org/cold"}}, + nil)}) + if ids := modelsList(t, h); len(ids) != 0 { + t.Errorf("with wake off a stopped node's model is not listed, got %v", ids) + } +} + +// The gateway cannot start a remote environment — a request never wakes one — +// so its source's model is not wakeable and not listed. +func TestModelsLeavesOutARemoteEnvironment(t *testing.T) { + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir(), Nodes: []fleet.NodeConfig{ + {Name: "env", Kind: fleet.KindRemote}, + }} + h := New(cfg, "", Options{ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"env": {Runner: "llamacpp", ModelID: "org/cold"}}, + nil)}) + if m := h.wakeableModels(); len(m) != 0 { + t.Errorf("a remote environment's source is not a request's, got %v", m) + } +} + +// Two sources describing one model list it once. +func TestModelsListsASharedSourceModelOnce(t *testing.T) { + a := newFakeNode(t, string(daemon.StateStopped), "") + b := newFakeNode(t, string(daemon.StateStopped), "") + cfg := fleetOf(t, []string{"a", "b"}, a, b) + h := New(cfg, "", Options{ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{ + "a": {Runner: "llamacpp", ModelID: "org/same"}, + "b": {Runner: "llamacpp", ModelID: "org/same"}, + }, + nil)}) + ids := modelsList(t, h) + if len(ids) != 1 || ids[0] != "org/same" { + t.Errorf("a model two sources describe is listed once, got %v", ids) + } +} + +// A burst of models requests resolves each node's source once; a request +// after the reading ages resolves it again. +func TestModelsResolvesEachSourceOnce(t *testing.T) { + cold := newFakeNode(t, string(daemon.StateStopped), "") + cfg := fleetOf(t, []string{"cold"}, cold) + now := time.Now() + calls := 0 + h := New(cfg, "", Options{ + Now: func() time.Time { return now }, + ConfigFor: func(entry fleet.NodeConfig) (remote.DeployConfig, error) { + calls++ + return remote.DeployConfig{Runner: "llamacpp", ModelID: "org/cold"}, nil + }, + }) + for range 3 { + modelsList(t, h) + } + if calls != 1 { + t.Fatalf("a burst of three resolved the source %d times, want once", calls) + } + now = now.Add(sourcesTTL + time.Second) + modelsList(t, h) + if calls != 2 { + t.Fatalf("the aged reading should have resolved the source once more, got %d total", calls) + } +} + +// --- routing ---------------------------------------------------------------- + +func TestRequestGoesToTheNodeServingItsModel(t *testing.T) { + t.Setenv("RIGHT_ENGINE_KEY", "engine-key") + right := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + wrong := newFakeNode(t, string(daemon.StateRunning), "org/other") + cfg := fleetOf(t, []string{"right", "wrong"}, right, wrong) + cfg.Nodes[0].EngineTokenEnv = "RIGHT_ENGINE_KEY" + right.engineAuth = "engine-key" + h := New(cfg, "caller-token", Options{}) + + resp, body := post(t, h, "caller-token", `{"model":"org/wanted","messages":[]}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + if !strings.Contains(body, "hello") { + t.Errorf("the engine's reply did not reach the caller: %s", body) + } + right.mu.Lock() + defer right.mu.Unlock() + // The engine got its own key, not the caller's token. + if right.engineGotAuth != "Bearer engine-key" { + t.Errorf("engine saw %q, want the fleet's key", right.engineGotAuth) + } + if strings.Contains(right.engineGotAuth, "caller-token") { + t.Error("the caller's token travelled past the gateway") + } +} + +func TestRequestNamesNoModel(t *testing.T) { + node := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{}) + resp, body := post(t, h, "", `{"messages":[]}`) + if resp.StatusCode != http.StatusBadRequest { + t.Fatalf("HTTP %d, want 400", resp.StatusCode) + } + if !strings.Contains(body, "no model") { + t.Errorf("the refusal should say the request names no model: %s", body) + } + node.mu.Lock() + defer node.mu.Unlock() + if node.statusHits != 0 { + t.Error("a refused request contacted a node") + } +} + +func TestStreamedReplyPassesThrough(t *testing.T) { + node := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{}) + resp, body := post(t, h, "", `{"model":"org/wanted","stream":true}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + if resp.Header.Get("Content-Type") != "text/event-stream" { + t.Errorf("content type = %q, want the engine's stream type", resp.Header.Get("Content-Type")) + } + for _, want := range []string{"data: one", "data: two", "data: [DONE]"} { + if !strings.Contains(body, want) { + t.Errorf("stream lost %q: %s", want, body) + } + } +} + +func TestEngineRefusalIsTheCallersError(t *testing.T) { + t.Setenv("KEY", "k") + failing := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + other := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"failing", "other"}, failing, other) + h := New(cfg, "", Options{}) + + resp, body := post(t, h, "", `{"model":"org/wanted","fail":true}`) + if resp.StatusCode != http.StatusInternalServerError { + t.Fatalf("the engine's 500 is the caller's: HTTP %d, body %s", resp.StatusCode, body) + } + if !strings.Contains(body, "on fire") { + t.Errorf("the engine's own refusal should reach the caller: %s", body) + } + other.mu.Lock() + defer other.mu.Unlock() + // The other node was never tried: its engine saw no request at all. + if other.engineGotAuth != "" { + t.Error("a failed request was retried at another node") + } +} + +func TestEngineDownFailsNamingTheNode(t *testing.T) { + node := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{}) + node.mu.Lock() + node.engineLn.Close() // the engine dies while the daemon still reports it + node.mu.Unlock() + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusBadGateway { + t.Fatalf("HTTP %d, want 502: %s", resp.StatusCode, body) + } + if !strings.Contains(body, "box") { + t.Errorf("the failure should name the node: %s", body) + } +} + +// A node reached by a non-loopback name with a loopback-bound engine, without +// an override, is not a candidate; the mark carries the daemon's explanation, +// which names the bind and the fix. +func TestReachableMarksLoopbackBoundEngines(t *testing.T) { + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir(), Nodes: []fleet.NodeConfig{ + {Name: "bound", Host: "remote-box", Kind: fleet.KindDaemon, Port: 14242}, + {Name: "overridden", Host: "remote-box", Kind: fleet.KindDaemon, Port: 14243, + Engine: &fleet.EngineOverride{Host: "proxy.local", Port: 9000, Path: "/v1"}}, + }} + h := New(cfg, "", Options{}) + results := []fleet.NodeResult{ + {Name: "bound", Outcome: fleet.OutcomeOK, Status: daemon.StatusResponse{ + State: string(daemon.StateRunning), Model: "m", + Engine: &daemon.EngineEndpoint{Port: 8080, LoopbackOnly: true}, + }}, + {Name: "overridden", Outcome: fleet.OutcomeOK, Status: daemon.StatusResponse{ + State: string(daemon.StateRunning), Model: "m", + Engine: &daemon.EngineEndpoint{Port: 8080, LoopbackOnly: true}, + }}, + } + marked := h.reachable(results) + if marked[0].OK() { + t.Fatal("a loopback-bound engine without an override is not a candidate") + } + if !strings.Contains(marked[0].Err.Error(), "loopback") { + t.Errorf("the mark should carry the explanation naming the bind: %v", marked[0].Err) + } + if !marked[1].OK() { + t.Errorf("an override takes responsibility for reachability: %v", marked[1].Err) + } +} + +// With nothing else matching, the failure the gateway gives names the bind +// and the fix rather than a bare "no node serving". +func TestLoopbackBoundEngineFailsNamingTheFix(t *testing.T) { + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir(), Nodes: []fleet.NodeConfig{ + {Name: "bound", Host: "remote-box", Kind: fleet.KindDaemon, Port: 14242}, + }} + h := New(cfg, "", Options{}) + results := []fleet.NodeResult{ + {Name: "bound", Outcome: fleet.OutcomeOK, Status: daemon.StatusResponse{ + State: string(daemon.StateRunning), Model: "org/wanted", + Engine: &daemon.EngineEndpoint{Port: 8080, LoopbackOnly: true}, + }}, + } + _, err := cfg.Choose(h.reachable(results), fleet.Want{Model: "org/wanted"}) + if err == nil { + t.Fatal("nothing reachable serves the model, so the request must fail") + } + for _, want := range []string{"bound", "loopback"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("failure should name %q: %v", want, err) + } + } +} + +// A remote environment's engine address arrives as its status's reported host — +// the control plane's published endpoint — so it is a candidate, not marked +// unreachable the way a loopback-bound engine is. Without this the gateway +// could list a remote node's model and then refuse to route a request to it. +func TestReachableRoutesToARemoteNode(t *testing.T) { + t.Setenv("TEST_ENGINE_KEY", "secret") + cfg := &fleet.Config{Path: "fleet.yaml", Dir: t.TempDir(), APIKeyEnv: "TEST_ENGINE_KEY", Nodes: []fleet.NodeConfig{ + {Name: "env", Kind: fleet.KindRemote}, + }} + h := New(cfg, "", Options{}) + results := []fleet.NodeResult{ + {Name: "env", Outcome: fleet.OutcomeOK, Status: daemon.StatusResponse{ + State: string(daemon.StateRunning), + Model: "org/wanted", + ServedName: "wanted", + Engine: &daemon.EngineEndpoint{Host: "1.2.3.4", Port: 8000, Path: "/v1"}, + }}, + } + choice, err := cfg.Choose(h.reachable(results), fleet.Want{Model: "wanted"}) + if err != nil { + t.Fatalf("a request for the remote node's model should route to it: %v", err) + } + if choice.Node.Name != "env" { + t.Errorf("choice = %q, want env", choice.Node.Name) + } + if choice.BaseURL != "http://1.2.3.4:8000/v1" { + t.Errorf("base URL = %q, want the control plane's published address", choice.BaseURL) + } + if choice.APIKey != "secret" { + t.Errorf("API key = %q, want the resolved engine key", choice.APIKey) + } +} + +func TestBurstFansOutOnce(t *testing.T) { + node := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + now := time.Now() + h := New(cfg, "", Options{Now: func() time.Time { return now }}) + + for i := range 3 { + resp, _ := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("request %d: HTTP %d", i, resp.StatusCode) + } + } + node.mu.Lock() + hits := node.statusHits + node.mu.Unlock() + if hits != 1 { + t.Fatalf("a burst of three fanned out %d times, want once", hits) + } + + // A request after the reading ages fans out again. + now = now.Add(3 * time.Second) + resp, _ := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("stale-reading request: HTTP %d", resp.StatusCode) + } + node.mu.Lock() + hits = node.statusHits + node.mu.Unlock() + if hits != 2 { + t.Fatalf("the aged reading should have fanned out once more, got %d total", hits) + } +} + +// --- waking ----------------------------------------------------------------- + +func TestColdRequestWakesANode(t *testing.T) { + t.Setenv("BOX_ENGINE_KEY", "engine-key") + node := newFakeNode(t, string(daemon.StateIdle), "") + node.engineDelay = 100 * time.Millisecond + cfg := fleetOf(t, []string{"box"}, node) + cfg.Nodes[0].EngineTokenEnv = "BOX_ENGINE_KEY" + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted", ServedModelName: "org/wanted"}}, + nil), + }) + + start := time.Now() + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + if time.Since(start) < 100*time.Millisecond { + t.Error("answered before the engine was up") + } + node.mu.Lock() + defer node.mu.Unlock() + if node.started != 1 { + t.Fatalf("node started %d times, want once", node.started) + } + // Started with its own source's config, gated with the fleet's key. + if node.pushed == nil || node.pushed.ModelID != "org/wanted" { + t.Errorf("the node's own config did not reach it: %+v", node.pushed) + } + if node.pushedKey != "engine-key" { + t.Errorf("the engine was gated with %q, want the fleet's key", node.pushedKey) + } +} + +func TestConcurrentColdRequestsShareOneWake(t *testing.T) { + node := newFakeNode(t, string(daemon.StateIdle), "") + node.engineDelay = 150 * time.Millisecond + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted"}}, + nil), + }) + + var wg sync.WaitGroup + codes := make([]int, 2) + for i := range codes { + wg.Add(1) + go func(i int) { + defer wg.Done() + resp, _ := post(t, h, "", `{"model":"org/wanted"}`) + codes[i] = resp.StatusCode + }(i) + } + wg.Wait() + for i, code := range codes { + if code != http.StatusOK { + t.Errorf("request %d: HTTP %d, want 200 from the shared wake", i, code) + } + } + node.mu.Lock() + defer node.mu.Unlock() + if node.started != 1 { + t.Fatalf("the node was started %d times, want once", node.started) + } +} + +func TestWakeTimeoutFailsTheRequestAndLeavesTheEngine(t *testing.T) { + old := fleet.WakeTimeout + fleet.WakeTimeout = 300 * time.Millisecond + t.Cleanup(func() { fleet.WakeTimeout = old }) + + node := newFakeNode(t, string(daemon.StateIdle), "") + node.noEngine = true // the engine never answers + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted"}}, + nil), + }) + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("HTTP %d, want 503: %s", resp.StatusCode, body) + } + if !strings.Contains(body, "box") { + t.Errorf("the timeout should name the node: %s", body) + } + node.mu.Lock() + defer node.mu.Unlock() + if node.started != 1 || node.state != string(daemon.StateRunning) { + t.Error("the started engine is left running on timeout, not stopped") + } +} + +func TestWakeOffRefusesNamingTheNodeAndCommand(t *testing.T) { + node := newFakeNode(t, string(daemon.StateIdle), "") + cfg := fleetOf(t, []string{"box"}, node) + cfg.WakePolicy = fleet.WakeOff + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted"}}, + nil), + }) + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("HTTP %d, want 503", resp.StatusCode) + } + for _, want := range []string{"box", "spinloop fleet start box"} { + if !strings.Contains(body, want) { + t.Errorf("refusal should name %q: %s", want, body) + } + } + node.mu.Lock() + defer node.mu.Unlock() + if node.started != 0 { + t.Error("wake off starts nothing") + } +} + +func TestWakeOffWithNoMatchingSource(t *testing.T) { + node := newFakeNode(t, string(daemon.StateIdle), "") + cfg := fleetOf(t, []string{"box"}, node) + cfg.WakePolicy = fleet.WakeOff + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/something-else"}}, + nil), + }) + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("HTTP %d, want 503", resp.StatusCode) + } + if !strings.Contains(body, "no node's source describes") { + t.Errorf("the refusal should say nothing describes the model: %s", body) + } +} + +func TestNothingCanServeNamesEveryRefusal(t *testing.T) { + a := newFakeNode(t, string(daemon.StateIdle), "") + b := newFakeNode(t, string(daemon.StateIdle), "") + cfg := fleetOf(t, []string{"a", "b"}, a, b) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"a": {Runner: "llamacpp", ModelID: "org/other"}}, + map[string]string{"b": "node \"b\" names no Spinloop source: no `file` field, no alias, no subdirectory"}, + ), + }) + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("HTTP %d, want 503", resp.StatusCode) + } + for _, want := range []string{"a", "b", "org/other", "Spinloop source"} { + if !strings.Contains(body, want) { + t.Errorf("the failure should name %q: %s", want, body) + } + } + a.mu.Lock() + b.mu.Lock() + defer func() { a.mu.Unlock(); b.mu.Unlock() }() + if a.started != 0 || b.started != 0 { + t.Error("nothing is started when nothing can serve") + } +} + +// --- logging ------------------------------------------------------------------ + +func TestRoutedRequestLeavesOneLogLine(t *testing.T) { + var buf bytes.Buffer + log := slog.New(slog.NewTextHandler(&buf, nil)) + node := newFakeNode(t, string(daemon.StateRunning), "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{Log: log}) + + resp, _ := post(t, h, "", `{"model":"org/wanted","messages":[{"content":"a secret prompt"}]}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d", resp.StatusCode) + } + lines := strings.Split(strings.TrimSpace(buf.String()), "\n") + if len(lines) != 1 { + t.Fatalf("one routed request leaves one log line, got %d: %q", len(lines), buf.String()) + } + for _, want := range []string{"org/wanted", "box"} { + if !strings.Contains(lines[0], want) { + t.Errorf("the log line should name %q: %s", want, lines[0]) + } + } + if strings.Contains(buf.String(), "secret prompt") { + t.Error("the log carries the request body") + } +} + +// --- a node that is running but has not answered yet ----------------------- + +// startingNode is a machine whose daemon reports running with its engine's +// port closed and readiness not-ready: llama.cpp fetching or loading weights. +// It is the state that made a request fail with a dial error before the +// gateway consulted readiness at all. +func startingNode(t *testing.T, model string) *fakeNode { + t.Helper() + f := newFakeNode(t, string(daemon.StateIdle), model) + f.mu.Lock() + f.state = string(daemon.StateRunning) + f.ready = daemon.ReadyNo + f.mu.Unlock() + return f +} + +// answersIn flips the node to ready and opens its engine port after d. +func (f *fakeNode) answersIn(d time.Duration) { + go func() { + time.Sleep(d) + f.upAsEngine() + f.mu.Lock() + f.ready = daemon.ReadyYes + f.mu.Unlock() + }() +} + +func TestRequestWaitsForAStartingEngine(t *testing.T) { + node := startingNode(t, "org/wanted") + node.answersIn(150 * time.Millisecond) + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted"}}, + nil), + }) + + start := time.Now() + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + if time.Since(start) < 150*time.Millisecond { + t.Error("answered before the engine was up") + } + node.mu.Lock() + defer node.mu.Unlock() + if node.started != 0 { + t.Errorf("the node was started %d times; it was already running", node.started) + } +} + +// The node loading the model is the one to wait for, not a reason to start a +// second engine somewhere else — which on a fleet of cloud nodes costs money +// for a model that is already minutes from being served. +func TestAStartingEngineIsWaitedForRatherThanWakingAnother(t *testing.T) { + starting := startingNode(t, "org/wanted") + starting.answersIn(100 * time.Millisecond) + spare := newFakeNode(t, string(daemon.StateIdle), "") + cfg := fleetOf(t, []string{"loading", "spare"}, starting, spare) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, map[string]remote.DeployConfig{ + "loading": {Runner: "llamacpp", ModelID: "org/wanted"}, + "spare": {Runner: "llamacpp", ModelID: "org/wanted"}, + }, nil), + }) + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusOK { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + spare.mu.Lock() + defer spare.mu.Unlock() + if spare.started != 0 { + t.Errorf("the spare node was started %d times, want none", spare.started) + } +} + +// An engine that never answers fails the request with a message naming the +// node, not the dial error the caller used to get on every attempt. +func TestStartingEngineThatNeverAnswersFailsNamingTheNode(t *testing.T) { + old := fleet.WakeTimeout + fleet.WakeTimeout = 200 * time.Millisecond + t.Cleanup(func() { fleet.WakeTimeout = old }) + + node := startingNode(t, "org/wanted") + cfg := fleetOf(t, []string{"box"}, node) + h := New(cfg, "", Options{ + ConfigFor: cfgForOf(t, + map[string]remote.DeployConfig{"box": {Runner: "llamacpp", ModelID: "org/wanted"}}, + nil), + }) + + resp, body := post(t, h, "", `{"model":"org/wanted"}`) + if resp.StatusCode != http.StatusServiceUnavailable { + t.Fatalf("HTTP %d, body %s", resp.StatusCode, body) + } + for _, want := range []string{"box", "still be loading"} { + if !strings.Contains(body, want) { + t.Errorf("body should mention %q, got: %s", want, body) + } + } + if strings.Contains(body, "connect: connection refused") { + t.Errorf("the caller should not be given a dial error: %s", body) + } +} diff --git a/internal/remote/remote.go b/internal/remote/remote.go index fc8144ac..2355926f 100644 --- a/internal/remote/remote.go +++ b/internal/remote/remote.go @@ -204,21 +204,28 @@ type Response struct { Environment string `json:"environment"` Message string `json:"message"` RetryAfterSeconds int `json:"retry_after_seconds"` - // Status-specific fields: the on-instance daemon's activity record, - // relayed by the status branch of the start Lambda. camelCase to match the - // daemon's own names, since these are copied through untouched — this - // struct is already mixed (see modelId, contextSize below). Absent when - // the instance is not running, when its daemon could not be reached, or - // when no engine has yet done any work. + // The on-instance daemon's activity record, relayed by the status branch of + // the start Lambda: when the engine last did work, and how long ago. + // camelCase to match the daemon's own names — this struct is already mixed + // (see modelId, contextSize below). Absent when the instance is not + // running, when its daemon could not be reached, or when no engine has yet + // done any work. LastActiveAt string `json:"lastActiveAt"` IdleSeconds int `json:"idleSeconds"` - // Deploy-specific fields. + // Deploy-specific fields. Runner, ModelID and ServedName are also relayed + // by the status reply, which reads them from the environment's deploy + // config — the same source the stats reply reads. Deployed bool `json:"deployed"` Seeding bool `json:"seeding"` // SeedID identifies the seed a deploy started, so it can be followed with // `spinloop remote seed status`. The instance id it replaces was an // implementation detail that changes if the seed is relaunched. - SeedID string `json:"seedId"` + SeedID string `json:"seedId"` + // ServedName is the name the engine answers to beside the model id — the + // served name the deploy gave it — relayed by the status reply from the + // environment's deploy config, so a caller may know the engine by either + // name. + ServedName string `json:"servedName"` Runner string `json:"runner"` ModelID string `json:"modelId"` ContextSize int `json:"contextSize"` diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/.openspec.yaml b/openspec/changes/archive/2026-09-12-add-fleet-gateway/.openspec.yaml new file mode 100644 index 00000000..1a62d62b --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-06 diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/design.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/design.md new file mode 100644 index 00000000..a15f76e5 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/design.md @@ -0,0 +1,248 @@ +## Context + +See proposal.md — Why. What matters for the approach is what already exists: + +- `internal/fleet` has the whole routing half: `Select` (fan-out plus the pure + `rank`), `Wake` (start-with-config, the warm-first candidate ordering, the + already-running race, `waitReady`), `EngineBaseURL` (composing the engine + address from the fleet file and the node's report), and `engineKeyFor` + (resolving a running engine's key). A gateway is a consumer of all of it. +- The wake path's one gap: when a start loses to another client's + already-running 409, `Wake` re-reads status and takes the node on the state + alone — a node can report `running` while still loading weights. Fine for a + client-side launch (the agent starts and the engine comes up moments later); + a gateway holding a caller's request must not proxy to an engine that is not + answering. +- The daemon's status says where the engine serves (`EngineEndpoint`) and, on + current daemons, whether it has answered its own health check (`Ready`), but + it reports the model id only, not the served name: a node started under an + alias answers requests for the alias and reports the id, so a router seeing + only status cannot match a request that carries the alias. +- The launch path consumes a `fleet.Choice`: `applyBeforeLaunch` writes + `choice.BaseURL` into the provider slot a `REMOTE` fills and wires + `choice.APIKey` through the same resolver the remote path uses. The endpoint + branch of `FLEET` is a new producer of the same `Choice`. +- The seam itself: `Selection.FleetIsEndpoint`, and the two + "gateway routing is not implemented yet" rejections in `routeThroughFleet` + and `fleet route`. + +## Goals / Non-Goals + +**Goals:** + +- One gateway that is the fleet client wearing a server: selection, waking, + endpoint resolution, and key handling all come from `internal/fleet`, so the + routing rules stay written once. +- A request held during a wake behaves like a slow first token, not an error: + the same readiness wait the launch path uses. +- The launch path's endpoint branch is a small new producer of `fleet.Choice`, + not a second routing mechanism. +- Every secret keeps its existing home: node tokens and engine keys in the + environment, named by the fleet file; the gateway's own token supplied the + way the daemon's is. + +**Non-Goals:** + +- TLS termination, budgets, spend tracking, per-user keys, retries across + nodes, request-body logging, queueing. The gateway stays a binary to run, + not a service to operate. +- New daemon endpoints, and new Spinloop keywords: `servedName` is one + additive status field, and `FLEET` already takes a URL. +- A gateway config file of its own: fleet.yaml plus flags is the configuration. + +## Decisions + +### A `gateway` command beside `serve`, a new `internal/gateway` package + +`spinloop gateway --fleet ./fleet.yaml --listen :4000` runs in the foreground +like `serve`: it holds a fleet file the way `serve` holds a Spinloop, and its +lifecycle is the process's. Top-level rather than `fleet gateway`, because it +does not observe the fleet the way `fleet status` does; it answers requests +through it, which is a different job. + +The HTTP surface lives in `internal/gateway` as an `http.Handler` built from a +`*fleet.Config`, the caller token, and a wake timeout. `cmd/spinloop/gateway.go` +resolves the fleet file, resolves the token, and serves. This keeps the +proxies, the model listing, the wake joining, and the status cache unit-testable +with a fake fleet, the same way the fleet client is tested. + +### The gateway wakes a node with the node's own config, never an invented one + +A completion request carries one name — the model — and a deploy config needs a +runner, a context, and serve args the request cannot name. The only config a +gateway can push is the one the node was told to run: its Spinloop source, +resolved exactly as `spinloop fleet start` resolves it (the `file` field, a +registered alias named after the node, a same-named directory beside the fleet +file). A node is therefore a wake candidate when it is not running and its +source's config names — as model or served name — the model the request asks +for. This is also the honest semantics: the gateway starts what the fleet file +says each node runs, and says so in the failures it reports. + +Consequently `fleet.Wake`'s single-config-per-all-candidates shape does not fit +the gateway, which needs one config per candidate. `Wake` is generalised to +take a per-candidate config function; the launch path passes a constant one +(derived from the Spinloop it wears), the gateway passes a resolver over the +node's own source. The ordering (warm-first), the refusal collection, the +already-running race, and the readiness wait all stay in `Wake`, written once. + +### Readiness is a gate before any request is proxied + +"Usable now" is three facts: the node reports `running`, the name it reports +serving matches the request, and its engine answers. The third is the daemon's +`Ready` field, and a node reporting not-ready is not a match at all, so +selection skips it — the state turns running when the engine's process exists, +which is before the weights are fetched and loaded, and during that window +nothing is listening on the engine's port. An absent reading is not evidence of +anything: older daemons and runners with no health-check convention report +none, and they still route. + +The gateway's per-request shape is therefore: select (from the cached or fresh +fan-out) → when nothing is serving, wait for a node already loading the model, +or wake one if the policy allows → `WaitReady` on the chosen node → proxy. The +wait for a starting node is deliberate: it is loading the weights the request +needs, so it serves sooner than anything a wake would start from cold, and +waking a second node would leave two engines up for one request. It is bounded +by the same wake timeout as a cold start, names the node it holds the request +for, a timeout still leaves the engine running, and it is not a wake — a fleet +whose wake is off still holds a request for a node already loading. The wait +and the wake share one readiness wait: `WaitReady` is lifted out of `Wake`'s +private helper into an exported form that both call. + +The same fix closes the client-side gap in the wake race: `Wake`'s +already-running branch currently takes a raced node on the state alone; it now +goes through the readiness wait, so no caller — launch or gateway — is handed +an engine that is still loading. The wait is bounded by the same wake timeout, +and a timeout still leaves the engine running. + +### Concurrent cold requests share one wake + +With the readiness fix, correctness already holds under concurrent cold +requests: the first starts the engine, the rest lose the 409, re-read, and +wait. But N requests would make N start attempts and N readiness polls against +one wake. The gateway keeps a single in-flight wake per (node, model) — a +mutex-guarded map — and joins it: the first request's wait is the wait the rest +take. At most one start per node, one poll loop, and every request in the +burst is answered from the same engine. + +### A short cache over the status fan-out + +Selecting fans out over every node, and the round waits on each producer with a +five-second per-node bound: one dead node would add up to five seconds to +*every* request an agent makes. The gateway therefore reuses a fan-out taken +within the last two seconds and re-runs the pure selection per request. The +ranking is on `idleSeconds`, which the fleet-harness-routing design already +calls a crude signal; a two-second-stale reading is unobservable, and the cache +also bounds how often wake decisions are made. The freshness is a package +variable so tests do not wait. A request that loses a race with a stopping node +fails with a connection error naming the node; the next request re-fans-out and +chooses elsewhere, which is the behaviour the no-retry rule gives. + +### The proxy: per-request, streaming, key-swapped + +Each routed request gets its own single-host reverse proxy at the resolved +engine address, with `FlushInterval` set so a streamed reply is flushed as the +engine sends it rather than in chunks. The proxy's client has no overall +timeout — a long generation is a long response — while the fan-out keeps the +fleet client's short one. The caller's `Authorization` is removed; the engine +key the gateway holds is set as the upstream's when the node reports its engine +gated, and nothing is set when it is not. The body and the reply pass through +unmodified, and a refused or failed upstream reply is the caller's reply: the +gateway does not retry at another node, because the engine's own error is the +honest one. + +### Auth and exposure mirror the daemon's rules + +One bearer token, the daemon's three sources (file, `SPINLOOP_API_TOKEN`, +command line), more than one a conflict, `401` without the right one, a +non-loopback listen without a token refused at startup, loopback without a +token allowed. The gateway is a longer-lived front door than the daemon's +control API and carries conversation content rather than control traffic, but +it sits on the fleet's own network, whose trust model is already +plain-HTTP-plus-bearer; TLS is an additive listener flag for later, not a +v1 gate. The rule is implemented in the gateway rather than imported from +`internal/daemon`, which keeps the daemon package a leaf the gateway depends +on rather than the other way round. + +### Status reports the served name beside the model id + +The daemon already stores the deploy config and reports its model id as the +served model; the served name is the same stored config's own field, so status +reports it when the config names one and omits it otherwise — one additive +`omitempty` field, no endpoint change. Older daemons and remotes omit it, and a +gateway matching a request then falls back to the model id alone, which is +exactly today's client-side behaviour. `docs/openapi.yaml` is updated to match; +`openapi_test.go` keeps the two in step. + +### The launch path's endpoint branch produces a `Choice` + +`routeThroughFleet` gains the branch its rejection used to occupy: a `FLEET` +with a scheme yields a `Choice` whose `BaseURL` is the named endpoint (with the +OpenAI-compatible prefix appended when the value carries no path) and a marker +that it names an endpoint rather than a node. The token is not resolved there: +`applyBeforeLaunch` already builds the resolver chain (environment, `.env` +beside the Spinloop, `ENV` instructions) and already special-cases a fleet +choice's key, so the endpoint's token is resolved through that same chain after +routing, a missing value fails before anything is written naming the variable, +and an already-set variable wins exactly as on the remote path. `fleet route` +gets the same branch as a report: the endpoint has already chosen, no node is +queried, nothing is started. + +### `wake:` sits in the fleet file beside `prefer:` + +Whether work may be started on the fleet's machines is a property of the fleet, +on the reasoning the `prefer` decision recorded: the same fleet shared by +several people and owned by one person wants different answers, and the fleet +file is the thing that differs. `wake: on|off`, default on (waking is today's +behaviour, so an existing fleet is unchanged), invalid values refused at parse +time. It decides *whether* to wake only: which node is chosen, the ranking, and +what a wake does are all untouched. The launch path's `--no-wake` still wins +over a file that allows waking — an explicit flag beats a file setting — and a +refused wake, by flag or by file, names the node that would have been woken and +the command that would start it. + +## Risks / Trade-offs + +- **A held request can outlive the caller's patience.** A cold wake is minutes + and the caller sees a slow response, with no progress to speak to. → It is + bounded by the wake timeout, the failure says the engine was left running, + and a retry is answered from the engine that is still loading. The + alternative — refuse and make the client retry — puts the burden on clients + that mostly will not retry. +- **A fleet whose nodes name no matching source cannot be woken through the + gateway**, even if a daemon could run the model from some other config. → + Deliberate: the gateway starts what the fleet file says a node runs, and the + failure names the node and the three ways a source could have been given. + The client-side wake, which carries its own Spinloop, is unaffected. +- **The two-second cache can pick a node that has just stopped.** → The proxy + fails with a connection error naming the node; the next request re-fans-out + and chooses elsewhere. No retry, by design. +- **Remotes in the fleet make cold wakes slow** — a scale-from-zero can exceed + the five-minute default. → Remotes wake through the same uniform `Node` + interface, and the timeout is a flag: a fleet with remotes sets a longer one + or declares `wake: off`. A wake that times out on a remote leaves the + instance starting, as the client-side path already does. +- **`servedName` is absent on older daemons and on remotes**, so aliases do not + match against them. → Matching falls back to the model id, which is today's + behaviour; the `no node is serving` failure lists each node's reported model, + which names the mismatch to anyone who knows the fleet. +- **The gateway holds every node's engine key in its environment.** A machine + compromise exposes all of them at once. → That is the trade the gateway is + bought for — the keys stop being distributed to every agent machine — and the + gateway is the one process the operator runs where the secrets already live. + +## Migration Plan + +Additive throughout. A fleet file with no `wake:` and a Spinloop with no +endpoint `FLEET` behave exactly as they do now; an older daemon that omits +`servedName` is matched on the model id; the wake race's readiness wait only +ever makes a launch wait longer, never shorter, and only in the race that +previously handed out a loading engine. Rollback is removing the `FLEET` URL +and stopping the gateway process; nothing else changes behaviour unless +something asks it to. + +## Open Questions + +None that block: TLS, budgets, per-user keys, and request logging beyond a line +per route are excluded by the proposal rather than deferred, and the wake +timeout default (five minutes, shared with the launch path) is a flag, not a +question. diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/proposal.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/proposal.md new file mode 100644 index 00000000..ae25d6c4 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/proposal.md @@ -0,0 +1,124 @@ +## Why + +A machine running an agent against a fleet today needs the fleet file, every +node's bearer token, and every node's engine key. A single +OpenAI-compatible endpoint in front of the fleet removes all of it: the agent +needs only a URL and one token, and the secrets stop at the gateway instead of +being distributed to every machine. The routing half was deliberately not built +when the fleet gained client-side selection (`fleet-harness-routing`, archived +2026-08-12) because it had nothing to stand on; the selector, the engine +endpoint reporting, and the wake path it assembles are all in place now, so the +gateway is mostly assembly. + +## What Changes + +- New `spinloop gateway` command: a foreground process that holds a fleet file + and serves one OpenAI-compatible endpoint for it — `/v1/models` (what a + request can reach: what the nodes are running, and, when the fleet's wake + allows it, what a stopped node's own source describes — the model a request + would start it with) and reverse proxies for `/v1/chat/completions` and + `/v1/completions`, picking a node with the existing fleet selector and + streaming the reply through untouched. +- The gateway authenticates callers with one bearer token, the way the daemon + does (file, environment, or command line; a non-loopback listen without one + refuses to start), and holds each node's engine key itself, supplying it to a + node when it wakes one — the client-side rule that the starter supplies the + key now applies to the gateway as the starter. +- The gateway wakes a node when a request names a model nothing is serving, + holding the request while the engine loads, bounded by a wake timeout. A + fleet file's new top-level `wake:` setting (`on`/`off`, default `on`) + controls this; a model no node can serve fails fast, naming each node's + refusal. +- Pointing a launch at the gateway is the fleet file's `gateway` section (added + by `add-fleet-harness`): a launch whose effective fleet file names a gateway + is pointed at it, with the gateway's token taken from the client's + environment, and `spinloop fleet route` answers such a Spinloop by saying the + gateway has already chosen. +- Daemon status reports the name an engine serves its model under (the deploy + config's served name) beside the model id, so a request that names an alias + can be matched against what a node reports. +- A remote environment's status now carries what it is serving — the model id, + and the served name beside it — read from the environment's stored deploy + config, the same source the stats reply uses, so a gateway (or `fleet + status`) can match a request to a running remote node the way it matches a + local one and the fleet and remote views name it the same. Without this a + running remote node reported its state but no model, and was invisible to + model-based routing. +- A remote environment's status also carries where its engine answers — the + instance's published address the control plane reports, which a daemon on the + instance cannot know. Routing resolves it as the engine's host, so the gateway + reaches a running remote node instead of listing its model and then refusing + to route a request to it. +- Waking: a start refused because another client woke the node first is + re-read, and the winner of that race is used only once its engine answers — + not on state alone — so no caller, client or gateway, is handed an engine + that is still loading weights. +- Selection no longer treats an engine that has not answered as serving: the + daemon's state turns running when the engine process exists, which is before + the weights are fetched and loaded, and during that window nothing is + listening on the engine's port. Such a node is not selected, and a request for + its model is held for the node to finish starting — bounded by the wake + timeout, the engine left running on it — rather than waking a second node. An + absent readiness reading still routes, so older daemons are unaffected. +- New standalone example `examples/gateway-docker/`: a fleet plus a gateway in + containers, with a client that holds nothing but a URL and one token. + +## Capabilities + +### New Capabilities + +- `fleet-gateway`: the `spinloop gateway` command — its OpenAI-compatible + surface, caller authentication, node selection and proxying, wake behaviour + and its timeout, and what it deliberately is not. + +### Modified Capabilities + +- `fleet-routing`: a launch whose effective fleet file names a gateway is + routed to that gateway (base URL and key resolution); `fleet route` answers + such a Spinloop; waking's race rule now requires the engine to answer before a + raced node is used; choosing no longer treats a node whose engine has not + answered as running what is wanted, and a pinned node still starting fails + saying so. +- `fleet-config`: a fleet file MAY declare a top-level `wake` setting + (`on`/`off`) deciding whether routing starts an engine on a node that is not + running one. +- `daemon-api`: status reports the served name of the model an engine runs, + beside the model id it already reports. +- `remote-node`: a running environment's status carries what it is serving + (model id and served name), read from the same stored deploy config the + stats reply reads, so the fleet and remote views name it the same. + +## Impact + +- `cmd/spinloop`: new `gateway.go` command; the launch path's route step and + `fleet route` gain the endpoint branch; completion and help cover the new + command. +- `internal/gateway` (new): the HTTP surface — auth, model listing (what runs, + plus what a wake can start, each node's source resolved at most once in a + short window), the reverse proxy, the in-flight wake joining, and a short + cache over the status fan-out so a burst of requests does not fan out per + request. +- `internal/fleet`: the wake race winner waits for the engine to answer; the + "usable now" test (running, matching, ready) is shared with the gateway; + selection skips a node whose daemon reports the engine not ready, and a + caller holding the no-node-serving failure can tell that from a node already + loading the wanted model and wait for it; the fleet file parses `wake:`. +- `internal/daemon`: status gains the served-name field, set from the stored + deploy config, and `EngineEndpoint` gains the host a node reports when it + knows its client-facing address (a remote environment does; a daemon never + will); `docs/openapi.yaml` updated to match (checked by `openapi_test.go`). +- `internal/fleet`: the remote node's status mapping carries the serving facts + the control plane now relays and the engine's published address as the + endpoint's host, and routing resolves a reported engine host in place of the + fleet file's; `internal/remote`: `Response` gains the served name. +- `remote/` (control plane): the start Lambda's status branch reads the + environment's stored deploy config for the serving facts (runner, model id, + served name) and the daemon for its activity. This changes what a deployed + start Lambda answers, so a control-plane redeploy is needed for live + environments to report a model; an environment deployed before the served-name + feature reports its model id but no served name until it is redeployed. +- `docs/`: command reference for `spinloop gateway`; + `examples/gateway-docker/` is the runnable end-to-end demonstration. +- No daemon endpoint changes, no new dependencies, no Spinloop keyword, no + change to the launch path's ordering (route before apply). A fleet with no + `wake:` and a Spinloop naming no gateway behave exactly as they do now. diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/daemon-api/spec.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/daemon-api/spec.md new file mode 100644 index 00000000..b9f66e2a --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/daemon-api/spec.md @@ -0,0 +1,24 @@ +## ADDED Requirements + +### Requirement: Status reports the name the model is served under + +Status SHALL report, beside the model id it already reports, the name the +engine serves the model under — the deploy config's served name — when the +stored config names one. A router that sees only status needs the name a +request will carry: a client that started the engine under an alias addresses +it by that alias, and matching a request only against the model id would not +recognise the node. The served name SHALL be omitted when the stored config +names none, and the model id SHALL be reported on the same terms it is today +whatever the served name is. + +#### Scenario: An aliased engine reports both names + +- **WHEN** an engine was started from a config that names both a model and a + served name, and a status request is made +- **THEN** the response reports the model id and the served name + +#### Scenario: No served name, no field + +- **WHEN** an engine was started from a config that names no served name, and a + status request is made +- **THEN** the response reports the model id and carries no served name diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-config/spec.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-config/spec.md new file mode 100644 index 00000000..b42e6cf7 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-config/spec.md @@ -0,0 +1,41 @@ +## ADDED Requirements + +### Requirement: Fleet-wide wake policy + +A fleet file MAY declare a top-level `wake` value of `on` or `off`, deciding +whether routing starts an engine on a node that is not running one when no +running node serves what is wanted. It belongs to the file rather than to each +node, for the reason `prefer` does: it describes how this cluster is to be +used — may work be started on its machines on demand, or only used where it is +already running — which is a property of the fleet, not of any one machine in +it. + +A file declaring nothing SHALL wake, as routing does when the setting is +absent: waking is the difference between a fleet that answers a request and +one that must be prepared by hand, and the file's author is the one who owns +the machines it names. A file declaring anything other than `on` or `off` SHALL +fail to parse, naming both accepted values, in keeping with the file's other +validation. + +The setting SHALL decide whether to wake only. It SHALL NOT change which node +is chosen, how matching nodes are ranked, or what a wake does: a fleet that +declares `wake: off` still reports, when nothing is running, the node whose +source describes the wanted model and the command that would start it. + +#### Scenario: A fleet that declares nothing wakes + +- **WHEN** a fleet file declares no `wake` setting and routing finds no node + serving what is wanted +- **THEN** routing starts an engine on a suitable node, as it does today + +#### Scenario: A fleet that refuses to wake + +- **WHEN** a fleet file declares `wake: off` and routing finds no node serving + what is wanted +- **THEN** nothing is started, and the failure names the node that would be + woken and the command that would start it + +#### Scenario: An unknown value is rejected at parse time + +- **WHEN** a fleet file declares `wake: sometimes` +- **THEN** parsing fails naming `on` and `off` diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-gateway/spec.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-gateway/spec.md new file mode 100644 index 00000000..8a58d3fb --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-gateway/spec.md @@ -0,0 +1,389 @@ +## Purpose + +One OpenAI-compatible endpoint in front of a fleet: a foreground process that +answers agent requests by choosing a node with the fleet's own selector, holds +each node's engine key, and wakes a node when nothing is serving what a request +asks for — so a machine running an agent needs nothing but a URL and one token. + +## ADDED Requirements + +### Requirement: The gateway command + +`spinloop gateway` SHALL run in the foreground, the way `spinloop serve` does, +holding the fleet file it serves: `--fleet ` (with the `-f ` short +form) when given, otherwise `./fleet.yaml` in the working directory, and a +missing file SHALL fail naming the expected path, as the fleet commands do. +The gateway SHALL listen on an address given by `--listen`, defaulting to +port 4000 on all interfaces, and SHALL print the address to name in a fleet +file's `gateway` section when it starts. + +The gateway's own startup failures SHALL be the fleet file's: a fleet file that +does not parse, or that names a token variable set nowhere, SHALL fail the +gateway at startup naming the problem, rather than listening and failing per +request. + +#### Scenario: The gateway starts and answers + +- **WHEN** the user runs `spinloop gateway --listen :4000` in a directory + holding a `fleet.yaml` +- **THEN** it prints the address to name in the fleet file's `gateway` section + and serves requests until it is stopped + +#### Scenario: An explicit fleet file is served + +- **WHEN** the gateway is given a `--fleet` path +- **THEN** that file is the one it serves + +#### Scenario: A missing fleet file names itself + +- **WHEN** the user runs `spinloop gateway` in a directory holding no + `fleet.yaml` and passes no `--fleet` +- **THEN** it fails naming `./fleet.yaml`, and nothing listens + +#### Scenario: A broken fleet file fails at startup + +- **WHEN** the gateway's fleet file names a token variable that is set nowhere +- **THEN** the gateway fails at startup naming the node and the variable, + rather than listening + +### Requirement: Caller authentication + +The gateway SHALL authenticate callers with one bearer token, on the same +terms the daemon's control API authenticates: the token MAY be supplied by a +file (`--api-token-file`), the environment (`SPINLOOP_API_TOKEN`), or the +command line (`--api-token`); giving more than one SHALL fail naming the +conflict. Requests without the correct token SHALL be rejected with `401`. +When no token is configured, the gateway SHALL refuse to listen on a +non-loopback address and SHALL say why, and listening on loopback without a +token SHALL be allowed. + +The caller's token authorises use of the gateway only. It SHALL NOT be +forwarded to a node's engine or control API: the gateway reaches a node with +the node's own credentials, resolved from the fleet file the way every other +fleet client does. + +#### Scenario: A wrong token is rejected + +- **WHEN** a request carries a missing or incorrect bearer token +- **THEN** the response is `401` and no node is contacted + +#### Scenario: A tokenless non-loopback listen refuses to start + +- **WHEN** the gateway would listen on a non-loopback address and no token is + configured +- **THEN** startup fails saying a token is required for non-loopback exposure, + naming every way one can be supplied + +#### Scenario: A tokenless loopback is permitted + +- **WHEN** the gateway listens on a loopback address with no token configured +- **THEN** it serves requests without authentication + +#### Scenario: The caller's token stops at the gateway + +- **WHEN** an authenticated request is routed to a node +- **THEN** the node is reached with the node's own credentials from the fleet + file, and the caller's token is not sent to it + +### Requirement: Listing the fleet's models + +The gateway SHALL serve `GET /v1/models` returning, in the OpenAI list shape, +the union of the models a request can reach. For each node whose state is +`running`, the list SHALL carry the name it reports serving — the served name +when it reports one, otherwise the model id. When the fleet's wake allows +starting an engine, the list SHALL additionally carry, for each node that is +not running and answers its status, the model that node's own Spinloop source +describes — the same served-name-first naming the wake would start it with — +since a request naming that model starts that node. A running node SHALL +contribute nothing but what it reports: a running engine is never displaced, +so its source's model is not a request the gateway would answer from it. A +node the gateway cannot start — a remote environment, which starts from +`spinloop remote deploy`, not from a request — and a fleet whose wake is off +SHALL contribute nothing beyond what is running, and duplicates SHALL be +listed once. The source a node describes SHALL be resolved at most once in a +short window shared by all models requests, so a burst does not re-read every +node's source. + +#### Scenario: Running models are listed + +- **WHEN** two nodes are running, one serving a model under an alias and one + under its id, and a models request is made +- **THEN** the response lists the alias and the id, each once + +#### Scenario: A stopped node's wakeable model is listed + +- **WHEN** a node is stopped, its own source describes a model, and the fleet + allows waking, and a models request is made +- **THEN** the response lists the model the source describes, beside what the + running nodes serve + +#### Scenario: A stopped node's model is not listed when wake is off + +- **WHEN** a node is stopped and the fleet's wake is off, and a models request + is made +- **THEN** the response lists only what the running nodes serve + +#### Scenario: A stopped remote environment's model is not listed + +- **WHEN** a remote environment is stopped and a models request is made +- **THEN** the response does not list what its source describes: the gateway + cannot start it, so a request naming that model would fail + +#### Scenario: A running node's source adds no second model + +- **WHEN** a running node reports one model and its source describes another, + and a models request is made +- **THEN** the response lists only what the node reports + +#### Scenario: A burst of models requests resolves each source once + +- **WHEN** several models requests arrive within the window in which a node's + source is resolved +- **THEN** each node's source is read once for the burst + +#### Scenario: Nothing reachable lists nothing + +- **WHEN** no node is running and nothing is wakeable — no stopped node's + source describes a model, or the fleet's wake is off — and a models request + is made +- **THEN** the response is an empty list, not an error + +### Requirement: Routing a request to a node + +The gateway SHALL serve `POST /v1/chat/completions` and `POST /v1/completions` +by choosing a node and reverse-proxying the request to that node's engine. The +choice SHALL be the fleet's own selection: every node's state is considered, a +node matches when what it reports serving — its model id or its served name — +equals the model the request names, and matching nodes are ranked by the fleet +file's activity preference, ties broken by fleet-file order. A request naming +no model SHALL be refused saying so, not routed at a guess. + +A node whose daemon reports the engine not ready SHALL NOT be selected: the +state turns running when the engine's process exists, which is before the +weights are fetched and loaded, and during that window nothing is listening on +the engine's port. A node whose daemon reports no readiness at all SHALL NOT be +disqualified: the absence of a reading is not evidence of not-readiness. + +When nothing is serving the model the request names, a node already running it +whose engine has not answered yet SHALL be waited for rather than passed over: +it is loading the weights the request needs, so it serves sooner than anything +a wake would start from cold, and waking a second node would leave two engines +up for one request. The wait SHALL be bounded by the wake timeout, SHALL name +the node it holds the request for, and a node that does not answer in time +SHALL fail the request naming it, leaving its engine running. Waiting is not +waking: a fleet whose wake is off SHALL still hold a request for a node already +loading the model it named, since no engine is being started. + +The request's body SHALL reach the chosen engine unmodified, and the engine's +reply SHALL reach the caller unmodified, including a streamed reply, which the +gateway SHALL pass through without holding it back. The gateway SHALL replace +the caller's authorisation with the engine key it holds for that node — the +value the node's fleet entry names, resolved the way every other fleet client +resolves it — and SHALL send no authorisation at all to an engine that needs +none. The gateway SHALL NOT retry a failed request at another node: the reply +the chosen engine gives is the reply the caller gets. + +The engine's address SHALL be resolved the way routing resolves it: the node's +declared engine override as given, otherwise the node's host with the port and +path the engine reports. Where a node reports its engine's host — a remote +environment, whose control plane publishes the instance's address and which the +fleet file names by environment alone — that reported host SHALL be used in +place of the node's host, so the request reaches the instance rather than an +address the fleet file never held. A node that reports its engine bound to +loopback, without an override taking responsibility for reachability, SHALL NOT +be selected: it answers only on its own machine, and where no other node serves +the wanted model the request SHALL fail saying so and naming the fix. + +The gateway SHALL reuse node state it has recently read — a reading taken +within the last couple of seconds — rather than query every node on every +request, so a burst of requests does not pay a fan-out each. The freshness of +the reading SHALL NOT matter to the choice: the same fleet in the same state +chooses the same node. + +Each routed request SHALL be logged as one line — the model, the node chosen, +and the outcome — and the gateway SHALL log no request body. + +#### Scenario: A request goes to the node serving its model + +- **WHEN** one node is running the model a request names and another is + running a different model +- **THEN** the request is proxied to the first node and the engine's reply + reaches the caller unmodified + +#### Scenario: The engine key is swapped, not the caller's token + +- **WHEN** a request is routed to a node whose engine is gated, and the node's + fleet entry names the key +- **THEN** the engine receives the key's value as its authorisation, and the + caller's token is not sent to the engine + +#### Scenario: A streamed reply passes through + +- **WHEN** a streamed completion is requested and the chosen engine streams its + reply +- **THEN** the caller receives the stream as the engine sent it + +#### Scenario: An engine error is the caller's error + +- **WHEN** the chosen engine refuses the request +- **THEN** the engine's refusal reaches the caller and no other node is tried + +#### Scenario: A request naming no model is refused + +- **WHEN** a completion request carries no model +- **THEN** it is refused saying the request names no model, and no node is + contacted + +#### Scenario: A loopback-bound engine is not selected + +- **WHEN** the only node serving the wanted model reports its engine bound to + loopback and names no engine override, and the node is not reached over + loopback +- **THEN** the request fails saying the engine answers only on that machine, + naming the bind and the override as the fixes + +#### Scenario: A not-ready engine is not selected + +- **WHEN** the only node running the model a request names reports its engine + not ready, and the fleet's wake is off +- **THEN** the request fails naming that node marked not ready, and nothing is + started + +#### Scenario: A request waits for an engine that is still starting + +- **WHEN** no node is serving the model a request names, one node is running + it but its engine has not answered yet, and no other node is running it +- **THEN** the request is held for that node and answered when its engine + answers, and no other node is started + +#### Scenario: A still-starting engine that does not answer fails the request + +- **WHEN** a node is running the model a request names, its engine has not + answered yet, and it does not answer within the wake timeout +- **THEN** the request fails saying so, naming the node, and the engine is left + running + +#### Scenario: Wake off still waits for an engine already starting + +- **WHEN** the fleet file declares the wake policy off and the only node + running the model a request names has not answered its engine yet +- **THEN** the request is held for that node and answered when the engine + answers, since waiting starts nothing + +#### Scenario: A burst of requests does not fan out per request + +- **WHEN** several requests arrive within a couple of seconds of each other +- **THEN** the fleet is queried once for the burst, and the requests are + answered from that reading + +#### Scenario: A routed request leaves one log line + +- **WHEN** a request is routed and answered +- **THEN** the gateway's log gains one line naming the model, the node, and the + outcome, and carries no part of the request body + +### Requirement: Waking a node for a request + +When no running node serves the model a request names, and the fleet file's +wake policy allows it, the gateway SHALL start an engine on a node that is not +running one, and SHALL hold the request until the engine answers. A node is a +wake candidate when it is not running and the Spinloop source it names — its +`file` field, a registered alias named after it, or a same-named directory +beside the fleet file, resolved the way `spinloop fleet start` resolves it — +describes a config whose model or served name is the one the request asks for: +a node is started with what it was told to run, never with a config invented +for the request. Candidates whose stored config already names the model SHALL +be tried first, since they have the weights, and the rest in fleet-file order. +A node that refuses the start — a runner or model it cannot serve — SHALL NOT +fail the request while other candidates remain. + +The started engine SHALL be gated with the key the node's fleet entry names, +supplied by the gateway: the gateway is the client that starts the engine, so +the key the client sets is the key the engine takes. The wait SHALL be bounded +by a wake timeout, defaulting to five minutes and overridable by +`--wake-timeout`; exceeding it SHALL fail the request saying the engine did +not answer in time, and the started engine SHALL be left running rather than +stopped, so a slow load is not thrown away. + +When several requests ask for a model nothing is serving at once, the gateway +SHALL start at most one engine per node and answer every request from it: the +first request's wait is the wait the rest join. A node another request woke +first SHALL be used the same way, and only once its engine answers. + +With the wake policy off, a request for a model nothing is serving SHALL fail +without starting anything, naming the nodes and what they could serve, and the +command that would start one. A model no node is running and no node's source +describes SHALL fail the same way, whatever the policy: nothing to wake with, +and the failure SHALL say so rather than trying to start a node with nothing. + +#### Scenario: A cold request wakes a node and is served + +- **WHEN** no node is running the model a request names, one node's Spinloop + source describes it, and the wake policy allows it +- **THEN** that node is started with its own config, gated with the key its + fleet entry names, and the request is answered once the engine answers + +#### Scenario: The request is held while the engine loads + +- **WHEN** the woken node reports running while its engine is still loading +- **THEN** the request waits, and is answered when the engine answers, rather + than failing against an endpoint that refuses connections + +#### Scenario: A wake that does not finish in time fails the request + +- **WHEN** a woken node's engine does not answer within the wake timeout +- **THEN** the request fails saying so, naming the node, and the engine is left + running + +#### Scenario: Concurrent cold requests share one wake + +- **WHEN** two requests arrive at once for a model nothing is serving, and one + node's source describes it +- **THEN** that node is started once, and both requests are answered from the + same engine + +#### Scenario: A woken engine takes the gateway's key + +- **WHEN** the gateway starts an engine on a node whose fleet entry names an + engine key +- **THEN** the engine is gated with that value, the gateway's requests to it + carry it, and no reply to any caller contains it + +#### Scenario: Wake refused by the fleet file + +- **WHEN** the fleet file declares the wake policy off and no node is serving + the model a request names +- **THEN** nothing is started, and the request fails naming the node whose + source describes the model and the command that would start it + +#### Scenario: Nothing can serve the model + +- **WHEN** no node is running the model a request names and no node's Spinloop + source describes it +- **THEN** the request fails, naming each node and why it cannot serve the + model, and nothing is started + +#### Scenario: A sourceless node is not woken + +- **WHEN** the only node that could take a request names no Spinloop source + that resolves +- **THEN** it is not started, and the failure names it and the ways a source + could have been given + +### Requirement: Paths the gateway does not serve + +A path other than `/v1/models`, `/v1/chat/completions`, `/v1/completions`, and +the gateway's own health path SHALL be answered with `404` naming the paths the +gateway serves. The health path SHALL answer that the gateway is up without +contacting any node, so an operator can tell the gateway down from the fleet +down. + +#### Scenario: An unknown path is named as such + +- **WHEN** a request is made to a path the gateway does not serve +- **THEN** the response is `404` and names the paths it does serve + +#### Scenario: The health path does not touch the fleet + +- **WHEN** the health path is requested and every node is unreachable +- **THEN** it still answers that the gateway is up diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-routing/spec.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-routing/spec.md new file mode 100644 index 00000000..a348a99b --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/fleet-routing/spec.md @@ -0,0 +1,199 @@ +## MODIFIED Requirements + +### Requirement: Choosing a node + +Selection SHALL query every candidate node concurrently, as `spinloop fleet +status` does, and SHALL prefer a node that is already running what is wanted: a +node whose state is `running` and whose served model matches the Spinloop's +`MODEL` (or its `ALIAS`, against the name the node reports serving). A Spinloop +that names no model SHALL match any running node. + +A node whose state is `running` but whose daemon reports the engine not ready is +not running what is wanted: the state turns running when the engine's process +exists, which is before the weights are fetched and loaded, and during that +window nothing is listening on the engine's port. Such a node SHALL NOT be +selected, and a failure that names it SHALL mark it not ready, so a refusal does +not read as if the node were serving the model it was asked for. A node whose +daemon reports no readiness at all — an older build, or a runner with no +health-check convention — SHALL NOT be disqualified: the absence of a reading is +not evidence of not-readiness. + +Matching nodes SHALL be ranked by the activity preference in force (see +"Preferring an idle or an active node"). Ties SHALL be broken by fleet-file +order, so the same fleet in the same state chooses the same node. + +A node that does not answer — unreachable, unauthorized, or a configuration +error — SHALL be skipped rather than aborting the selection, exactly as it is a +row rather than a failure in `spinloop fleet status`. + +`--node ` SHALL pin the selection to one node, skipping the search. An +unknown name SHALL fail naming the known nodes, and a pinned node that cannot be +reached SHALL fail rather than falling back to another node — a pin is an +instruction, not a preference. A pinned node that is running the wanted model +but has not answered yet SHALL fail saying it is still starting it — it may be +fetching or loading weights — and naming the command for the node's log, rather +than saying that nothing serves the model or restarting the node: this node is +about to serve it. + +A running engine SHALL NEVER be stopped or restarted to make room, including a +pinned one: another person may be using it. A node running a different model is +therefore not a candidate, and pinning one SHALL fail saying what it is serving. + +#### Scenario: The preference decides between matching nodes + +- **WHEN** two nodes are running the wanted model and one reports a longer time + since it last did work +- **THEN** the one the activity preference favours is chosen, and the same fleet + in the same state chooses the same node every time + +#### Scenario: An unreachable node is skipped + +- **WHEN** one node in the fleet cannot be reached and another is running the + wanted model +- **THEN** the reachable node is chosen and the launch proceeds + +#### Scenario: A pinned node is used as given + +- **WHEN** the user runs `spinloop harness --node gpu-box` and that node is + running the wanted model +- **THEN** `gpu-box` is chosen without regard to what the other nodes are doing + +#### Scenario: A pinned node that cannot be reached fails + +- **WHEN** the user pins a node whose daemon is unreachable +- **THEN** the command fails naming that node, and no other node is selected + +#### Scenario: A not-ready engine is not a match + +- **WHEN** the only node running the wanted model reports its engine not ready, + and no other node is running it +- **THEN** the selection reports that nothing is serving the model, and the + failure names the node marked not ready + +#### Scenario: A missing readiness reading still routes + +- **WHEN** a node is running the wanted model and its daemon reports no + readiness at all +- **THEN** the node is chosen, as if its engine had answered + +#### Scenario: A pinned node that is still starting names its state + +- **WHEN** the user pins a node whose engine is running the wanted model but has + not answered yet +- **THEN** the command fails saying the node is still starting the model, that + it may be fetching or loading weights, and names the command for the node's + log, without restarting the node + +#### Scenario: A busy node is left alone + +- **WHEN** every reachable node is running a model other than the one wanted +- **THEN** no running engine is stopped, and selection falls through to waking + an idle node + +#### Scenario: Pinning a node serving something else fails + +- **WHEN** the user pins a node that is running a different model +- **THEN** the command fails saying what that node is serving, and the engine is + untouched + +### Requirement: Waking a node + +When no running node is serving what is wanted, routing SHALL wake one: it SHALL +choose a node that is not running, push what the Spinloop asks for as that node's +deploy config, start it through the daemon's start endpoint, and wait before +launching the agent — not merely until the node reports `running`, which says +only that a process exists, but until its engine endpoint answers. A node whose +stored config already matches the wanted model SHALL be preferred, since it has +the weights. + +The pushed config is the node-side counterpart of what `spinloop serve` would run +for that Spinloop, translated per engine. A node may be woken for an engine that +binds its model at launch — `llamacpp`, `vllm`, and `mtplx` — and a `MODEL` that +names a file on the node's own disk is a valid wake for it: the node has the +file, and only a destination that fetches its weights itself refuses a local +path. + +A node that refuses the config — a runner or model it cannot serve — SHALL NOT +fail the launch while other candidates remain: the next candidate SHALL be +tried, and the refusals SHALL be reported when none succeeds. + +Two clients may wake the same node at once. A start refused because an engine is +already running SHALL NOT fail the launch: the node's state SHALL be re-read, +and a node now serving what was wanted SHALL be used — and the launch SHALL +wait for that node's engine to answer before launching the agent, exactly as it +waits for a node it woken itself: the node that won the race may still be +loading weights, and the wait is bounded by the same timeout. Losing that race +is another route to the same place, not an error. + +The wait SHALL be bounded by a timeout and SHALL report what it is waiting for, +because a cold node loads weights before it answers. Exceeding the timeout SHALL +fail naming the node and the endpoint that did not come up; the started engine +SHALL be left running rather than stopped, so a slow load is not thrown away. + +`--no-wake` SHALL turn waking off: with no running node serving what is wanted +the command SHALL then fail, listing the nodes and their states and naming the +command that would start one. + +#### Scenario: An idle node is woken and used + +- **WHEN** a fleet-routed launch finds no node serving the wanted model and one + node is idle and able to serve it +- **THEN** that node is given the Spinloop's model as its deploy config, started, + and the agent launches against it once its engine answers + +#### Scenario: A node is woken for a Mac-only engine + +- **WHEN** a fleet-routed launch finds no node serving the wanted model, and an + idle node's daemon can run MTPLX +- **THEN** that node is woken with a config that runs the wanted model under + `mtplx serve`, and the agent launches against it once its engine answers + +#### Scenario: A local model path wakes the node that has it + +- **WHEN** the Spinloop's `MODEL` names a file on the woken node's disk +- **THEN** the wake carries that path as the model to load, rather than + refusing it as a local file + +#### Scenario: A Spinloop that pins a bind wakes a node bound to it + +- **WHEN** the Spinloop names a `BASEURL` and a node is woken for it +- **THEN** the engine the node starts binds to the address the `BASEURL` names, + exactly as `spinloop serve` would bind it, and the node reports that engine + as reachable rather than on the engine's own default + +#### Scenario: A started engine that is not yet loaded is waited for + +- **WHEN** a woken node reports `running` while its engine is still loading + weights and not yet answering +- **THEN** the launch waits for the engine to answer rather than launching the + agent against an endpoint that refuses connections + +#### Scenario: A node that cannot serve the model is passed over + +- **WHEN** the first idle candidate rejects the pushed config as unservable and + a second idle node accepts it +- **THEN** the second node is started and used + +#### Scenario: No node can serve it + +- **WHEN** every idle node rejects the config +- **THEN** the command fails, naming each node and the reason it refused + +#### Scenario: Losing the race to another client + +- **WHEN** a start is refused because another client woke the same node first, + and that node is now serving the wanted model +- **THEN** the launch uses that node rather than failing, waiting for its + engine to answer first if it is still loading + +#### Scenario: A node that never comes up + +- **WHEN** a woken node does not report running within the timeout +- **THEN** the command fails naming the node, and the engine it started is left + running rather than stopped + +#### Scenario: Waking is refused + +- **WHEN** `--no-wake` is passed and no node is serving the wanted model +- **THEN** the command fails, listing the nodes with their states and naming the + `spinloop fleet start` command that would start one, and nothing is started diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/remote-node/spec.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/remote-node/spec.md new file mode 100644 index 00000000..f42a6445 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/specs/remote-node/spec.md @@ -0,0 +1,59 @@ +## MODIFIED Requirements + +### Requirement: A remote environment is a fleet node + +A registered remote environment SHALL be representable as one member of the fleet's node +set, answering the same operations a local node answers: its status, its metrics, and +being started, stopped, and read for logs. The control plane's replies SHALL be mapped +onto the same status and metrics shapes a local node yields, so downstream fan-out and +rendering treat the two identically. A running environment's status SHALL in particular +carry what its engine is serving — the model it runs, and the served name the deploy gave +it beside the model id when there is one — so a client choosing a node by model matches +it the way it matches a local node, and the fleet view and the remote view name it the +same. It SHALL also carry where its engine answers — the instance's published address, +which the control plane knows and a daemon on the instance cannot — so a client can +reach the engine, not only name it; a stopped or undeployed environment reports none. + +A remote environment that cannot be reached, or whose control call is rejected — +including a rejected AWS credential — SHALL be reported as a typed outcome against that +environment, the same way an unreachable or unauthorized node is, rather than failing the +command or being silently dropped. + +Because a remote endpoint is provisioned by deployment rather than woken like a node, a +node-level start asked to run on a supplied deploy configuration SHALL be refused with a +message naming the deployment path, rather than attempted. + +#### Scenario: A remote environment answers status like a node + +- **WHEN** a remote environment is asked for its status as a member of a node set +- **THEN** it returns a status carrying the endpoint's state, what its engine is serving + (the model, and the served name beside it when the deploy gave one), where its engine + answers (the instance's published address), and, when the engine has done work, its + last-active time, in the same shape a local node's status carries + +#### Scenario: A freshly loaded engine shows its model before it has done work + +- **WHEN** a remote environment's engine is serving a model but has not yet answered a + request, so it reports no last-active time +- **THEN** its status still carries the model it is serving, so a router can match a + request to it before the first request has landed + +#### Scenario: A remote environment answers metrics like a node + +- **WHEN** a running remote environment is asked for its metrics as a member of a node set +- **THEN** it returns the token and system figures in the same stats shape a local node + returns + +#### Scenario: A rejected control call is a typed outcome + +- **WHEN** a remote environment's status or metrics call is rejected, for example because + the caller's credentials are not valid +- **THEN** the environment is reported with a failure outcome and the reason, and it does + not abort or blank the rest of the node set + +#### Scenario: Waking a remote environment is refused + +- **WHEN** a node-level start is requested for a remote environment, carrying a deploy + configuration +- **THEN** it is refused with a message naming the deployment path, and the environment + is not started diff --git a/openspec/changes/archive/2026-09-12-add-fleet-gateway/tasks.md b/openspec/changes/archive/2026-09-12-add-fleet-gateway/tasks.md new file mode 100644 index 00000000..0ae5d299 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-add-fleet-gateway/tasks.md @@ -0,0 +1,43 @@ +# Tasks + +## 1. Fleet foundations + +- [x] 1.1 Parse a top-level `wake` setting in fleet.yaml (`on`/`off`, default on when absent, any other value refused naming both) with a `fleet.Config` accessor for the policy in force, and verify with `internal/fleet` config tests: a file with `wake: off` reports off, a bare file reports on, and `wake: sometimes` fails naming `on` and `off` +- [x] 1.2 Generalise `fleet.Wake` from one deploy config for all candidates to a per-candidate config function, keeping the existing warm-first ordering, refusal collection, race handling, and readiness wait, and verify the existing `wake_test.go` suite passes with the launch path's constant config plus a new test where two candidates take two different configs +- [x] 1.3 Gate every use of a woken or raced node on its engine answering: lift the readiness wait out of `Wake`'s private helper into an exported form both call, make `Wake`'s already-running branch wait for the engine to answer rather than taking the node on state alone, and add a "usable now" test of running + name match (model id or served name) + ready-or-answering, verified by a wake test where a node that won the race is still loading is waited for, not returned +- [x] 1.4 Make daemon status report the served name beside the model id: the stored deploy config's served name is recorded with the served model and reported as an omitted-unless-set `servedName` field, and `docs/openapi.yaml` is updated to match, verified by a daemon test (an aliased start reports both names, an unaliased one reports no field) and `openapi_test.go` passing +- [x] 1.5 Keep a not-ready engine out of selection and wait for one that is loading: a node whose daemon reports not-ready is not a match (the pre-existing selection tests all route, which is the absent-reading case), a node already running the wanted model is waited for — first in fleet-file order, bounded by the wake timeout, left running on it — rather than passed over to a wake, a pinned node still starting fails saying so and naming the command for the node's log, and the failure rows mark a not-ready node, verified by select tests (a not-ready node is skipped, the skip is the same as nothing serving, the pinned failure names the state, `Loading` names the starting nodes) and gateway tests (a request holds for a starting node, the wait rather than waking another, a never-answering engine fails naming the node) + +## 2. The gateway + +- [x] 2.1 Build the `internal/gateway` handler: constructed from a `*fleet.Config`, a caller token, a wake timeout, and a per-node config callback, it authenticates with the daemon's token rules (file, `SPINLOOP_API_TOKEN`, command line; more than one a conflict; `401` without the right one; a non-loopback listen without a token refuses to start, loopback allowed), answers `GET /health` without touching any node, and refuses unknown paths with `404` naming the paths served, verified by handler tests against each rule +- [x] 2.2 Serve `GET /v1/models` as the OpenAI list of what the fleet's nodes are running — the served name when a running node reports one, otherwise the model id, duplicates once, nothing for a stopped node, an empty list not an error when nothing runs — verified by handler tests over a fake fleet in each state +- [x] 2.3 Reuse a status fan-out taken within the last two seconds rather than fanning out per request, the freshness a package variable tests can shorten, verified by a test counting fan-outs: a burst inside the window fans out once, a request after it fans out again +- [x] 2.4 Route `POST /v1/chat/completions` and `POST /v1/completions`: read the model from the body (a request naming none is refused saying so), select with the fleet's own ranking (name match on model id or served name, the fleet file's preference, fleet-file tie-break, a loopback-bound engine without an override never selected and named in the failure when it is the only candidate), wait for the chosen node to be usable, then reverse-proxy per request with streaming flushed as sent, the caller's authorisation swapped for the engine key the fleet entry names (none sent to an ungated engine), no read timeout on the upstream, no retry at another node, an upstream failure reaching the caller as an error naming the node, and one log line per routed request naming model, node, and outcome with no request body, verified by handler tests over fake engines: right node by model, key swapped not forwarded, a streamed reply passes through, an engine error is the caller's error, and the log line appears +- [x] 2.5 Wake for a request: when no running node matches and the fleet's wake policy allows it, start a node with the config its own Spinloop source resolves to (only nodes whose source names the requested model or served name are candidates, a node whose stored config already matches tried first), gate it with the key its fleet entry names, hold the request until the engine answers bounded by the wake timeout (a timeout failing the request saying so and leaving the engine running), share one in-flight wake per node between concurrent requests so a burst starts at most one engine, and with the policy off — or no node's source describing the model — fail without starting anything, naming the nodes and the ways a source could have been given, verified by handler tests: a cold request is woken and answered, concurrent cold requests produce one start, a timeout leaves the engine running, `wake: off` refuses naming the node, and a sourceless node is named in the failure +- [x] 2.6 List what a request can reach from `/v1/models`: beside what the running nodes report, when the fleet's wake allows it each node that is not running and answers its status contributes the model its own source describes — the served name when the source gives one — with each node's source resolved at most once in a short window shared by models requests, and nothing contributed for a node the gateway cannot start (a remote environment), a fleet whose wake is off, or a running node beyond what it reports, verified by handler tests: a stopped node's model appears beside what runs under its served name, a running node's source adds no second model, `wake: off` and a remote environment list nothing beyond what runs, a model two sources describe is listed once, and a burst resolves each source once + +## 3. The CLI + +- [x] 3.1 Add the `spinloop gateway` command beside `serve`: `--fleet`/`-f`, `--listen` defaulting to port 4000 on all interfaces, the daemon's three token sources, `--wake-timeout`; it resolves the fleet file and token at startup (failures naming the fix, nothing listening on failure), prints the address to name in a Spinloop's `FLEET`, and serves until stopped, verified by a command test (a gateway answers `/v1/models` on the printed address, a non-loopback listen without a token refuses naming the ways to supply one, a missing fleet file names `./fleet.yaml`) and by the completion dispatch-coverage scan passing with the new command +- [x] 3.2 Implement the launch path's endpoint branch: a `FLEET` — instruction or flag — carrying a scheme yields a choice naming the endpoint (the OpenAI-compatible prefix appended when the value has no path, a value with a path used as given), no fleet file read and no node contacted, the endpoint's token resolved through the launch's existing key chain (an ENV instruction, then the environment, then the `.env` beside the Spinloop, an already-set variable winning) with a missing value failing before anything is written and naming `OPENAI_API_KEY`, node-steering flags inert, and a pinned `BASEURL` still winning, verified by launch tests: a Spinloop with `FLEET http://gw:4000` gives the agent `OPENAI_BASE_URL` of `http://gw:4000/v1` and the token as its key, a Spinloop with a path in the URL is used as given, a missing token fails naming the variable and writes nothing, and a pinned `BASEURL` is not routed +- [x] 3.3 Make `spinloop fleet route` answer a `FLEET` that names an endpoint: it reports the address the launch will be given, queries no node, and starts nothing, replacing its "not implemented yet" refusal, verified by a command test on a Spinloop with an endpoint `FLEET` in a directory holding no fleet file +- [x] 3.4 Gate the launch path's waking on the fleet's `wake` setting: `wake: off` refuses to start a node the way `--no-wake` does — the failure names the node that would be woken and the `spinloop fleet start` command that would start it — and an explicit `--no-wake` still wins over a file that allows waking, verified by launch tests for each of the three combinations + +## 4. Example and documentation + +- [x] 4.1 Add a standalone `examples/gateway-docker/` — its own Dockerfile, shim, engine files, `compose.yaml` (two fleet nodes plus a gateway service, no reference to `examples/fleet-docker/`), a `fleet.yaml` whose nodes name their tokens and engine keys by variable, a `.env.example`, a client Spinloop whose `FLEET` is the gateway's address, a `run-tests.sh` asserting the gateway's surface (a model is listed once running, a cold request wakes a node and streams a reply, the engine key is injected and never reaches the caller, a wrong token is `401`, `wake: off` refuses), and a `README.md` with the run steps, verified by `./run-tests.sh` passing from a clean checkout +- [x] 4.2 Document the change where the rest of it is documented: `docs/commands/gateway.md` for the new command (flags, the token's three sources and the non-loopback rule, the address it prints), the endpoint form of `FLEET` and the token it resolves in `docs/spinloop-file.md`, the `wake` setting in the fleet file documentation, and `docs/README.md` pointing at the new example, verified by the docs links resolving and the new pages reading against the implemented flags + +## 5. Remote nodes report what they serve + +A running remote node reported its state but no model — the control plane's status branch read the on-instance daemon's status but relayed only its activity, dropping the serving facts. A gateway (or `fleet status`) matching a request by model could not see a running remote node. + +- [x] 5.1 Make the start Lambda's status branch read the environment's stored deploy config for the serving facts: `readDeployFacts` returns the runner, model id and served name from the same config the stats reply reads (the daemon's own model is the on-disk weights path, which a router cannot match, so it is never reported), read concurrently with the daemon's activity and never turning a working status into a failing one, verified by the start-status tests (facts reported with and without activity, a model present before the engine has done work, an absent served name stays absent, a failed config read drops only the facts) +- [x] 5.2 Map the relayed serving facts onto the node's status: `remote.Response` gains the served name, and `statusFromRemote` carries runner, model and served name, leaving them empty when the daemon reports none, verified by a remote-node status test (facts mapped across, and an absent fact stays empty) +- [x] 5.3 Note in the proposal and the `remote-node` spec that a running environment's status carries what its engine is serving, and that a control-plane redeploy is needed for live environments to report it +- [x] 5.4 Carry the engine's address on a remote node's status and resolve it in routing: `EngineEndpoint` gains the host a node reports, `statusFromRemote` maps the control plane's published base url to the endpoint's host, port and path (none when the environment is not running), `EngineBaseURL` uses a reported engine host in place of the fleet file's, and `docs/openapi.yaml` is updated to match, verified by the remote-node status tests (the address maps across, absent when stopped), the `EngineBaseURL` tests (a reported host is used, an override still wins), a gateway test (a remote node is a candidate and a request is routed to its published address with the resolved engine key), and `openapi_test.go` passing + +## 6. Final verification + +- [x] 6.1 Run the full gate: `go test ./... -cover` at or above the 80% total, `go vet ./...`, `gofmt -l .` clean, the control plane's `pnpm build` and `pnpm test`, and `openspec validate add-fleet-gateway` passing, fixing whatever each reports diff --git a/openspec/specs/daemon-api/spec.md b/openspec/specs/daemon-api/spec.md index fcf878b5..f367068b 100644 --- a/openspec/specs/daemon-api/spec.md +++ b/openspec/specs/daemon-api/spec.md @@ -12,7 +12,9 @@ a non-loopback listen without a bearer token refuses to start outright rather than serving unauthenticated. And it answers questions rather than handing out raw material — status reports how long the engine has been idle, not the counters a caller would have to compare for itself. + ## Requirements + ### Requirement: API exposure `spinloop daemon` SHALL always expose the control API — it is the command's @@ -485,3 +487,26 @@ taken, the field SHALL be omitted rather than empty. - **WHEN** a metrics request is made on a daemon that has never run an engine - **THEN** the response carries no history field + +### Requirement: Status reports the name the model is served under + +Status SHALL report, beside the model id it already reports, the name the +engine serves the model under — the deploy config's served name — when the +stored config names one. A router that sees only status needs the name a +request will carry: a client that started the engine under an alias addresses +it by that alias, and matching a request only against the model id would not +recognise the node. The served name SHALL be omitted when the stored config +names none, and the model id SHALL be reported on the same terms it is today +whatever the served name is. + +#### Scenario: An aliased engine reports both names + +- **WHEN** an engine was started from a config that names both a model and a + served name, and a status request is made +- **THEN** the response reports the model id and the served name + +#### Scenario: No served name, no field + +- **WHEN** an engine was started from a config that names no served name, and a + status request is made +- **THEN** the response reports the model id and carries no served name diff --git a/openspec/specs/fleet-config/spec.md b/openspec/specs/fleet-config/spec.md index dc2af4df..a5d0b398 100644 --- a/openspec/specs/fleet-config/spec.md +++ b/openspec/specs/fleet-config/spec.md @@ -4,7 +4,9 @@ Define `fleet.yaml`: the file that names the machines a `spinloop fleet` client observes and how to reach each one's daemon control API — the fleet's single source of what nodes exist, kept separate from the per-node secrets. + ## Requirements + ### Requirement: Fleet file format A `fleet.yaml` SHALL declare a list of nodes, each with a unique `name`, a @@ -318,3 +320,43 @@ given: the `file` field, a `spinloop alias` named after the node, or a - **THEN** `fleet start` fails for that node, naming the `file` field, the alias registry, and the subdirectory convention as the three ways a source could have been given + +### Requirement: Fleet-wide wake policy + +A fleet file MAY declare a top-level `wake` value of `on` or `off`, deciding +whether routing starts an engine on a node that is not running one when no +running node serves what is wanted. It belongs to the file rather than to each +node, for the reason `prefer` does: it describes how this cluster is to be +used — may work be started on its machines on demand, or only used where it is +already running — which is a property of the fleet, not of any one machine in +it. + +A file declaring nothing SHALL wake, as routing does when the setting is +absent: waking is the difference between a fleet that answers a request and +one that must be prepared by hand, and the file's author is the one who owns +the machines it names. A file declaring anything other than `on` or `off` SHALL +fail to parse, naming both accepted values, in keeping with the file's other +validation. + +The setting SHALL decide whether to wake only. It SHALL NOT change which node +is chosen, how matching nodes are ranked, or what a wake does: a fleet that +declares `wake: off` still reports, when nothing is running, the node whose +source describes the wanted model and the command that would start it. + +#### Scenario: A fleet that declares nothing wakes + +- **WHEN** a fleet file declares no `wake` setting and routing finds no node + serving what is wanted +- **THEN** routing starts an engine on a suitable node, as it does today + +#### Scenario: A fleet that refuses to wake + +- **WHEN** a fleet file declares `wake: off` and routing finds no node serving + what is wanted +- **THEN** nothing is started, and the failure names the node that would be + woken and the command that would start it + +#### Scenario: An unknown value is rejected at parse time + +- **WHEN** a fleet file declares `wake: sometimes` +- **THEN** parsing fails naming `on` and `off` diff --git a/openspec/specs/fleet-gateway/spec.md b/openspec/specs/fleet-gateway/spec.md new file mode 100644 index 00000000..43922180 --- /dev/null +++ b/openspec/specs/fleet-gateway/spec.md @@ -0,0 +1,390 @@ +# fleet-gateway Specification + +## Purpose +One OpenAI-compatible endpoint in front of a fleet: a foreground process that +answers agent requests by choosing a node with the fleet's own selector, holds +each node's engine key, and wakes a node when nothing is serving what a request +asks for — so a machine running an agent needs nothing but a URL and one token. + +## Requirements + +### Requirement: The gateway command + +`spinloop gateway` SHALL run in the foreground, the way `spinloop serve` does, +holding the fleet file it serves: `--fleet ` (with the `-f ` short +form) when given, otherwise `./fleet.yaml` in the working directory, and a +missing file SHALL fail naming the expected path, as the fleet commands do. +The gateway SHALL listen on an address given by `--listen`, defaulting to +port 4000 on all interfaces, and SHALL print the address to name in a fleet +file's `gateway` section when it starts. + +The gateway's own startup failures SHALL be the fleet file's: a fleet file that +does not parse, or that names a token variable set nowhere, SHALL fail the +gateway at startup naming the problem, rather than listening and failing per +request. + +#### Scenario: The gateway starts and answers + +- **WHEN** the user runs `spinloop gateway --listen :4000` in a directory + holding a `fleet.yaml` +- **THEN** it prints the address to name in the fleet file's `gateway` section + and serves requests until it is stopped + +#### Scenario: An explicit fleet file is served + +- **WHEN** the gateway is given a `--fleet` path +- **THEN** that file is the one it serves + +#### Scenario: A missing fleet file names itself + +- **WHEN** the user runs `spinloop gateway` in a directory holding no + `fleet.yaml` and passes no `--fleet` +- **THEN** it fails naming `./fleet.yaml`, and nothing listens + +#### Scenario: A broken fleet file fails at startup + +- **WHEN** the gateway's fleet file names a token variable that is set nowhere +- **THEN** the gateway fails at startup naming the node and the variable, + rather than listening + +### Requirement: Caller authentication + +The gateway SHALL authenticate callers with one bearer token, on the same +terms the daemon's control API authenticates: the token MAY be supplied by a +file (`--api-token-file`), the environment (`SPINLOOP_API_TOKEN`), or the +command line (`--api-token`); giving more than one SHALL fail naming the +conflict. Requests without the correct token SHALL be rejected with `401`. +When no token is configured, the gateway SHALL refuse to listen on a +non-loopback address and SHALL say why, and listening on loopback without a +token SHALL be allowed. + +The caller's token authorises use of the gateway only. It SHALL NOT be +forwarded to a node's engine or control API: the gateway reaches a node with +the node's own credentials, resolved from the fleet file the way every other +fleet client does. + +#### Scenario: A wrong token is rejected + +- **WHEN** a request carries a missing or incorrect bearer token +- **THEN** the response is `401` and no node is contacted + +#### Scenario: A tokenless non-loopback listen refuses to start + +- **WHEN** the gateway would listen on a non-loopback address and no token is + configured +- **THEN** startup fails saying a token is required for non-loopback exposure, + naming every way one can be supplied + +#### Scenario: A tokenless loopback is permitted + +- **WHEN** the gateway listens on a loopback address with no token configured +- **THEN** it serves requests without authentication + +#### Scenario: The caller's token stops at the gateway + +- **WHEN** an authenticated request is routed to a node +- **THEN** the node is reached with the node's own credentials from the fleet + file, and the caller's token is not sent to it + +### Requirement: Listing the fleet's models + +The gateway SHALL serve `GET /v1/models` returning, in the OpenAI list shape, +the union of the models a request can reach. For each node whose state is +`running`, the list SHALL carry the name it reports serving — the served name +when it reports one, otherwise the model id. When the fleet's wake allows +starting an engine, the list SHALL additionally carry, for each node that is +not running and answers its status, the model that node's own Spinloop source +describes — the same served-name-first naming the wake would start it with — +since a request naming that model starts that node. A running node SHALL +contribute nothing but what it reports: a running engine is never displaced, +so its source's model is not a request the gateway would answer from it. A +node the gateway cannot start — a remote environment, which starts from +`spinloop remote deploy`, not from a request — and a fleet whose wake is off +SHALL contribute nothing beyond what is running, and duplicates SHALL be +listed once. The source a node describes SHALL be resolved at most once in a +short window shared by all models requests, so a burst does not re-read every +node's source. + +#### Scenario: Running models are listed + +- **WHEN** two nodes are running, one serving a model under an alias and one + under its id, and a models request is made +- **THEN** the response lists the alias and the id, each once + +#### Scenario: A stopped node's wakeable model is listed + +- **WHEN** a node is stopped, its own source describes a model, and the fleet + allows waking, and a models request is made +- **THEN** the response lists the model the source describes, beside what the + running nodes serve + +#### Scenario: A stopped node's model is not listed when wake is off + +- **WHEN** a node is stopped and the fleet's wake is off, and a models request + is made +- **THEN** the response lists only what the running nodes serve + +#### Scenario: A stopped remote environment's model is not listed + +- **WHEN** a remote environment is stopped and a models request is made +- **THEN** the response does not list what its source describes: the gateway + cannot start it, so a request naming that model would fail + +#### Scenario: A running node's source adds no second model + +- **WHEN** a running node reports one model and its source describes another, + and a models request is made +- **THEN** the response lists only what the node reports + +#### Scenario: A burst of models requests resolves each source once + +- **WHEN** several models requests arrive within the window in which a node's + source is resolved +- **THEN** each node's source is read once for the burst + +#### Scenario: Nothing reachable lists nothing + +- **WHEN** no node is running and nothing is wakeable — no stopped node's + source describes a model, or the fleet's wake is off — and a models request + is made +- **THEN** the response is an empty list, not an error + +### Requirement: Routing a request to a node + +The gateway SHALL serve `POST /v1/chat/completions` and `POST /v1/completions` +by choosing a node and reverse-proxying the request to that node's engine. The +choice SHALL be the fleet's own selection: every node's state is considered, a +node matches when what it reports serving — its model id or its served name — +equals the model the request names, and matching nodes are ranked by the fleet +file's activity preference, ties broken by fleet-file order. A request naming +no model SHALL be refused saying so, not routed at a guess. + +A node whose daemon reports the engine not ready SHALL NOT be selected: the +state turns running when the engine's process exists, which is before the +weights are fetched and loaded, and during that window nothing is listening on +the engine's port. A node whose daemon reports no readiness at all SHALL NOT be +disqualified: the absence of a reading is not evidence of not-readiness. + +When nothing is serving the model the request names, a node already running it +whose engine has not answered yet SHALL be waited for rather than passed over: +it is loading the weights the request needs, so it serves sooner than anything +a wake would start from cold, and waking a second node would leave two engines +up for one request. The wait SHALL be bounded by the wake timeout, SHALL name +the node it holds the request for, and a node that does not answer in time +SHALL fail the request naming it, leaving its engine running. Waiting is not +waking: a fleet whose wake is off SHALL still hold a request for a node already +loading the model it named, since no engine is being started. + +The request's body SHALL reach the chosen engine unmodified, and the engine's +reply SHALL reach the caller unmodified, including a streamed reply, which the +gateway SHALL pass through without holding it back. The gateway SHALL replace +the caller's authorisation with the engine key it holds for that node — the +value the node's fleet entry names, resolved the way every other fleet client +resolves it — and SHALL send no authorisation at all to an engine that needs +none. The gateway SHALL NOT retry a failed request at another node: the reply +the chosen engine gives is the reply the caller gets. + +The engine's address SHALL be resolved the way routing resolves it: the node's +declared engine override as given, otherwise the node's host with the port and +path the engine reports. Where a node reports its engine's host — a remote +environment, whose control plane publishes the instance's address and which the +fleet file names by environment alone — that reported host SHALL be used in +place of the node's host, so the request reaches the instance rather than an +address the fleet file never held. A node that reports its engine bound to +loopback, without an override taking responsibility for reachability, SHALL NOT +be selected: it answers only on its own machine, and where no other node serves +the wanted model the request SHALL fail saying so and naming the fix. + +The gateway SHALL reuse node state it has recently read — a reading taken +within the last couple of seconds — rather than query every node on every +request, so a burst of requests does not pay a fan-out each. The freshness of +the reading SHALL NOT matter to the choice: the same fleet in the same state +chooses the same node. + +Each routed request SHALL be logged as one line — the model, the node chosen, +and the outcome — and the gateway SHALL log no request body. + +#### Scenario: A request goes to the node serving its model + +- **WHEN** one node is running the model a request names and another is + running a different model +- **THEN** the request is proxied to the first node and the engine's reply + reaches the caller unmodified + +#### Scenario: The engine key is swapped, not the caller's token + +- **WHEN** a request is routed to a node whose engine is gated, and the node's + fleet entry names the key +- **THEN** the engine receives the key's value as its authorisation, and the + caller's token is not sent to the engine + +#### Scenario: A streamed reply passes through + +- **WHEN** a streamed completion is requested and the chosen engine streams its + reply +- **THEN** the caller receives the stream as the engine sent it + +#### Scenario: An engine error is the caller's error + +- **WHEN** the chosen engine refuses the request +- **THEN** the engine's refusal reaches the caller and no other node is tried + +#### Scenario: A request naming no model is refused + +- **WHEN** a completion request carries no model +- **THEN** it is refused saying the request names no model, and no node is + contacted + +#### Scenario: A loopback-bound engine is not selected + +- **WHEN** the only node serving the wanted model reports its engine bound to + loopback and names no engine override, and the node is not reached over + loopback +- **THEN** the request fails saying the engine answers only on that machine, + naming the bind and the override as the fixes + +#### Scenario: A not-ready engine is not selected + +- **WHEN** the only node running the model a request names reports its engine + not ready, and the fleet's wake is off +- **THEN** the request fails naming that node marked not ready, and nothing is + started + +#### Scenario: A request waits for an engine that is still starting + +- **WHEN** no node is serving the model a request names, one node is running + it but its engine has not answered yet, and no other node is running it +- **THEN** the request is held for that node and answered when its engine + answers, and no other node is started + +#### Scenario: A still-starting engine that does not answer fails the request + +- **WHEN** a node is running the model a request names, its engine has not + answered yet, and it does not answer within the wake timeout +- **THEN** the request fails saying so, naming the node, and the engine is left + running + +#### Scenario: Wake off still waits for an engine already starting + +- **WHEN** the fleet file declares the wake policy off and the only node + running the model a request names has not answered its engine yet +- **THEN** the request is held for that node and answered when the engine + answers, since waiting starts nothing + +#### Scenario: A burst of requests does not fan out per request + +- **WHEN** several requests arrive within a couple of seconds of each other +- **THEN** the fleet is queried once for the burst, and the requests are + answered from that reading + +#### Scenario: A routed request leaves one log line + +- **WHEN** a request is routed and answered +- **THEN** the gateway's log gains one line naming the model, the node, and the + outcome, and carries no part of the request body + +### Requirement: Waking a node for a request + +When no running node serves the model a request names, and the fleet file's +wake policy allows it, the gateway SHALL start an engine on a node that is not +running one, and SHALL hold the request until the engine answers. A node is a +wake candidate when it is not running and the Spinloop source it names — its +`file` field, a registered alias named after it, or a same-named directory +beside the fleet file, resolved the way `spinloop fleet start` resolves it — +describes a config whose model or served name is the one the request asks for: +a node is started with what it was told to run, never with a config invented +for the request. Candidates whose stored config already names the model SHALL +be tried first, since they have the weights, and the rest in fleet-file order. +A node that refuses the start — a runner or model it cannot serve — SHALL NOT +fail the request while other candidates remain. + +The started engine SHALL be gated with the key the node's fleet entry names, +supplied by the gateway: the gateway is the client that starts the engine, so +the key the client sets is the key the engine takes. The wait SHALL be bounded +by a wake timeout, defaulting to five minutes and overridable by +`--wake-timeout`; exceeding it SHALL fail the request saying the engine did +not answer in time, and the started engine SHALL be left running rather than +stopped, so a slow load is not thrown away. + +When several requests ask for a model nothing is serving at once, the gateway +SHALL start at most one engine per node and answer every request from it: the +first request's wait is the wait the rest join. A node another request woke +first SHALL be used the same way, and only once its engine answers. + +With the wake policy off, a request for a model nothing is serving SHALL fail +without starting anything, naming the nodes and what they could serve, and the +command that would start one. A model no node is running and no node's source +describes SHALL fail the same way, whatever the policy: nothing to wake with, +and the failure SHALL say so rather than trying to start a node with nothing. + +#### Scenario: A cold request wakes a node and is served + +- **WHEN** no node is running the model a request names, one node's Spinloop + source describes it, and the wake policy allows it +- **THEN** that node is started with its own config, gated with the key its + fleet entry names, and the request is answered once the engine answers + +#### Scenario: The request is held while the engine loads + +- **WHEN** the woken node reports running while its engine is still loading +- **THEN** the request waits, and is answered when the engine answers, rather + than failing against an endpoint that refuses connections + +#### Scenario: A wake that does not finish in time fails the request + +- **WHEN** a woken node's engine does not answer within the wake timeout +- **THEN** the request fails saying so, naming the node, and the engine is left + running + +#### Scenario: Concurrent cold requests share one wake + +- **WHEN** two requests arrive at once for a model nothing is serving, and one + node's source describes it +- **THEN** that node is started once, and both requests are answered from the + same engine + +#### Scenario: A woken engine takes the gateway's key + +- **WHEN** the gateway starts an engine on a node whose fleet entry names an + engine key +- **THEN** the engine is gated with that value, the gateway's requests to it + carry it, and no reply to any caller contains it + +#### Scenario: Wake refused by the fleet file + +- **WHEN** the fleet file declares the wake policy off and no node is serving + the model a request names +- **THEN** nothing is started, and the request fails naming the node whose + source describes the model and the command that would start it + +#### Scenario: Nothing can serve the model + +- **WHEN** no node is running the model a request names and no node's Spinloop + source describes it +- **THEN** the request fails, naming each node and why it cannot serve the + model, and nothing is started + +#### Scenario: A sourceless node is not woken + +- **WHEN** the only node that could take a request names no Spinloop source + that resolves +- **THEN** it is not started, and the failure names it and the ways a source + could have been given + +### Requirement: Paths the gateway does not serve + +A path other than `/v1/models`, `/v1/chat/completions`, `/v1/completions`, and +the gateway's own health path SHALL be answered with `404` naming the paths the +gateway serves. The health path SHALL answer that the gateway is up without +contacting any node, so an operator can tell the gateway down from the fleet +down. + +#### Scenario: An unknown path is named as such + +- **WHEN** a request is made to a path the gateway does not serve +- **THEN** the response is `404` and names the paths it does serve + +#### Scenario: The health path does not touch the fleet + +- **WHEN** the health path is requested and every node is unreachable +- **THEN** it still answers that the gateway is up diff --git a/openspec/specs/fleet-routing/spec.md b/openspec/specs/fleet-routing/spec.md index c63566ae..5414ad68 100644 --- a/openspec/specs/fleet-routing/spec.md +++ b/openspec/specs/fleet-routing/spec.md @@ -5,7 +5,9 @@ Connecting a harness launch to the fleet: choosing which node serves the agent spinloop is about to launch, waking that node when nothing is serving yet, and turning the choice into the base URL and key the launched agent authenticates with — so a machine that can reach the fleet needs no addresses of its own. + ## Requirements + ### Requirement: A fleet-routed launch `spinloop harness` SHALL route through a fleet when the Spinloop it wears names one @@ -82,6 +84,16 @@ node whose state is `running` and whose served model matches the Spinloop's `MODEL` (or its `ALIAS`, against the name the node reports serving). A Spinloop that names no model SHALL match any running node. +A node whose state is `running` but whose daemon reports the engine not ready is +not running what is wanted: the state turns running when the engine's process +exists, which is before the weights are fetched and loaded, and during that +window nothing is listening on the engine's port. Such a node SHALL NOT be +selected, and a failure that names it SHALL mark it not ready, so a refusal does +not read as if the node were serving the model it was asked for. A node whose +daemon reports no readiness at all — an older build, or a runner with no +health-check convention — SHALL NOT be disqualified: the absence of a reading is +not evidence of not-readiness. + Matching nodes SHALL be ranked by the activity preference in force (see "Preferring an idle or an active node"). Ties SHALL be broken by fleet-file order, so the same fleet in the same state chooses the same node. @@ -93,7 +105,11 @@ row rather than a failure in `spinloop fleet status`. `--node ` SHALL pin the selection to one node, skipping the search. An unknown name SHALL fail naming the known nodes, and a pinned node that cannot be reached SHALL fail rather than falling back to another node — a pin is an -instruction, not a preference. +instruction, not a preference. A pinned node that is running the wanted model +but has not answered yet SHALL fail saying it is still starting it — it may be +fetching or loading weights — and naming the command for the node's log, rather +than saying that nothing serves the model or restarting the node: this node is +about to serve it. A running engine SHALL NEVER be stopped or restarted to make room, including a pinned one: another person may be using it. A node running a different model is @@ -123,6 +139,27 @@ therefore not a candidate, and pinning one SHALL fail saying what it is serving. - **WHEN** the user pins a node whose daemon is unreachable - **THEN** the command fails naming that node, and no other node is selected +#### Scenario: A not-ready engine is not a match + +- **WHEN** the only node running the wanted model reports its engine not ready, + and no other node is running it +- **THEN** the selection reports that nothing is serving the model, and the + failure names the node marked not ready + +#### Scenario: A missing readiness reading still routes + +- **WHEN** a node is running the wanted model and its daemon reports no + readiness at all +- **THEN** the node is chosen, as if its engine had answered + +#### Scenario: A pinned node that is still starting names its state + +- **WHEN** the user pins a node whose engine is running the wanted model but has + not answered yet +- **THEN** the command fails saying the node is still starting the model, that + it may be fetching or loading weights, and names the command for the node's + log, without restarting the node + #### Scenario: A busy node is left alone - **WHEN** every reachable node is running a model other than the one wanted @@ -223,8 +260,11 @@ tried, and the refusals SHALL be reported when none succeeds. Two clients may wake the same node at once. A start refused because an engine is already running SHALL NOT fail the launch: the node's state SHALL be re-read, -and a node now serving what was wanted SHALL be used. Losing that race is -another route to the same place, not an error. +and a node now serving what was wanted SHALL be used — and the launch SHALL +wait for that node's engine to answer before launching the agent, exactly as it +waits for a node it woken itself: the node that won the race may still be +loading weights, and the wait is bounded by the same timeout. Losing that race +is another route to the same place, not an error. The wait SHALL be bounded by a timeout and SHALL report what it is waiting for, because a cold node loads weights before it answers. Exceeding the timeout SHALL @@ -255,6 +295,13 @@ command that would start one. - **THEN** the wake carries that path as the model to load, rather than refusing it as a local file +#### Scenario: A Spinloop that pins a bind wakes a node bound to it + +- **WHEN** the Spinloop names a `BASEURL` and a node is woken for it +- **THEN** the engine the node starts binds to the address the `BASEURL` names, + exactly as `spinloop serve` would bind it, and the node reports that engine + as reachable rather than on the engine's own default + #### Scenario: A started engine that is not yet loaded is waited for - **WHEN** a woken node reports `running` while its engine is still loading @@ -277,7 +324,8 @@ command that would start one. - **WHEN** a start is refused because another client woke the same node first, and that node is now serving the wanted model -- **THEN** the launch uses that node rather than failing +- **THEN** the launch uses that node rather than failing, waiting for its + engine to answer first if it is still loading #### Scenario: A node that never comes up @@ -435,4 +483,3 @@ authenticate is worse than a message that says so. - **WHEN** `OPENAI_API_KEY` is already set in the user's environment and a fleet-routed launch runs - **THEN** the existing value reaches the agent unchanged - diff --git a/openspec/specs/remote-node/spec.md b/openspec/specs/remote-node/spec.md index 658e63ec..20ecb8e5 100644 --- a/openspec/specs/remote-node/spec.md +++ b/openspec/specs/remote-node/spec.md @@ -4,14 +4,22 @@ Let a remote, scale-to-zero inference environment — one driven through its cloud control plane — be observed and driven the same way a fleet node is, so the two clients share one driver and one source for its status facts instead of each keeping its own copy. + ## Requirements + ### Requirement: A remote environment is a fleet node A registered remote environment SHALL be representable as one member of the fleet's node set, answering the same operations a local node answers: its status, its metrics, and being started, stopped, and read for logs. The control plane's replies SHALL be mapped onto the same status and metrics shapes a local node yields, so downstream fan-out and -rendering treat the two identically. +rendering treat the two identically. A running environment's status SHALL in particular +carry what its engine is serving — the model it runs, and the served name the deploy gave +it beside the model id when there is one — so a client choosing a node by model matches +it the way it matches a local node, and the fleet view and the remote view name it the +same. It SHALL also carry where its engine answers — the instance's published address, +which the control plane knows and a daemon on the instance cannot — so a client can +reach the engine, not only name it; a stopped or undeployed environment reports none. A remote environment that cannot be reached, or whose control call is rejected — including a rejected AWS credential — SHALL be reported as a typed outcome against that @@ -25,8 +33,17 @@ message naming the deployment path, rather than attempted. #### Scenario: A remote environment answers status like a node - **WHEN** a remote environment is asked for its status as a member of a node set -- **THEN** it returns a status carrying the endpoint's state and, when the engine has done - work, its last-active time, in the same shape a local node's status carries +- **THEN** it returns a status carrying the endpoint's state, what its engine is serving + (the model, and the served name beside it when the deploy gave one), where its engine + answers (the instance's published address), and, when the engine has done work, its + last-active time, in the same shape a local node's status carries + +#### Scenario: A freshly loaded engine shows its model before it has done work + +- **WHEN** a remote environment's engine is serving a model but has not yet answered a + request, so it reports no last-active time +- **THEN** its status still carries the model it is serving, so a router can match a + request to it before the first request has landed #### Scenario: A remote environment answers metrics like a node @@ -202,4 +219,3 @@ follow of the same node. - **WHEN** a follow of a remote node's log is closed and reopened - **THEN** the reopened follow shows the node's current tail, not an empty result because those events were already shown by the previous follow - diff --git a/remote/lambda/shared/daemon.ts b/remote/lambda/shared/daemon.ts index 3f3a72de..03844e3b 100644 --- a/remote/lambda/shared/daemon.ts +++ b/remote/lambda/shared/daemon.ts @@ -80,6 +80,11 @@ export interface DaemonStatus { state: string; runner?: string; model?: string; + /** + * The name the engine answers to beside the model id — the served name the + * deploy named it. Absent when the deploy gave the engine no other name. + */ + servedName?: string; uptimeSeconds?: number; logPath?: string; lastActiveAt?: string; diff --git a/remote/lambda/start/index.ts b/remote/lambda/start/index.ts index 5eec1f24..00c82b96 100644 --- a/remote/lambda/start/index.ts +++ b/remote/lambda/start/index.ts @@ -163,17 +163,20 @@ async function status(env: string): Promise { }); } // Concurrently, not in sequence: status is what you type repeatedly while - // waiting for a box, and this branch should cost the slower of the two SSM - // calls rather than their sum. - const [healthy, activity] = await Promise.all([ + // waiting for a box, and this branch should cost the slowest of these calls + // rather than their sum. The model facts come from the deploy config — the + // same source the stats reply reads — and the activity from the daemon. + const [healthy, activity, deploy] = await Promise.all([ checkHealth(instance.instanceId), readDaemonActivity(instance.instanceId), + readDeployFacts(env), ]); const result: Record = { state: 'running', environment: env, healthy, base_url: baseUrl, + ...deploy, ...activity, }; if (instance.retainUntil) { @@ -183,12 +186,12 @@ async function status(env: string): Promise { } /** - * Ask the instance's daemon when its engine last did work. Every failure — - * SSM error, unreachable daemon, unparseable reply, an engine that has done - * nothing yet — yields an empty object, so the caller spreads nothing and the - * report is exactly what it would have been. This must never be able to turn - * a working status into a failing one, which is why it is kept out of the - * `healthy` expression. + * Ask the instance's daemon when its engine last did work. Every failure — an + * SSM error, an unreachable daemon, an unparseable reply, an engine that has + * not done any work yet — yields an empty object, so the caller spreads + * nothing and the report is exactly what it would have been without this. It + * is kept out of the `healthy` expression so it can never turn a working + * status into a failing one. */ async function readDaemonActivity( instanceId: string, @@ -298,6 +301,40 @@ async function seedingGate( return seedingReply(seedId, `seeding the weights — ${follow}`); } +/** + * Read the environment's stored deploy config and keep the facts that name + * what it is serving: the runner, the model id, and the served name where the + * deploy gave one. This is the same source the stats reply — and so the remote + * status view — reads, so the fleet and remote views name a model identically. + * The daemon's own model is the on-disk weights path, which a router cannot + * match a request against, so the path is never reported here. A failed read + * yields an empty object, so an environment whose config cannot be read still + * reports its state and activity without a made-up model. + */ +async function readDeployFacts(env: string): Promise<{ + runner?: string; + modelId?: string; + servedName?: string; +}> { + try { + const cfg = await readDeployConfig(deployConfigParam(env)); + const facts: { runner?: string; modelId?: string; servedName?: string } = {}; + if (cfg.runner) { + facts.runner = cfg.runner; + } + if (cfg.modelId) { + facts.modelId = cfg.modelId; + } + if (cfg.servedModelName) { + facts.servedName = cfg.servedModelName; + } + return facts; + } catch (err) { + console.log(JSON.stringify({ phase: 'deploy-facts', error: errorName(err) })); + return {}; + } +} + /** POST — launch the environment's instance if needed and block until serving. */ async function wake( env: string, diff --git a/remote/test/start-status.test.ts b/remote/test/start-status.test.ts index 56aeed8b..a7ce6f77 100644 --- a/remote/test/start-status.test.ts +++ b/remote/test/start-status.test.ts @@ -29,6 +29,7 @@ const LAMBDA_ENV = { const findManagedInstance = vi.fn(); const isSsmAgentOnline = vi.fn(); const runShellCommand = vi.fn(); +const readDeployConfig = vi.fn(); const findEnvEip = vi.fn(); vi.mock('../lambda/shared/aws', async (importOriginal) => ({ @@ -36,6 +37,7 @@ vi.mock('../lambda/shared/aws', async (importOriginal) => ({ findManagedInstance: (...args: unknown[]) => findManagedInstance(...args), isSsmAgentOnline: (...args: unknown[]) => isSsmAgentOnline(...args), runShellCommand: (...args: unknown[]) => runShellCommand(...args), + readDeployConfig: (...args: unknown[]) => readDeployConfig(...args), })); vi.mock('../lambda/shared/environments', async (importOriginal) => ({ @@ -79,11 +81,23 @@ const daemonReply = (fields: Record) => ({ stdout: JSON.stringify({ state: 'running', ...fields }), }); +/** The stored deploy config the status branch reads its model facts from. */ +const deployConfig = (fields: Record = {}) => + readDeployConfig.mockResolvedValue({ + runner: 'llamacpp', + modelId: 'org/Qwen3.8-27B', + servedModelName: 'qwen3.8-27b', + ...fields, + }); + beforeEach(() => { vi.clearAllMocks(); findEnvEip.mockResolvedValue({ publicIp: '198.51.100.7' }); findManagedInstance.mockResolvedValue({ instanceId: 'i-abc', state: 'running' }); isSsmAgentOnline.mockResolvedValue(true); + // A deploy config is present for a running environment by default; tests + // that care about the model facts override it, the rest ignore it. + deployConfig(); }); /** Route each SSM invocation by the command it was given. */ @@ -105,6 +119,65 @@ describe('remote status activity reporting', () => { expect(body.idleSeconds).toBe(42); }); + it('reports what the environment is serving alongside its activity', async () => { + // The model facts come from the stored deploy config (the default one + // here); the daemon supplies only the activity. + runShellCommand.mockImplementation( + ssmRouter(daemonReply({ lastActiveAt: '2026-08-09T12:00:00Z', idleSeconds: 42 })), + ); + + const body = bodyOf(await handler(statusEvent, {} as Context)); + expect(body.state).toBe('running'); + expect(body.runner).toBe('llamacpp'); + expect(body.modelId).toBe('org/Qwen3.8-27B'); + expect(body.servedName).toBe('qwen3.8-27b'); + expect(body.lastActiveAt).toBe('2026-08-09T12:00:00Z'); + expect(body.idleSeconds).toBe(42); + }); + + it('reports the model before the engine has done any work', async () => { + // The model facts come from the deploy config, so they are present + // whenever the instance is running — even before the engine has answered a + // request. That is the case a router needs, so the model is not gated on + // activity. + runShellCommand.mockImplementation(ssmRouter(daemonReply({}))); + + const body = bodyOf(await handler(statusEvent, {} as Context)); + expect(body.state).toBe('running'); + expect(body.modelId).toBe('org/Qwen3.8-27B'); + expect(body.servedName).toBe('qwen3.8-27b'); + expect(body).not.toHaveProperty('lastActiveAt'); + expect(body).not.toHaveProperty('idleSeconds'); + }); + + it('omits the served name when the deploy named none', async () => { + // A deploy before the served-name feature stored no servedModelName, so + // the report carries the model id but no served name. + deployConfig({ servedModelName: '' }); + runShellCommand.mockImplementation( + ssmRouter(daemonReply({ lastActiveAt: '2026-08-09T12:00:00Z', idleSeconds: 5 })), + ); + + const body = bodyOf(await handler(statusEvent, {} as Context)); + expect(body.modelId).toBe('org/Qwen3.8-27B'); + expect(body).not.toHaveProperty('servedName'); + }); + + it('omits the model facts when the deploy config cannot be read', async () => { + readDeployConfig.mockRejectedValue(new Error('InvalidParameter')); + runShellCommand.mockImplementation( + ssmRouter(daemonReply({ lastActiveAt: '2026-08-09T12:00:00Z', idleSeconds: 5 })), + ); + + const result = await handler(statusEvent, {} as Context); + expect(structured(result).statusCode).toBe(200); + const body = bodyOf(result); + expect(body.lastActiveAt).toBe('2026-08-09T12:00:00Z'); + expect(body).not.toHaveProperty('runner'); + expect(body).not.toHaveProperty('modelId'); + expect(body).not.toHaveProperty('servedName'); + }); + it('treats an omitted idleSeconds as zero, not as absent', async () => { // The daemon omits idleSeconds rather than sending 0 while the engine is // working right now. The timestamp is the gate, so this must survive. From 8905886208e0e063e1dd4fd126587c20b8cf37af Mon Sep 17 00:00:00 2001 From: Pete Cornish Date: Sun, 13 Sep 2026 00:44:02 +0100 Subject: [PATCH 2/4] perf: serve the daemon's host figures from the sampler --- cmd/spinloop/dashboard_model.go | 8 +- internal/daemon/activity.go | 18 ++- internal/daemon/daemon.go | 12 +- internal/daemon/daemon_test.go | 3 + internal/daemon/history.go | 67 ++++++++++- internal/daemon/history_test.go | 110 +++++++++++++++++- .../.openspec.yaml | 2 + .../design.md | 70 +++++++++++ .../proposal.md | 59 ++++++++++ .../specs/daemon-api/spec.md | 44 +++++++ .../specs/engine-activity/spec.md | 45 +++++++ .../tasks.md | 11 ++ openspec/specs/daemon-api/spec.md | 43 +++++++ openspec/specs/engine-activity/spec.md | 19 ++- 14 files changed, 489 insertions(+), 22 deletions(-) create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/.openspec.yaml create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/design.md create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/proposal.md create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/daemon-api/spec.md create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/engine-activity/spec.md create mode 100644 openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/tasks.md diff --git a/cmd/spinloop/dashboard_model.go b/cmd/spinloop/dashboard_model.go index 613357ac..6c0f4cad 100644 --- a/cmd/spinloop/dashboard_model.go +++ b/cmd/spinloop/dashboard_model.go @@ -18,8 +18,12 @@ import ( ) // dashboardRefreshInterval is how often the board re-reads the local daemon -// machines. It is a variable so a test never waits for a slow node. -var dashboardRefreshInterval = 2 * time.Second +// machines. It is one call per machine per tick, so the cost is the fleet's +// size times this rate; the figures it draws — utilization, token counters, +// idle time — are sampled by each daemon every 15 seconds anyway, so polling +// faster than this re-fetches readings that have not changed. It is a +// variable so a test never waits for a slow node. +var dashboardRefreshInterval = 5 * time.Second // dashboardRemoteRefreshInterval is the cadence for kind: remote // environments instead. Each of their statuses is a signed call through the diff --git a/internal/daemon/activity.go b/internal/daemon/activity.go index ab7d9bfa..90814141 100644 --- a/internal/daemon/activity.go +++ b/internal/daemon/activity.go @@ -101,7 +101,7 @@ func (d *Daemon) SampleActivity(ctx context.Context) { // can answer, instead of up to a full interval later. A tick with no // engine running costs a state check. wait := interval - if !d.sample.haveTokens() { + if d.awaitingFirstSample() { wait = catchUpInterval } select { @@ -118,6 +118,22 @@ func (d *Daemon) SampleActivity(ctx context.Context) { // engine-state check. var catchUpInterval = time.Second +// awaitingFirstSample reports whether the short catch-up interval still +// applies: a scrape target is known and no counters have come back from it +// yet. The target check is what bounds the catch-up. Without it, an engine +// whose runner exposes no metrics endpoint never yields counters, so the +// sampler would stay at the catch-up interval for the engine's whole life — +// running the host commands every second for a reading that is not coming. +func (d *Daemon) awaitingFirstSample() bool { + if d.sample.haveTokens() { + return false + } + d.mu.Lock() + scrape := d.scrape + d.mu.Unlock() + return scrape.BaseURL != "" && scrape.Engine != "" +} + // sampleOnce takes one reading, feeding both a success and a failure through // observe so there is exactly one place where a sample becomes activity. The // reading is also kept, because /v1/metrics reports it rather than scraping diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 69aeaad4..bf5ccfb1 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -66,6 +66,7 @@ type Daemon struct { sample engineSample ready readiness hist systemHistory + system systemSample mu sync.Mutex runner string @@ -272,6 +273,7 @@ func (d *Daemon) StartEngine() error { // against this one, for the same reason its counter baseline is dropped. d.sample.forget() d.ready.forget() + d.system.forget() d.hist.clear() return nil } @@ -432,9 +434,13 @@ func (d *Daemon) Metrics(ctx context.Context) metrics.Stats { UptimeSeconds: uptime, } if state == StateRunning { - if d.Collector != nil { - d.Collector.System(ctx, &stats) - } + // The host's figures come from the background sampler, never from a + // collection taken here. Reading CPU costs a host command — `top -l 1` + // on macOS — that takes longer the busier the host is, so collecting + // inline made this handler block for seconds on exactly the machine + // someone had opened the dashboard to watch, past the fleet client's + // timeout, and the node rendered as unreachable. + d.system.apply(&stats) // The engine's counters come from the background sampler, never from // a scrape taken here. A busy engine does not answer its own metrics // endpoint — llama.cpp serves it from the queue it serves inference diff --git a/internal/daemon/daemon_test.go b/internal/daemon/daemon_test.go index 826c6183..e2bdf0d5 100644 --- a/internal/daemon/daemon_test.go +++ b/internal/daemon/daemon_test.go @@ -379,6 +379,9 @@ while true; do sleep 0.05; done`) } // Metrics while running: state, runner, and the collector's memory stat. + // The host figures reach the reply through the background sampler, which + // this test drives one tick of by hand rather than running the loop. + d.systemSampleOnce(context.Background()) if resp, body := do("GET", "/v1/metrics", "sekrit", ""); resp.StatusCode != 200 || body["state"] != "running" || body["runner"] != "llamacpp" || body["memory"] == nil { t.Fatalf("metrics = %d %v", resp.StatusCode, body) diff --git a/internal/daemon/history.go b/internal/daemon/history.go index 1bd237b0..ee484cd1 100644 --- a/internal/daemon/history.go +++ b/internal/daemon/history.go @@ -77,10 +77,14 @@ func (h *systemHistory) clear() { // come from host commands, not from the engine — so an engine with no metrics // endpoint still yields a window to draw. It runs only while an engine is // running: a stopped engine has no utilization to chart, and the readings -// taken before the stop remain until the next start. A reading that yields no -// figure at all records nothing and reports nothing: a failed sample is a -// non-observation here as in the activity record, and the on-request -// collection keeps its own error reporting. +// taken before the stop remain until the next start. +// +// One collection serves both readers: the full figures go to systemSample, +// which is what /v1/metrics reports, and the reduced percentages go to the +// history. A reading that yields no figure at all adds nothing to the +// history — a failed sample is a non-observation there as in the activity +// record — but is still recorded as the current reading, because its errors +// are how a broken source gets reported at all. func (d *Daemon) systemSampleOnce(ctx context.Context) { if state, _, _ := d.Sup.Status(); state != StateRunning { return @@ -90,6 +94,7 @@ func (d *Daemon) systemSampleOnce(ctx context.Context) { } var stats metrics.Stats d.Collector.System(ctx, &stats) + d.system.record(stats) sample := metrics.HistorySample{Time: d.now().Unix()} if stats.CPU != nil { v := stats.CPU.Utilization @@ -112,3 +117,57 @@ func (d *Daemon) systemSampleOnce(ctx context.Context) { } d.hist.add(sample) } + +// systemSample holds the most recent reading of the host's figures, taken by +// the background sampler. /v1/metrics reports this rather than collecting on +// demand, for the same reason engineSample exists: collecting costs host +// commands, and one of them is slow in proportion to how busy the host is. +// macOS reads CPU with `top -l 1`, which on a machine loaded enough to be +// worth watching takes several seconds — so a handler that collected inline +// blocked past the fleet client's timeout exactly when someone was looking at +// the dashboard, and the node rendered as unreachable while it was answering +// fine. The sampler also removes a second cost: the figures were collected +// once for the history and again for every request, so a polled daemon ran +// the host commands far more often than the readings changed. +// +// The cost is staleness bounded by the sample interval, which for utilisation +// figures in a refreshing view is not a cost at all. +type systemSample struct { + mu sync.Mutex + gpus []metrics.GpuStat + cpu *metrics.CpuStat + memory *metrics.MemoryStat + // errs are the collection failures from that reading, kept because they + // are now the only place a broken source is reported: the handler no + // longer collects, so it has no errors of its own to add. + errs []string + have bool +} + +// record stores one reading, including its failures. +func (s *systemSample) record(stats metrics.Stats) { + s.mu.Lock() + defer s.mu.Unlock() + s.gpus, s.cpu, s.memory, s.errs, s.have = stats.GPUs, stats.CPU, stats.Memory, stats.Errors, true +} + +// apply copies the last reading onto stats. Before the first sample lands it +// copies nothing, so an unsampled figure stays absent rather than reading as a +// host with no CPU. +func (s *systemSample) apply(stats *metrics.Stats) { + s.mu.Lock() + defer s.mu.Unlock() + if !s.have { + return + } + stats.GPUs, stats.CPU, stats.Memory = s.gpus, s.cpu, s.memory + stats.Errors = append(stats.Errors, s.errs...) +} + +// forget drops the reading, so a stopped engine's host figures are not +// reported against the next one. +func (s *systemSample) forget() { + s.mu.Lock() + defer s.mu.Unlock() + s.gpus, s.cpu, s.memory, s.errs, s.have = nil, nil, nil, nil, false +} diff --git a/internal/daemon/history_test.go b/internal/daemon/history_test.go index b13b6d02..159cbee9 100644 --- a/internal/daemon/history_test.go +++ b/internal/daemon/history_test.go @@ -10,6 +10,7 @@ import ( "math" "os/exec" "strings" + "sync/atomic" "testing" "time" @@ -310,8 +311,10 @@ while true; do sleep 0.05; done`) d.Now = clock.now d.Collector = linuxCollector() // No scrape target is set: the system readings must not depend on one. - old := catchUpInterval - catchUpInterval = 5 * time.Millisecond + // The cadence comes from the tick interval rather than the catch-up, + // which without a scrape target does not apply — there are no counters + // coming, so there is nothing to catch up to. + d.SampleInterval = 5 * time.Millisecond ctx, cancel := context.WithCancel(context.Background()) done := make(chan struct{}) @@ -319,13 +322,9 @@ while true; do sleep 0.05; done`) d.SampleActivity(ctx) close(done) }() - // The loop reads the catch-up cadence on every tick while it has no - // reading to report — with no scrape target, always — so the sampler - // must have exited before the cadence is restored. defer func() { cancel() <-done - catchUpInterval = old }() if err := d.Push(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}); err != nil { @@ -359,3 +358,102 @@ while true; do sleep 0.05; done`) t.Errorf("a stopped engine kept accumulating readings: %d -> %d", n, got) } } + +// The whole point of the sampler: a metrics request runs no host command, so +// a handler cannot be held up by a slow one. `top -l 1` on a loaded macOS host +// takes seconds, which is what made a polled node render as unreachable while +// it was answering fine. +func TestMetricsRunsNoHostCommands(t *testing.T) { + d := testDaemon(t, `trap 'exit 0' TERM +while true; do sleep 0.05; done`) + var during int32 + base := linuxCollector() + inner := base.Run + base.Run = func(ctx context.Context, name string, args ...string) (string, error) { + atomic.AddInt32(&during, 1) + return inner(ctx, name, args...) + } + d.Collector = base + if err := d.Push(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}); err != nil { + t.Fatal(err) + } + if err := d.StartEngine(); err != nil { + t.Fatal(err) + } + defer d.Sup.Stop() + waitForState(t, d.Sup, StateRunning) + + d.systemSampleOnce(context.Background()) + sampled := atomic.LoadInt32(&during) + if sampled == 0 { + t.Fatal("the sampler ran no host commands; it is what collects them") + } + for i := 0; i < 5; i++ { + if stats := d.Metrics(context.Background()); stats.Memory == nil { + t.Fatalf("request %d reported no memory figure from the last sample", i) + } + } + if got := atomic.LoadInt32(&during); got != sampled { + t.Errorf("metrics requests ran %d host commands, want none", got-sampled) + } +} + +// A collection failure is reported to the caller. The handler no longer +// collects, so the sample is the only thing that can carry the error — losing +// it would turn a broken source into figures that are silently absent. +func TestSampledCollectionErrorsReachMetrics(t *testing.T) { + d := testDaemon(t, `trap 'exit 0' TERM +while true; do sleep 0.05; done`) + d.Collector = &metrics.Collector{ + GOOS: "linux", + Run: func(ctx context.Context, name string, args ...string) (string, error) { + return "", errors.New("vmstat exploded") + }, + } + if err := d.Push(remote.DeployConfig{Runner: "llamacpp", ModelID: "m"}); err != nil { + t.Fatal(err) + } + if err := d.StartEngine(); err != nil { + t.Fatal(err) + } + defer d.Sup.Stop() + waitForState(t, d.Sup, StateRunning) + + d.systemSampleOnce(context.Background()) + stats := d.Metrics(context.Background()) + var found bool + for _, e := range stats.Errors { + if strings.Contains(e, "vmstat exploded") { + found = true + } + } + if !found { + t.Errorf("the collection failure did not reach the caller: %v", stats.Errors) + } + // Reported once, from the one collection, not once per reader. + if n := len(stats.Errors); n != len(d.Metrics(context.Background()).Errors) { + t.Errorf("errors accumulate across requests: %d", n) + } +} + +// The catch-up interval is for a reading that is actually coming. An engine +// whose runner exposes no metrics endpoint yields no counters ever, so the +// loop must settle at its tick rather than spin — it runs the host commands +// on every pass. +func TestNoScrapeTargetLeavesTheCatchUpInterval(t *testing.T) { + d := testDaemon(t, `trap 'exit 0' TERM +while true; do sleep 0.05; done`) + if d.awaitingFirstSample() { + t.Error("no scrape target: the catch-up interval should not apply") + } + // An address with no dialect is the readiness-probe-only case: still + // nothing to scrape, so still no catching up to do. + d.SetScrape(metrics.ScrapeTarget{BaseURL: "http://127.0.0.1:8080"}) + if d.awaitingFirstSample() { + t.Error("an address with no metrics dialect should not hold the catch-up") + } + d.SetScrape(metrics.ScrapeTarget{BaseURL: "http://127.0.0.1:8080", Engine: "llamacpp"}) + if !d.awaitingFirstSample() { + t.Error("a scrape target with no reading yet should hold the catch-up") + } +} diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/.openspec.yaml b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/.openspec.yaml new file mode 100644 index 00000000..7a8e2be6 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-08 diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/design.md b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/design.md new file mode 100644 index 00000000..2e8e73bc --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/design.md @@ -0,0 +1,70 @@ +## Context + +The daemon's sampler already takes one system reading per tick for the retained +history, and the metrics handler collected the same host figures again on every +request. See proposal.md for why the second collection had to go. + +## Goals / Non-Goals + +**Goals:** + +- One collection per tick serves both the retained history and the reported + reading. +- A broken host source stays visible: its failure is reported somewhere a + caller sees. + +**Non-Goals:** + +- The engine's token counters: they already come from the sampler's reading, + and this change does not touch how they are reported. +- The sampler's interval: it stays at its own interval; only what a reading + serves changes. + +## Decisions + +### The handler copies, never collects + +The daemon keeps the most recent system reading: the figures, the collection's +failures, and whether a reading has landed. The sampler records each reading +there; the metrics handler copies it onto the response. Before the first +reading the handler copies nothing, so an unsampled figure stays absent rather +than reading as a host with no CPU. A start after a stop drops the reading, so +the new engine does not report the last engine's host. + +The cost is staleness bounded by the sampling interval. For utilisation figures +in a refreshing view, a reading at most one tick old is not a cost: the view is +already a sample of a moving host, and the alternative — collecting inline — +made the endpoint slow in proportion to how busy the host was, which is +backwards for a watched machine. + +### The reading carries the collection's failures + +With no collection taken on the request, the handler has no errors of its own. +A failed reading is recorded as the current reading with its failures, and the +handler reports them among the response's errors. A failed reading still +contributes no figures to the retained history: there, a failure is a +non-observation, as in the activity record. + +### The catch-up interval bounds itself with a scrape target + +The short interval that runs until the first engine sample exists applied +whenever no counters had come back. For an engine whose runner exposes no +metrics endpoint, counters never come back, so the sampler would have run the +host commands every second for the engine's whole life. The interval now +applies only while a scrape target is known: no target, the sampler settles at +the tick. + +### The dashboard's cadence matches the data's + +The board re-reads the local daemons every 5 seconds rather than 2: it is one +call per machine per tick, and the figures it draws are sampled every 15 +seconds, so polling faster re-fetches readings that have not changed. + +## Risks / Trade-offs + +- A caller who polls metrics faster than the tick sees figures that do not move + between ticks. That is the sampler's contract already, and the dashboard is + the only in-repo caller that relies on these figures refreshing. +- A host source that breaks between ticks is visible once a tick's reading + fails — at most one tick later than an inline collection would have shown + it. diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/proposal.md b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/proposal.md new file mode 100644 index 00000000..dcaa4152 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/proposal.md @@ -0,0 +1,59 @@ +## Why + +The daemon's metrics endpoint collected the host's figures inline: reading CPU +costs a host command that takes longer the busier the host is, so on exactly the +machine someone has opened the dashboard to watch, the handler blocked past the +fleet client's timeout and the node rendered as unreachable while it was +answering fine. The figures were also collected twice — once by the background +sampler for the retained history, and again on every request. + +## What Changes + +- `/v1/metrics` reports the background sampler's last system reading — the + GPU, CPU and memory figures — instead of taking a fresh collection on the + request: one collection per tick serves both the retained history and the + reported reading, and a reported figure is at most one sampling interval + stale. +- A system reading that fails still becomes the reported reading, carrying its + errors: with no collection taken on the request, the reading is the only + place a broken source is reported from. The retained history still gets no + figures for that tick. +- Before the first reading lands, the host figures are absent rather than a + fresh collection or zero, and a start after a stop drops the previous + engine's reading. +- The sampler's short catch-up interval now also requires a known scrape + target: an engine whose runner exposes no metrics endpoint settles at the + tick instead of running the host commands every second for counters that are + never coming. +- The fleet dashboard polls the local daemons every 5 seconds rather than 2: + one call per machine per tick, and the figures it draws are sampled every 15 + seconds anyway. + +## Capabilities + +### New Capabilities + +(none) + +### Modified Capabilities + +- `daemon-api`: the metrics endpoint's host figures are the sampler's last + reading, not a collection taken on the request; before the first reading they + are absent, a start after a stop drops the previous reading, and a reading's + collection failures are reported among the response's errors. +- `engine-activity`: one system reading per tick serves both the retained + history and the reported reading; a failed reading gets no figures in the + history but is still the reported reading, carrying its errors. + +## Impact + +- `internal/daemon`: the daemon keeps the most recent system reading (figures, + the collection's failures, whether a reading has landed), recorded by the + sampler and dropped on a start after a stop; the metrics handler copies the + last reading instead of calling the collector; the catch-up interval applies + only while a scrape target is known. +- `cmd/spinloop`: the dashboard's local refresh interval moves from 2 seconds + to 5 seconds. +- No API-shape change: the same fields carry the figures, and the response's + errors already existed, so `docs/openapi.yaml` is untouched. No `remote/` + change, no new dependencies. diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/daemon-api/spec.md b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/daemon-api/spec.md new file mode 100644 index 00000000..bc743456 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/daemon-api/spec.md @@ -0,0 +1,44 @@ +## ADDED Requirements + +### Requirement: Metrics reports the host's last sampled reading + +The metrics endpoint's host figures — the GPU, CPU and memory readings — SHALL +be the background sampler's last reading, not a collection taken on the +request: one collection per tick serves the retained history and the reported +reading together, and reading the host's CPU costs a command that takes longer +the busier the host is, so a request that collected inline would block on +exactly the machine a caller is watching. A reported figure SHALL be at most +one sampling interval stale. + +Before the first reading has landed, the host figures SHALL be absent rather +than zero or a fresh collection. A start after a stop SHALL drop the previous +engine's reading, so a new engine does not report the host's figures as they +stood for the last one. A reading's collection failures SHALL be reported among +the response's errors, naming the source: with no collection taken on the +request, the reading is the only place a broken source is reported from. + +#### Scenario: A running engine's metrics report the sampled reading + +- **WHEN** an engine has been running long enough for a sampling tick, and a + metrics request is made +- **THEN** the response's host figures are the sampler's last reading, at most + one sampling interval old + +#### Scenario: No reading yet, no figures + +- **WHEN** an engine has just started and no sampling tick has landed a + reading, and a metrics request is made +- **THEN** the response carries no host figures + +#### Scenario: A failed reading reports its errors + +- **WHEN** the sampler's last system reading failed, and a metrics request is + made +- **THEN** the response reports the failure among its errors, naming the + source, and carries no figures the reading did not yield + +#### Scenario: A start after a stop drops the previous reading + +- **WHEN** an engine is stopped and a new engine is started, and a metrics + request is made before the new engine's first tick +- **THEN** the response carries no host figures from the previous engine diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/engine-activity/spec.md b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/engine-activity/spec.md new file mode 100644 index 00000000..3e42a8d5 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/specs/engine-activity/spec.md @@ -0,0 +1,45 @@ +## MODIFIED Requirements + +### Requirement: System readings for the retained history + +While an engine is running, the sampler SHALL take one system reading — the +host's CPU, memory, and GPU figures — on each tick at its own interval, +independently of any request to the control API and independently of whether a +scrape target for the engine's counters is known. Each reading SHALL be +retained in the history the metrics endpoint reports, for at most the last +10 minutes. + +One reading SHALL serve both what is retained and what is reported: the same +collection feeds the retained history and the current reading the metrics +endpoint answers with, so a request never takes a second collection. A reading +that yields no figure at all SHALL contribute no sample to the retained +history — a failed sample is a non-observation there as in the activity +record — and SHALL still become the current reading, carrying its errors: with +no collection taken on the request, the reading is the only place a broken +source is reported from. + +#### Scenario: System readings happen without being asked + +- **WHEN** an engine is running and no client calls the control API +- **THEN** the daemon still takes a system reading on each sampler tick and + retains it + +#### Scenario: System readings do not depend on a scrape target + +- **WHEN** the running engine exposes no metrics endpoint to scrape +- **THEN** the system readings are still taken and retained, since they come + from the host, not from the engine + +#### Scenario: A failed system reading records nothing + +- **WHEN** a system reading fails on a tick because a host command is missing + or fails +- **THEN** no sample is recorded for that tick in the retained history, and + the reading the metrics endpoint reports is the failed one, its errors + included + +#### Scenario: Reading stops with the engine, retention does not end + +- **WHEN** the engine is stopped +- **THEN** no further system readings are taken, and the readings taken before + the stop remain retained until the next engine starts diff --git a/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/tasks.md b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/tasks.md new file mode 100644 index 00000000..4d3efab2 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-serve-host-figures-from-sampler/tasks.md @@ -0,0 +1,11 @@ +# Tasks + +## 1. The sampler's reading serves the metrics endpoint + +- [x] 1.1 Record each system reading on the daemon as the current reading — the figures, the collection's failures, and whether a reading has landed — from the sampler's per-tick collection, drop it on a start after a stop, and copy it (not a fresh collection) onto the metrics response, so an unsampled figure stays absent and the handler runs no host commands, verified by `TestSystemSampleOnce` (a reading is recorded and applied), `TestSystemSampleOnceRecordsNothingOnFailure` (a failed reading is still recorded, with its failures), `TestSystemHistoryClearsOnStartAndSurvivesAStop` (the reading is dropped on a start after a stop), and `TestMetricsRunsNoHostCommands` (the handler takes no collection) +- [x] 1.2 Report a reading's collection failures among the metrics response's errors, naming the source, so a broken host source is reported from the reading now that the handler no longer collects, verified by `TestSampledCollectionErrorsReachMetrics` +- [x] 1.3 Bound the short catch-up interval with a known scrape target — it applies only while a target is known and no counters have come back, so an engine whose runner exposes no metrics endpoint settles at the tick instead of running the host commands every second — verified by `TestNoScrapeTargetLeavesTheCatchUpInterval` + +## 2. The dashboard's cadence + +- [x] 2.1 Move the dashboard's local refresh interval from 2 seconds to 5 — one call per machine per tick, the figures it draws are sampled every 15 seconds anyway — verified by the dashboard refresh tests passing against the new constant diff --git a/openspec/specs/daemon-api/spec.md b/openspec/specs/daemon-api/spec.md index f367068b..f1c12450 100644 --- a/openspec/specs/daemon-api/spec.md +++ b/openspec/specs/daemon-api/spec.md @@ -510,3 +510,46 @@ whatever the served name is. - **WHEN** an engine was started from a config that names no served name, and a status request is made - **THEN** the response reports the model id and carries no served name + +### Requirement: Metrics reports the host's last sampled reading + +The metrics endpoint's host figures — the GPU, CPU and memory readings — SHALL +be the background sampler's last reading, not a collection taken on the +request: one collection per tick serves the retained history and the reported +reading together, and reading the host's CPU costs a command that takes longer +the busier the host is, so a request that collected inline would block on +exactly the machine a caller is watching. A reported figure SHALL be at most +one sampling interval stale. + +Before the first reading has landed, the host figures SHALL be absent rather +than zero or a fresh collection. A start after a stop SHALL drop the previous +engine's reading, so a new engine does not report the host's figures as they +stood for the last one. A reading's collection failures SHALL be reported among +the response's errors, naming the source: with no collection taken on the +request, the reading is the only place a broken source is reported from. + +#### Scenario: A running engine's metrics report the sampled reading + +- **WHEN** an engine has been running long enough for a sampling tick, and a + metrics request is made +- **THEN** the response's host figures are the sampler's last reading, at most + one sampling interval old + +#### Scenario: No reading yet, no figures + +- **WHEN** an engine has just started and no sampling tick has landed a + reading, and a metrics request is made +- **THEN** the response carries no host figures + +#### Scenario: A failed reading reports its errors + +- **WHEN** the sampler's last system reading failed, and a metrics request is + made +- **THEN** the response reports the failure among its errors, naming the + source, and carries no figures the reading did not yield + +#### Scenario: A start after a stop drops the previous reading + +- **WHEN** an engine is stopped and a new engine is started, and a metrics + request is made before the new engine's first tick +- **THEN** the response carries no host figures from the previous engine diff --git a/openspec/specs/engine-activity/spec.md b/openspec/specs/engine-activity/spec.md index 291f09e1..adc8cd00 100644 --- a/openspec/specs/engine-activity/spec.md +++ b/openspec/specs/engine-activity/spec.md @@ -126,10 +126,16 @@ host's CPU, memory, and GPU figures — on each tick at its own interval, independently of any request to the control API and independently of whether a scrape target for the engine's counters is known. Each reading SHALL be retained in the history the metrics endpoint reports, for at most the last -10 minutes. A failed system reading SHALL record no sample for its tick and -SHALL NOT be reported as an error: the on-request collection keeps its own -error reporting, and a transient sampling failure is neither data nor a -condition worth surfacing on every tick. +10 minutes. + +One reading SHALL serve both what is retained and what is reported: the same +collection feeds the retained history and the current reading the metrics +endpoint answers with, so a request never takes a second collection. A reading +that yields no figure at all SHALL contribute no sample to the retained +history — a failed sample is a non-observation there as in the activity +record — and SHALL still become the current reading, carrying its errors: with +no collection taken on the request, the reading is the only place a broken +source is reported from. #### Scenario: System readings happen without being asked @@ -147,8 +153,9 @@ condition worth surfacing on every tick. - **WHEN** a system reading fails on a tick because a host command is missing or fails -- **THEN** no sample is recorded for that tick and no error is reported for - it +- **THEN** no sample is recorded for that tick in the retained history, and + the reading the metrics endpoint reports is the failed one, its errors + included #### Scenario: Reading stops with the engine, retention does not end From d172a52007dfac4872bed849377adcdbf5363931 Mon Sep 17 00:00:00 2001 From: Pete Cornish Date: Sun, 13 Sep 2026 00:45:04 +0100 Subject: [PATCH 3/4] feat: add the fleet harness command --- cmd/spinloop/commands.go | 127 ++++--- cmd/spinloop/fleet.go | 111 ++++++ cmd/spinloop/fleet_harness_test.go | 354 ++++++++++++++++++ cmd/spinloop/harness_test.go | 64 ++++ cmd/spinloop/main.go | 56 ++- cmd/spinloop/route.go | 17 +- cmd/spinloop/route_test.go | 239 ++++++++++++ docs/README.md | 2 +- docs/commands/fleet.md | 58 +++ docs/commands/gateway.md | 17 + docs/commands/harness.md | 7 + examples/gateway-docker/README.md | 19 +- examples/gateway-docker/client/Spinloop | 14 +- examples/gateway-docker/fleet.yaml | 13 +- examples/gateway-docker/run-tests.sh | 24 +- internal/fleet/config.go | 52 +++ internal/fleet/config_test.go | 84 +++++ internal/fleet/select.go | 8 +- .../.openspec.yaml | 2 + .../2026-09-08-add-fleet-harness/design.md | 58 +++ .../2026-09-08-add-fleet-harness/proposal.md | 63 ++++ .../specs/fleet-client/spec.md | 54 +++ .../specs/fleet-config/spec.md | 34 ++ .../specs/fleet-routing/spec.md | 55 +++ .../2026-09-08-add-fleet-harness/tasks.md | 23 ++ openspec/specs/fleet-client/spec.md | 55 +++ openspec/specs/fleet-config/spec.md | 33 ++ openspec/specs/fleet-routing/spec.md | 54 +++ 28 files changed, 1591 insertions(+), 106 deletions(-) create mode 100644 cmd/spinloop/fleet_harness_test.go create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/.openspec.yaml create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/design.md create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/proposal.md create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-client/spec.md create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-config/spec.md create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-routing/spec.md create mode 100644 openspec/changes/archive/2026-09-08-add-fleet-harness/tasks.md diff --git a/cmd/spinloop/commands.go b/cmd/spinloop/commands.go index 5ee4ff7c..3b5e4954 100644 --- a/cmd/spinloop/commands.go +++ b/cmd/spinloop/commands.go @@ -197,65 +197,7 @@ exits. Honours -H/--harness and SPINLOOP_HARNESS.`, } else if route.fleetPath != "" { return fmt.Errorf("--fleet needs a Spinloop: it is the Spinloop's model that decides which node can serve you") } - // The resolver the launch uses knows the remote key too, so every - // key the agent is given comes from the same place the apply step - // reported. - resolveKey := remoteLaunchResolver(opencode.EnvResolver(envDir), remoteResp) - - // Launch the harness, forwarding stdio and any trailing args. - bin := h.Command() - cmd := exec.Command(bin, rest...) - cmd.Stdin = os.Stdin - cmd.Stdout = os.Stdout - cmd.Stderr = os.Stderr - cmd.Env = harnessEnv(providers, resolveKey, remoteResp) - // A routed launch points the agent at the node that was chosen. As - // on the remote path, an explicit setting in the environment already - // won: routing fills what is unset rather than overriding a - // deliberate choice. - if choice != nil { - cmd.Env = setEnvIfBlank(cmd.Env, "OPENAI_BASE_URL", choice.BaseURL) - if choice.APIKey != "" { - cmd.Env = setEnvIfBlank(cmd.Env, "OPENAI_API_KEY", choice.APIKey) - } - } - // A worn Spinloop brings its whole local environment to the launched - // agent: its adjacent .env fills any gaps left above, and its ENV - // instructions override everything. These shape only the child's - // environment — spinloop never mutates its own — and follow the same - // precedence the remote commands use: ENV > process environment > - // .env. - if spinloopPath.set { - cmd.Env = overlayLocalEnv(cmd.Env, sel, envDir) - } - // lucinate reads an OpenAI-compatible key from - // LUCINATE_OPENAI_API_KEY when its stored secret is empty — which is - // exactly how spinloop configures it, with no secret on disk. Supply - // the active provider's key here so the launched agent can - // authenticate the model it boots into, without ever writing it to - // lucinate's config. An explicit setting already in the child's env - // wins. - if h.Name() == "lucinate" { - if choice != nil && choice.APIKey != "" { - cmd.Env = setEnvIfBlank(cmd.Env, "LUCINATE_OPENAI_API_KEY", choice.APIKey) - } - if key, ok := lucinateLaunchKey(providers, resolveKey, sel, spinloopPath.set); ok { - cmd.Env = setEnvIfAbsent(cmd.Env, "LUCINATE_OPENAI_API_KEY", key) - } - } - if err := cmd.Run(); err != nil { - if errors.Is(err, exec.ErrNotFound) || errors.Is(err, os.ErrNotExist) { - return fmt.Errorf("%s not found — install the %s harness or add it to your PATH", bin, h.Name()) - } - var exitErr *exec.ExitError - if errors.As(err, &exitErr) { - // The harness ran and chose its own exit code; surface it - // verbatim. - os.Exit(exitErr.ExitCode()) - } - return err - } - return nil + return launchAgent(h, rest, providers, envDir, remoteResp, sel, spinloopPath.set, choice) }, } fs := c.Flags() @@ -276,6 +218,72 @@ exits. Honours -H/--harness and SPINLOOP_HARNESS.`, return c } +// launchAgent runs the harness as the launch's child: stdio and any trailing +// args forwarded, and the environment the apply step reported as the source of +// every key the agent is given. Both launch commands end here, so the agent a +// launch is given can only differ the way the apply that preceded it did. +func launchAgent(h harness.Harness, rest []string, providers, envDir string, remoteResp *remote.Response, sel spinloop.Selection, worn bool, choice *fleet.Choice) error { + // The resolver the launch uses knows the remote key too, so every + // key the agent is given comes from the same place the apply step + // reported. + resolveKey := remoteLaunchResolver(opencode.EnvResolver(envDir), remoteResp) + + // Launch the harness, forwarding stdio and any trailing args. + bin := h.Command() + cmd := exec.Command(bin, rest...) + cmd.Stdin = os.Stdin + cmd.Stdout = os.Stdout + cmd.Stderr = os.Stderr + cmd.Env = harnessEnv(providers, resolveKey, remoteResp) + // A routed launch points the agent at the node that was chosen. As + // on the remote path, an explicit setting in the environment already + // won: routing fills what is unset rather than overriding a + // deliberate choice. + if choice != nil { + cmd.Env = setEnvIfBlank(cmd.Env, "OPENAI_BASE_URL", choice.BaseURL) + if choice.APIKey != "" { + cmd.Env = setEnvIfBlank(cmd.Env, "OPENAI_API_KEY", choice.APIKey) + } + } + // A worn Spinloop brings its whole local environment to the launched + // agent: its adjacent .env fills any gaps left above, and its ENV + // instructions override everything. These shape only the child's + // environment — spinloop never mutates its own — and follow the same + // precedence the remote commands use: ENV > process environment > + // .env. + if worn { + cmd.Env = overlayLocalEnv(cmd.Env, sel, envDir) + } + // lucinate reads an OpenAI-compatible key from + // LUCINATE_OPENAI_API_KEY when its stored secret is empty — which is + // exactly how spinloop configures it, with no secret on disk. Supply + // the active provider's key here so the launched agent can + // authenticate the model it boots into, without ever writing it to + // lucinate's config. An explicit setting already in the child's env + // wins. + if h.Name() == "lucinate" { + if choice != nil && choice.APIKey != "" { + cmd.Env = setEnvIfBlank(cmd.Env, "LUCINATE_OPENAI_API_KEY", choice.APIKey) + } + if key, ok := lucinateLaunchKey(providers, resolveKey, sel, worn); ok { + cmd.Env = setEnvIfAbsent(cmd.Env, "LUCINATE_OPENAI_API_KEY", key) + } + } + if err := cmd.Run(); err != nil { + if errors.Is(err, exec.ErrNotFound) || errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("%s not found — install the %s harness or add it to your PATH", bin, h.Name()) + } + var exitErr *exec.ExitError + if errors.As(err, &exitErr) { + // The harness ran and chose its own exit code; surface it + // verbatim. + os.Exit(exitErr.ExitCode()) + } + return err + } + return nil +} + // versionCmd prints the version, the same spelling the old dispatch gave // `spinloop version`. func versionCmd() *cobra.Command { @@ -331,6 +339,7 @@ error — only a problem with the fleet file itself fails a command.`, fleetLogsCmd(), fleetDashboardCmd(), fleetRouteCmd(), + fleetHarnessCmd(), fleetStartCmd(), fleetStopCmd(), fleetDeployCmd(), diff --git a/cmd/spinloop/fleet.go b/cmd/spinloop/fleet.go index d437c3e2..64c98b41 100644 --- a/cmd/spinloop/fleet.go +++ b/cmd/spinloop/fleet.go @@ -26,6 +26,8 @@ import ( "github.com/spinloop-ai/spinloop/internal/config" "github.com/spinloop-ai/spinloop/internal/daemon" "github.com/spinloop-ai/spinloop/internal/fleet" + "github.com/spinloop-ai/spinloop/internal/harness" + "github.com/spinloop-ai/spinloop/internal/spinloop" ) // cmdFleet runs the fleet subcommands through the tree — the seam the suite @@ -811,6 +813,13 @@ func runFleetRoute(path, node, prefer string, args []string) error { fmt.Printf("This Spinloop pins BASEURL %s, so a launch would not route at all.\n", sel.BaseURL) return nil } + if gw, ok := cfg.GatewaySection(); ok { + // The file names a gateway: as for an endpoint FLEET, the choosing is + // already done — no node is queried, and nothing is started. + fmt.Printf("The fleet file names a gateway: a launch would point the agent at %s.\n", endpointBaseURL(gw.URL)) + fmt.Println("No node is queried, and nothing is started.") + return nil + } choice, err := cfg.Select(context.Background(), want) if err == nil { @@ -844,3 +853,105 @@ func runFleetRoute(path, node, prefer string, args []string) error { // cmdFleetRoute runs the command through the tree — the seam the suite calls. func cmdFleetRoute(args []string) error { return execCmd(fleetRouteCmd(), args) } + +// fleetHarnessCmd configures the active harness for the fleet file and +// launches it: the fleet-level form of a fleet-routed launch, in which the +// fleet file comes from the command rather than from a Spinloop's FLEET. +func fleetHarnessCmd() *cobra.Command { + var spinloopPath spinloopPathFlag + var fleetPath, node, prefer, harnessName string + var noWake bool + var wakeTimeout time.Duration + c := &cobra.Command{ + Use: "harness", + Short: "configure the active harness for this fleet and launch it", + Long: `configures the active harness for the fleet in the fleet file and launches it: +the fleet-level form of a fleet-routed launch. The Spinloop is taken the way +spinloop harness takes one — a leading alias or path, -O/--spinloop, or the +Spinloop beside the fleet file — and the fleet file from -f/--fleet, defaulting +to the fleet.yaml beside it. Where no -f is given, a Spinloop's FLEET — a file +or an endpoint — is used, exactly as --fleet overrides an instruction on +spinloop harness; a Spinloop's pinned BASEURL is not routed, as on the launch.`, + Args: cobra.MaximumNArgs(1), + SilenceErrors: true, + SilenceUsage: true, + ValidArgsFunction: aliasSlot, + RunE: func(c *cobra.Command, args []string) error { + resolve(c) + return runFleetHarness(spinloopPath, fleetPath, node, prefer, harnessName, noWake, wakeTimeout, args) + }, + } + fs := c.Flags() + fs.VarP(&spinloopPath, "spinloop", "O", "the Spinloop to route (bare: ./"+spinloop.DefaultFile+")") + // Bare -O arrives as NoOptDefVal; spinloopPathFlag maps it to the empty + // path readSpinloop resolves as SPINLOOP_ALIAS > ./Spinloop. + fs.Lookup("spinloop").NoOptDefVal = "true" + fs.StringVarP(&fleetPath, "fleet", "f", "", fleetFileUsage) + fs.StringVar(&node, "node", "", "route to this node rather than choosing one") + fs.StringVar(&prefer, "prefer", "", "rank nodes by `idle` or `active` (overrides the fleet file)") + fs.BoolVar(&noWake, "no-wake", false, "fail rather than starting an engine on an idle fleet node") + fs.DurationVar(&wakeTimeout, "wake-timeout", 0, "how long to wait for a woken node's engine") + fs.StringVarP(&harnessName, "harness", "H", "", "which harness to launch") + compRegister(c, "fleet", compFiles) + return c +} + +// cmdFleetHarness runs the command through the tree — the seam the suite calls. +func cmdFleetHarness(args []string) error { return execCmd(fleetHarnessCmd(), args) } + +// runFleetHarness is the body of `spinloop fleet harness`: read the Spinloop, +// route the launch at the fleet's gateway or a chosen node, apply, launch. +func runFleetHarness(sp spinloopPathFlag, fleetPath, node, prefer, harnessName string, noWake bool, wakeTimeout time.Duration, args []string) error { + h, _, err := harness.Resolve(harnessName) + if err != nil { + return err + } + + // The Spinloop is what gets routed: its model is what the fleet is chosen + // against. A leading argument names one the way spinloop harness takes one; + // -O does; with neither, the Spinloop beside the fleet file. + var spinloopArg string + given := false + if sp.set && sp.path != "" { + spinloopArg, given = sp.path, true + } else if len(args) > 0 { + if !namesAnSpinloopOrAlias(args[0]) { + return fmt.Errorf("%s does not name a Spinloop: pass one with -O, or put one beside the fleet file", args[0]) + } + spinloopArg, given = args[0], true + } else if sp.set { + // A bare -O: pflag delivers the flag's NoOptDefVal, which stands for + // the default Spinloop, resolved as SPINLOOP_ALIAS > ./Spinloop. + given = true + } + sel, resolvedPath, err := readSpinloop("spinloop fleet harness ", spinloopArg) + if err != nil { + if !given { + // Nothing was given, so the default file was looked for and not + // found: say what a launch cannot do without it. + return fmt.Errorf("a launch needs a Spinloop to know which model to route: %v", err) + } + return err + } + + // The fleet file: -f when given, the Spinloop's FLEET when -f is not, and + // the fleet.yaml beside it when neither is — the launch's own precedence, + // with the default the launch does not have, because routing is what this + // command exists to do. + route := routeOptions{ + fleetPath: fleetPath, + node: node, + prefer: prefer, + noWake: noWake, + wakeTimeout: wakeTimeout, + } + if route.fleetTarget(sel) == "" { + route.fleetPath = fleet.DefaultFile + } + + sel, envDir, remoteResp, choice, err := applyRoutedSpinloop(sel, resolvedPath, "", h, route) + if err != nil { + return err + } + return launchAgent(h, nil, "", envDir, remoteResp, sel, true, choice) +} diff --git a/cmd/spinloop/fleet_harness_test.go b/cmd/spinloop/fleet_harness_test.go new file mode 100644 index 00000000..ff265c58 --- /dev/null +++ b/cmd/spinloop/fleet_harness_test.go @@ -0,0 +1,354 @@ +package main + +import ( + "os" + "path/filepath" + "strconv" + "strings" + "testing" +) + +// fleetHarnessDir writes a fleet file and a Spinloop into a temp dir and moves +// the test into it: the command's defaults — the fleet.yaml beside it, the +// Spinloop beside it — are working-directory stories. +func fleetHarnessDir(t *testing.T, fleetBody, spinloopBody string) { + t.Helper() + dir := t.TempDir() + if fleetBody != "" { + mustWrite(t, filepath.Join(dir, "fleet.yaml"), fleetBody) + } + if spinloopBody != "" { + mustWrite(t, filepath.Join(dir, "Spinloop"), spinloopBody) + } + t.Chdir(dir) +} + +// No flags at all: the fleet file beside the command names a gateway, so the +// agent is pointed at it with the section's token, and no node is contacted. +func TestCmdFleetHarnessUsesTheFleetsGateway(t *testing.T) { + home := isolateConfig(t) + t.Setenv("GATEWAY_TOKEN", "gw-token") + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetHarnessDir(t, + "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\ngateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n", + "PROVIDER llamacpp\nMODEL qwen3-27b\n") + stderr := captureStderr(t, func() { + captureStdout(t, func() { + if err := cmdFleetHarness(nil); err != nil { + t.Fatalf("cmdFleetHarness: %v", err) + } + }) + }) + if !strings.Contains(stderr, "Routing at http://gw.internal:4000/v1 — the fleet file names a gateway") { + t.Errorf("the choice should be reported on stderr before the launch, got:\n%s", stderr) + } + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("the harness was not launched: %v", err) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + out := string(data) + if !strings.Contains(out, "BASE=http://gw.internal:4000/v1") { + t.Errorf("the agent's base URL should be the section's address with the prefix, got:\n%s", out) + } + if !strings.Contains(out, "KEY=gw-token") { + t.Errorf("the agent should carry the token the section's variable holds, got:\n%s", out) + } + config, err := os.ReadFile(filepath.Join(home, ".config", "opencode", "opencode.json")) + if err != nil { + t.Fatalf("the harness config was not written: %v", err) + } + if !strings.Contains(string(config), "http://gw.internal:4000/v1") { + t.Errorf("the applied provider's base URL should be the section's address, got:\n%s", config) + } +} + +// A fleet file naming no gateway routes to a node, as a fleet-routed launch +// would. +func TestCmdFleetHarnessRoutesToANode(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", true, 300) + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetHarnessDir(t, + "nodes:\n"+node.entry("gpu-box"), + "PROVIDER llamacpp\nMODEL qwen3-27b\n") + captureStdout(t, func() { + if err := cmdFleetHarness(nil); err != nil { + t.Fatalf("cmdFleetHarness: %v", err) + } + }) + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + want := "BASE=http://127.0.0.1:" + strconv.Itoa(node.enginePort) + "/v1" + if !strings.Contains(string(data), want) { + t.Errorf("the agent's base URL should be the node's engine, got:\n%s", data) + } +} + +// -f beats the Spinloop's FLEET: the section in the named file is where the +// agent is pointed, not the endpoint the Spinloop names. +func TestCmdFleetHarnessFileFlagBeatsTheSpinloopsFleet(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "gw-token") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetPath := fleetFileIn(t, t.TempDir(), + "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\ngateway:\n url: http://b.internal:4000\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://a.internal:4000") + captureStdout(t, func() { + if err := cmdFleetHarness([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { + t.Fatalf("cmdFleetHarness: %v", err) + } + }) + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(data), "BASE=http://b.internal:4000/v1") { + t.Errorf("the command's file should be the one routed through, got:\n%s", data) + } +} + +// With no -f, the Spinloop's FLEET — an endpoint here — is used, exactly as +// --fleet overrides an instruction on spinloop harness. +func TestCmdFleetHarnessSpinloopFleetIsUsedWithoutAFlag(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "gw-token") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + spinloopDir := routedSpinloop(t, "qwen3-27b", "http://a.internal:4000") + captureStdout(t, func() { + if err := cmdFleetHarness([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + t.Fatalf("cmdFleetHarness: %v", err) + } + }) + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(data), "BASE=http://a.internal:4000/v1") { + t.Errorf("without -f the Spinloop's FLEET should be routed through, got:\n%s", data) + } +} + +// A directory holding no Spinloop, with none given: the command fails saying a +// launch needs a Spinloop to know which model to route, and launches nothing. +func TestCmdFleetHarnessNeedsASpinloop(t *testing.T) { + isolateConfig(t) + t.Setenv("SPINLOOP_ALIAS", "") + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + t.Chdir(t.TempDir()) + captureStdout(t, func() { + err := cmdFleetHarness(nil) + if err == nil { + t.Fatal("a launch with no Spinloop should fail") + } + if !strings.Contains(err.Error(), "a launch needs a Spinloop to know which model to route") { + t.Errorf("the failure should say a launch needs a Spinloop, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the harness was launched with no Spinloop to route") + } +} + +// A leading argument that is not a Spinloop is refused: there is no forwarding +// on this command, so a stray positional names nothing it could be. +func TestCmdFleetHarnessRefusesANonSpinloopArgument(t *testing.T) { + isolateConfig(t) + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + fleetHarnessDir(t, + "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\ngateway:\n url: http://gw.internal:4000\n", + "PROVIDER llamacpp\nMODEL qwen3-27b\n") + captureStdout(t, func() { + err := cmdFleetHarness([]string{"not-a-spinloop"}) + if err == nil { + t.Fatal("a positional that names no Spinloop should fail") + } + if !strings.Contains(err.Error(), "does not name a Spinloop") { + t.Errorf("the failure should say the argument names no Spinloop, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the harness was launched for an argument that names no Spinloop") + } +} + +// The Spinloop is taken the way spinloop harness takes one: -O, in each of its +// spellings — valueless, space, and equals. +func TestCmdFleetHarnessSpinloopFlag(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", true, 300) + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetHarnessDir(t, "nodes:\n"+node.entry("gpu-box"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + forms := []struct { + name string + args []string + }{ + {"valueless", []string{"-O"}}, + {"space", []string{"-O", "Spinloop"}}, + {"equals", []string{"-O=Spinloop"}}, + } + want := "BASE=http://127.0.0.1:" + strconv.Itoa(node.enginePort) + "/v1" + for _, form := range forms { + t.Run(form.name, func(t *testing.T) { + if err := os.Remove(envFile); err != nil && !os.IsNotExist(err) { + t.Fatal(err) + } + captureStdout(t, func() { + if err := cmdFleetHarness(form.args); err != nil { + t.Fatalf("cmdFleetHarness %v: %v", form.args, err) + } + }) + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatalf("the harness was not launched: %v", err) + } + if !strings.Contains(string(data), want) { + t.Errorf("the agent's base URL should be the node's engine, got:\n%s", data) + } + }) + } +} + +// A -O that names a Spinloop there is no read of fails naming the path, and +// launches nothing. +func TestCmdFleetHarnessNamesTheSpinloopItCannotRead(t *testing.T) { + isolateConfig(t) + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + missing := filepath.Join(t.TempDir(), "Spinloop") + fleetHarnessDir(t, "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\n", "") + captureStdout(t, func() { + err := cmdFleetHarness([]string{"-O=" + missing}) + if err == nil { + t.Fatal("a -O naming a missing Spinloop should fail") + } + if !strings.Contains(err.Error(), missing) { + t.Errorf("the failure should name the path, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the harness was launched for a Spinloop that could not be read") + } +} + +// A harness the command cannot resolve fails before anything is routed or +// written. +func TestCmdFleetHarnessRefusesAnUnknownHarness(t *testing.T) { + home := isolateConfig(t) + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + fleetHarnessDir(t, "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\n", + "PROVIDER llamacpp\nMODEL qwen3-27b\n") + captureStdout(t, func() { + err := cmdFleetHarness([]string{"-H", "nosuchharness"}) + if err == nil { + t.Fatal("an unresolvable harness should fail") + } + if !strings.Contains(err.Error(), "nosuchharness") { + t.Errorf("the failure should name the harness, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the harness was launched for a name that resolves to nothing") + } + if _, err := os.Stat(filepath.Join(home, ".config", "opencode", "opencode.json")); !os.IsNotExist(err) { + t.Errorf("a harness config was written for a launch that never routed (stat: %v)", err) + } +} + +// Nothing serving, wake allowed: the command starts the idle node's engine and +// points the agent at it, the way a fleet-routed launch does — the wake +// timeout flag bounds the wait. +func TestCmdFleetHarnessWakesANode(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", false, 0) + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetHarnessDir(t, "nodes:\n"+node.entry("idle-box"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + stderr := captureStderr(t, func() { + captureStdout(t, func() { + if err := cmdFleetHarness([]string{"--wake-timeout", "5s"}); err != nil { + t.Fatalf("cmdFleetHarness: %v", err) + } + }) + }) + if !node.started { + t.Error("the idle node was not started") + } + if !strings.Contains(stderr, "Using idle-box at") { + t.Errorf("the woken node should be named on stderr, got:\n%s", stderr) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatalf("the harness was not launched: %v", err) + } + want := "BASE=http://127.0.0.1:" + strconv.Itoa(node.enginePort) + "/v1" + if !strings.Contains(string(data), want) { + t.Errorf("the agent's base URL should be the woken engine, got:\n%s", data) + } +} + +// --no-wake with nothing serving fails rather than starting an engine, and the +// refusal names the node and the flag to drop. +func TestCmdFleetHarnessNoWakeRefuses(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", false, 0) + isolateConfig(t) + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + fleetHarnessDir(t, "nodes:\n"+node.entry("idle-box"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + captureStdout(t, func() { + err := cmdFleetHarness([]string{"--no-wake"}) + if err == nil { + t.Fatal("--no-wake with nothing serving should fail") + } + for _, want := range []string{"idle-box", "spinloop fleet start", "drop --no-wake"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal should say %q, got:\n%v", want, err) + } + } + }) + if node.started { + t.Error("--no-wake started an engine") + } + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the harness was launched for a route that refused to wake") + } +} diff --git a/cmd/spinloop/harness_test.go b/cmd/spinloop/harness_test.go index b0700771..8bcbc865 100644 --- a/cmd/spinloop/harness_test.go +++ b/cmd/spinloop/harness_test.go @@ -1126,3 +1126,67 @@ func TestHarness_LucinateInjectsKeyAtLaunch(t *testing.T) { t.Errorf("launched agent's env is missing the injected key:\n%s", got) } } + +// A routed launch gives lucinate the fleet's key, not the provider's: the +// endpoint's token is the one the agent will authenticate with. +func TestHarness_LucinateCarriesTheFleetKeyAtLaunch(t *testing.T) { + isolateConfig(t) + t.Setenv("SPINLOOP_HARNESS", "lucinate") + t.Setenv("OPENAI_API_KEY", "gw-token") + t.Setenv("OPENAI_BASE_URL", "") + t.Setenv("DEEPSEEK_API_KEY", "sk-or-v1-test") + + envFile := filepath.Join(t.TempDir(), "env") + dir := t.TempDir() + body := "#!/bin/sh\nenv > " + envFile + "\n" + if err := os.WriteFile(filepath.Join(dir, "lucinate"), []byte(body), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), + "PROVIDER openrouter\nMODEL deepseek/deepseek-v4-flash\nFLEET http://gw.internal:4000\n") + + captureStdout(t, func() { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir}); err != nil { + t.Fatalf("cmdHarness --spinloop: %v", err) + } + }) + + got, err := os.ReadFile(envFile) + if err != nil { + t.Fatalf("lucinate was not launched: %v", err) + } + if !strings.Contains(string(got), "LUCINATE_OPENAI_API_KEY=gw-token") { + t.Errorf("the fleet's key should reach lucinate in preference to the provider's:\n%s", got) + } + if !strings.Contains(string(got), "OPENAI_BASE_URL=http://gw.internal:4000/v1") { + t.Errorf("the endpoint's address should reach lucinate:\n%s", got) + } +} + +// A harness whose binary is not on the PATH fails naming the fix, after the +// apply — the config is written, the agent is not started. +func TestHarness_NamesTheMissingHarness(t *testing.T) { + home := isolateConfig(t) + t.Setenv("PATH", t.TempDir()) + + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), + "PROVIDER openrouter\nMODEL deepseek/deepseek-v4-flash\n") + + captureStdout(t, func() { + err := cmdHarness([]string{"-H", "lucinate", "--spinloop=" + spinloopDir}) + if err == nil { + t.Fatal("a launch whose harness is not installed should fail") + } + if !strings.Contains(err.Error(), "lucinate not found") || + !strings.Contains(err.Error(), "install the lucinate harness") { + t.Errorf("the failure should name the harness and the fix, got:\n%v", err) + } + }) + if _, err := os.Stat(filepath.Join(home, ".lucinate", "connections.json")); err != nil { + t.Errorf("the apply should have written the config before the launch failed (stat: %v)", err) + } +} diff --git a/cmd/spinloop/main.go b/cmd/spinloop/main.go index a03e576c..e9c24e41 100644 --- a/cmd/spinloop/main.go +++ b/cmd/spinloop/main.go @@ -1171,6 +1171,20 @@ func applyBeforeLaunch(f spinloopPathFlag, providers string, h harness.Harness, if err != nil { return spinloop.Selection{}, "", nil, nil, err } + sel, envDir, remoteResp, choice, err := applyRoutedSpinloop(sel, path, providers, h, route) + if err != nil { + return spinloop.Selection{}, "", nil, nil, err + } + return sel, envDir, remoteResp, choice, nil +} + +// applyRoutedSpinloop routes an already-read Spinloop and applies it to the +// harness that is about to be launched: routing first, so a launch that cannot +// find a node leaves the harness config exactly as it was, then the remote +// fetch and the apply themselves. `spinloop harness` reads its Spinloop on the +// way in; `spinloop fleet harness` reads its own, because with none it fails on +// its own terms. Both then run this one path. +func applyRoutedSpinloop(sel spinloop.Selection, path string, providers string, h harness.Harness, route routeOptions) (spinloop.Selection, string, *remote.Response, *fleet.Choice, error) { // As for apply, --providers overrides the catalogue the selection resolves // against (a Spinloop never names one). sel.Providers = providers @@ -1200,16 +1214,22 @@ func applyBeforeLaunch(f spinloopPathFlag, providers string, h harness.Harness, resolve = fleetLaunchResolver(resolve, choice.APIKey) } if choice != nil && choice.Gateway { - // A FLEET naming an endpoint authenticates with a token the client - // holds itself, resolved the way a key is resolved elsewhere: an ENV - // instruction, else the process environment, else the .env beside the - // Spinloop. Set nowhere, the launch cannot authenticate, and an - // endpoint that refuses every request is not one to point an agent at. - key := localKey(sel, localResolve) + // A FLEET naming an endpoint — or a fleet file naming a gateway — + // authenticates with a token the client holds itself, resolved the way + // a key is resolved elsewhere: an ENV instruction, else the process + // environment, else the .env beside the Spinloop. The variable is the + // one the section names, or the endpoint's default where the section + // names none. Set nowhere, the launch cannot authenticate, and a + // gateway that refuses every request is not one to point an agent at. + env := choice.GatewayTokenEnv + if env == "" { + env = remoteAPIKeyEnv + } + key := localKeyUnder(sel, localResolve, env) if key == "" { return spinloop.Selection{}, "", nil, nil, fmt.Errorf( - "no token to reach the FLEET endpoint %s: export %s, or set it in the .env beside %s", - choice.BaseURL, remoteAPIKeyEnv, path) + "no token to reach the gateway %s: export %s, or set it in the .env beside %s", + choice.BaseURL, env, path) } choice.APIKey = key resolve = fleetLaunchResolver(resolve, key) @@ -1279,17 +1299,23 @@ func fetchRemoteEnv(sel spinloop.Selection, spinloopPath string, resolve func(st return nil, nil } -// localKey returns the API key the launch can supply without the control -// plane: an ENV instruction in the Spinloop, which overrides everything at launch -// (see overlayLocalEnv), otherwise whatever the environment or the adjacent -// `.env` holds. -func localKey(sel spinloop.Selection, resolve func(string) string) string { +// localKeyUnder returns the key the launch can supply without a control +// plane, under the variable name: an ENV instruction in the Spinloop that +// names it, which overrides everything at launch (see overlayLocalEnv), +// otherwise whatever the environment or the adjacent `.env` holds. +func localKeyUnder(sel spinloop.Selection, resolve func(string) string, name string) string { for _, e := range sel.Env { - if e.Key == remoteAPIKeyEnv { + if e.Key == name { return e.Value } } - return resolve(remoteAPIKeyEnv) + return resolve(name) +} + +// localKey resolves under the remote API key's variable — the one a REMOTE +// endpoint and a FLEET endpoint authenticate under. +func localKey(sel spinloop.Selection, resolve func(string) string) string { + return localKeyUnder(sel, resolve, remoteAPIKeyEnv) } // remoteLaunchResolver extends an environment-variable lookup with the key diff --git a/cmd/spinloop/route.go b/cmd/spinloop/route.go index 25c5daca..4b4dd48f 100644 --- a/cmd/spinloop/route.go +++ b/cmd/spinloop/route.go @@ -79,6 +79,21 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp if err != nil { return nil, err } + // A fleet file that names a gateway routes the way an endpoint FLEET + // does: the gateway has already done the choosing, so there is no node to + // contact and nothing to wake. The token is not resolved here: the launch + // resolves it through the same chain, under the variable the section + // names. + if gw, ok := cfg.GatewaySection(); ok { + choice := &fleet.Choice{ + Gateway: true, + BaseURL: endpointBaseURL(gw.URL), + GatewayTokenEnv: gw.TokenEnv, + Reason: "the fleet file names a gateway", + } + announceChoice(choice) + return choice, nil + } prefer, err := cfg.Preference(opts.prefer) if err != nil { return nil, err @@ -149,7 +164,7 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp // an unexpected route says so at the time rather than at the first request. func announceChoice(c *fleet.Choice) { if c.Gateway { - fmt.Fprintf(os.Stderr, "Routing at the FLEET endpoint %s\n", c.BaseURL) + fmt.Fprintf(os.Stderr, "Routing at %s — %s\n", c.BaseURL, c.Reason) return } fmt.Fprintf(os.Stderr, "Using %s at %s — %s\n", c.Node.Name, c.BaseURL, c.Reason) diff --git a/cmd/spinloop/route_test.go b/cmd/spinloop/route_test.go index 057ffc28..2310232f 100644 --- a/cmd/spinloop/route_test.go +++ b/cmd/spinloop/route_test.go @@ -378,6 +378,90 @@ func TestPinnedBaseURLBeatsAnEndpointFleet(t *testing.T) { } } +// gatewayFleetFile writes a fleet file naming a gateway whose nodes point at a +// port nothing listens on — so a route that consults them fails loudly rather +// than passing quietly. +func gatewayFleetFile(t *testing.T, section string) string { + t.Helper() + return fleetFileIn(t, t.TempDir(), + "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\n"+section) +} + +// A fleet file naming a gateway routes at it the way an endpoint FLEET does: +// no node is consulted — the dead node below would fail a route that tried. +func TestRouteToFileNamingAGateway(t *testing.T) { + fleetPath := gatewayFleetFile(t, + "gateway:\n url: http://gw.internal:4000\n tokenEnv: GW_TOKEN\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + stderr := captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{node: "nobody", prefer: "sideways", noWake: true}) + if err != nil { + t.Fatalf("a file naming a gateway should not consult any node: %v", err) + } + if !c.Gateway { + t.Fatalf("the choice should mark itself as a gateway, got %+v", c) + } + if c.BaseURL != "http://gw.internal:4000/v1" { + t.Errorf("a section without a path gets the prefix, got %s", c.BaseURL) + } + if c.GatewayTokenEnv != "GW_TOKEN" { + t.Errorf("the choice should carry the section's variable, got %q", c.GatewayTokenEnv) + } + }) + // The choice is reported before anything launches, the way a node choice is. + if !strings.Contains(stderr, "Routing at http://gw.internal:4000/v1 — the fleet file names a gateway") { + t.Errorf("the choice should be reported on stderr, got:\n%s", stderr) + } +} + +// A section url carrying a path is used as given, like an endpoint's. +func TestRouteToFileNamingAGatewayWithAPath(t *testing.T) { + fleetPath := gatewayFleetFile(t, + "gateway:\n url: http://gw.internal:4000/proxy/v1\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{}) + if err != nil { + t.Fatal(err) + } + if !c.Gateway || c.BaseURL != "http://gw.internal:4000/proxy/v1" { + t.Errorf("a section carrying a path is used as given, got %+v", c) + } + }) +} + +// A pinned BASEURL wins over a gateway section, as it wins over an endpoint. +func TestPinnedBaseURLBeatsAGatewaySection(t *testing.T) { + fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), + "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\nFLEET "+fleetPath+"\n") + sel, path, err := readSpinloop("test", spinloopDir) + if err != nil { + t.Fatal(err) + } + stderr := captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{}) + if err != nil { + t.Fatal(err) + } + if c != nil { + t.Errorf("a pinned BASEURL is not routed, got %+v", c) + } + }) + if !strings.Contains(stderr, "Not routing") { + t.Errorf("spinloop should say it is not routing, got:\n%s", stderr) + } +} + // stubHarnessBinaryWithEnv is stubHarnessBinary plus a dump of the two // variables a routed launch injects, for asserting what the agent actually // got. @@ -452,6 +536,139 @@ func TestLaunchWithEndpointFleetFailsWithoutAToken(t *testing.T) { } } +// A fleet file naming a gateway points the agent at it: the address is the +// applied provider's base URL, and the token is resolved under the variable +// the section names, not the endpoint's default. +func TestLaunchWithAGatewaySectionPointsTheAgentAtTheGateway(t *testing.T) { + home := isolateConfig(t) + t.Setenv("GATEWAY_TOKEN", "gw-token") + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetPath := gatewayFleetFile(t, + "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + captureStdout(t, func() { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("harness was not launched: %v", err) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + out := string(data) + if !strings.Contains(out, "BASE=http://gw.internal:4000/v1") { + t.Errorf("the agent's base URL should be the section's address with the prefix, got:\n%s", out) + } + if !strings.Contains(out, "KEY=gw-token") { + t.Errorf("the agent should carry the token the section's variable holds, got:\n%s", out) + } + config, err := os.ReadFile(filepath.Join(home, ".config", "opencode", "opencode.json")) + if err != nil { + t.Fatalf("the harness config was not written: %v", err) + } + if !strings.Contains(string(config), "http://gw.internal:4000/v1") { + t.Errorf("the applied provider's base URL should be the section's address, got:\n%s", config) + } +} + +// A section naming no tokenEnv resolves under the endpoint FLEET's variable, +// so moving a launch from an endpoint to a section changes nothing the client +// has to export. +func TestLaunchWithAGatewaySectionDefaultsToOpenAIKey(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_API_KEY", "gw-token") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + captureStdout(t, func() { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(data), "KEY=gw-token") { + t.Errorf("the agent should carry the token under the default variable, got:\n%s", data) + } +} + +// A launch at a gateway section whose variable is set nowhere fails before the +// agent launches and before the harness config is written, naming the variable +// the section names. +func TestLaunchWithAGatewaySectionFailsNamingItsVariable(t *testing.T) { + home := isolateConfig(t) + t.Setenv("GATEWAY_TOKEN", "") + t.Setenv("OPENAI_API_KEY", "") + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + + fleetPath := gatewayFleetFile(t, + "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + captureStdout(t, func() { + err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}) + if err == nil { + t.Fatal("a launch that cannot authenticate the gateway should fail") + } + if !strings.Contains(err.Error(), "GATEWAY_TOKEN") { + t.Errorf("the failure should name the variable the section names, got:\n%v", err) + } + }) + if _, err := os.ReadFile(argsFile); err == nil { + t.Error("the agent launched without a token to reach the gateway") + } + if _, err := os.Stat(filepath.Join(home, ".config", "opencode", "opencode.json")); err == nil { + t.Error("the harness config was written for a launch that could not authenticate") + } +} + +// The section's token may sit in the `.env` beside the Spinloop, the way any +// key the launch resolves does: set nowhere in the environment, found beside +// the file. +func TestLaunchWithAGatewaySectionTokenFromDotEnv(t *testing.T) { + isolateConfig(t) + t.Setenv("GATEWAY_TOKEN", "") + t.Setenv("OPENAI_API_KEY", "") + t.Setenv("OPENAI_BASE_URL", "") + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + + fleetPath := gatewayFleetFile(t, + "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + mustWrite(t, filepath.Join(spinloopDir, ".env"), "GATEWAY_TOKEN=dotenv-token\n") + captureStdout(t, func() { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("harness was not launched: %v", err) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(data), "KEY=dotenv-token") { + t.Errorf("the agent should carry the token the .env beside the Spinloop holds, got:\n%s", data) + } +} + // A fleet that declares wake: off refuses to start anything when nothing is // serving, and names the node that would have been woken with the command that // would start it. @@ -627,6 +844,28 @@ func TestCmdFleetRouteAgainstAnEndpoint(t *testing.T) { } } +// A fleet file naming a gateway is answered the way an endpoint is: the +// address is named, and the dead node below proves none is queried. +func TestCmdFleetRouteAgainstAFileNamingAGateway(t *testing.T) { + fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") + spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + + out := captureStdout(t, func() { + if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + t.Fatal(err) + } + }) + for _, want := range []string{ + "names a gateway", + "would point the agent at http://gw.internal:4000/v1", + "nothing is started", + } { + if !strings.Contains(out, want) { + t.Errorf("output should mention %q, got:\n%s", want, out) + } + } +} + // The flag lets the two preferences be compared on a live fleet without // editing the file. func TestCmdFleetRoutePreferenceFlagBeatsTheFile(t *testing.T) { diff --git a/docs/README.md b/docs/README.md index 587c6f44..6cf8ae9e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -62,7 +62,7 @@ Four words carry the whole tool: | [`spinloop serve`](commands/serve.md) | Run the inference server for the model a `Spinloop` names | | [`spinloop up`](commands/up.md) | Start the engine this directory holds: the fleet, or the `Spinloop`'s server | | [`spinloop daemon`](commands/serve.md#the-control-api---api-and-spinloop-daemon) | Supervise an engine over the [control API](http-api.md) | -| [`spinloop fleet`](commands/fleet.md) | Observe and drive the engines on every machine you run | +| [`spinloop fleet`](commands/fleet.md) | Observe and drive the engines on every machine you run, and launch your agent against them | | [`spinloop gateway`](commands/gateway.md) | Serve the fleet under one OpenAI-compatible endpoint | | [`spinloop remote`](commands/remote.md) | Run the model on a cloud GPU that stops when you do | | [`spinloop export`](commands/export.md) | Capture the current setup as a `Spinloop` | diff --git a/docs/commands/fleet.md b/docs/commands/fleet.md index 939d0039..5c73cc20 100644 --- a/docs/commands/fleet.md +++ b/docs/commands/fleet.md @@ -214,6 +214,35 @@ An explicit `--no-wake` still refuses to start anything, whatever the file says; an explicit `spinloop fleet start` does the opposite — it always starts, because it was asked. +### Gateway + +`gateway` names the address this fleet is served under by a +[`spinloop gateway`](gateway.md): a launch routed through this file is pointed +at the gateway rather than at a node. The gateway has done the choosing, so +the launch queries no node and wakes none: + +```yaml +nodes: … +gateway: + url: http://gateway.internal:4000 # required, with a scheme, like a FLEET endpoint + tokenEnv: GATEWAY_TOKEN # optional; OPENAI_API_KEY when absent +``` + +A launch through such a file is dressed exactly as a launch whose `FLEET` +names an endpoint: the section's address is the agent's base URL, and the +token is resolved from the variable the section names — `OPENAI_API_KEY` when +it names none — the way a key is resolved elsewhere: an `ENV` instruction, +then the process environment, then the `.env` beside the Spinloop. A variable +set nowhere fails the launch before anything is written, naming the variable. +As with a node's choice, the launch reports the address on stderr before the +agent starts. + +This is how a machine that holds the fleet file points a harness at the fleet: +`spinloop fleet harness` reads the section when it is there, so a Spinloop +beside the file needs only the model, and the address travels with the file. +`spinloop fleet route` answers a file that names a gateway the same way — the +address, and that no node is queried and nothing is started. + ### Tokens `tokenEnv` names an environment variable; the value is resolved from the @@ -504,6 +533,10 @@ Would use gpu-box at http://gpu-box:8080/v1 serving qwen3-27b, active 312s ago (prefer idle) ``` +A file that names a [gateway](#gateway) is answered the way a launch answers +it — the gateway's address, and that no node is queried and nothing is +started. + When nothing is serving that model it shows the whole fleet's state and names the node a real launch would wake, without waking it: @@ -519,6 +552,31 @@ A launch would wake studio and wait for its engine. Nothing has been started. Use it to check a route before an agent depends on it, to see what the other `prefer` setting would choose, or to work out why a launch landed where it did. +## Launching the harness + +`spinloop fleet harness` is the fleet-level form of a +[harness launch](harness.md#launching-against-your-fleet): the fleet file comes +from the command — `--fleet`, or the `fleet.yaml` beside it — rather than from +the Spinloop's `FLEET`, which stands in when `--fleet` is not given. A fleet +file that names a [gateway](#gateway) points the agent there, so the address +lives in the file, not in every Spinloop: + +```sh +spinloop fleet harness # the Spinloop and fleet.yaml beside it +spinloop fleet harness my-spinloop # a leading Spinloop +spinloop fleet harness -O=./client/Spinloop +spinloop fleet harness --node gpu-box # the launch's steering flags +``` + +Routing is the launch's routing: at the gateway where the file names one, +otherwise by node selection and, where the file's +[wake policy](#waking) allows, a wake — `--node`, `--prefer`, `--no-wake` and +`--wake-timeout` steer it as on the launch. A Spinloop that pins a `BASEURL` is +not routed, and a variable already set in spinloop's environment wins, in each +case as on the launch. With no Spinloop to route — none passed, none beside the +fleet file — the command fails saying a launch needs a Spinloop to know which +model to route. + ## Starting and stopping `fleet start` and `fleet stop` take one or more node names, or `--all` for diff --git a/docs/commands/gateway.md b/docs/commands/gateway.md index c50eeb17..31836f98 100644 --- a/docs/commands/gateway.md +++ b/docs/commands/gateway.md @@ -49,6 +49,23 @@ credential; the node tokens and engine keys live with the gateway, which presents them to the nodes and the engines. See [The `Spinloop` file](../spinloop-file.md#running-the-model-on-another-machine-you-own). +The same pointing can live in the fleet file instead of the Spinloop: a +[`gateway` section](fleet.md#gateway) beside its `nodes` names the address and +the variable holding the token, and `spinloop fleet harness` — the fleet-level +form of a launch — reads it. A Spinloop beside that file then needs only the +model, and the address travels with the file when the gateway moves: + +```yaml +# fleet.yaml +gateway: + url: http://gateway.internal:4000 + tokenEnv: GATEWAY_TOKEN +``` + +```sh +spinloop fleet harness -O=./Spinloop # from the fleet file's directory +``` + ## What it answers | Path | Meaning | diff --git a/docs/commands/harness.md b/docs/commands/harness.md index 9ad31fcb..767fdd1c 100644 --- a/docs/commands/harness.md +++ b/docs/commands/harness.md @@ -96,6 +96,12 @@ Which node wins among several that could all serve you is a takes the machine that has been quiet longest, keeping a second agent off an engine that is mid-request; `active` consolidates onto the busy one instead. +[`spinloop fleet harness`](fleet.md#launching-the-harness) is the fleet-level +form of this launch: the fleet file comes from the command rather than from the +Spinloop's `FLEET`, and a fleet file that names a +[gateway](fleet.md#gateway) points the agent there, so the address lives in the +file, not in every Spinloop. + ## Notes - Trailing arguments and stdio go to the agent untouched, and its exit code is @@ -108,4 +114,5 @@ engine that is mid-request; `active` consolidates onto the busy one instead. - [`spinloop show`](show.md) — what the active harness has configured - [`spinloop apply`](apply.md) — configure without launching - [`spinloop fleet route`](fleet.md#which-node-would-i-get) — which node a launch would pick +- [`spinloop fleet harness`](fleet.md#launching-the-harness) — launch against the fleet file beside you - [`examples/fleet-local/`](../../examples/fleet-local/) — routing at a single local node, end to end diff --git a/examples/gateway-docker/README.md b/examples/gateway-docker/README.md index 5c279f87..dec54fc0 100644 --- a/examples/gateway-docker/README.md +++ b/examples/gateway-docker/README.md @@ -59,10 +59,11 @@ is the one place that holds all of them. There are three Spinloops here: -- [`client/Spinloop`](client/Spinloop) — what an *agent's* machine wears. Its - `FLEET` is the gateway's address, a URL rather than a file: the gateway has - done the choosing, and the agent is only pointed at it, with the gateway's - token as its key. +- [`client/Spinloop`](client/Spinloop) — what an *agent's* machine wears: just + the model. Where it is served lives in [`fleet.yaml`](fleet.yaml)'s `gateway` + section, and `spinloop fleet harness` from this directory reads it: the + gateway has done the choosing, and the agent is only pointed at it, with the + gateway's token as its key. - [`node/Spinloop`](node/Spinloop) — what a *node* runs when started. Its `BASEURL` binds the engine to every interface, which is why the gateway — a different container — can reach it at all. @@ -103,9 +104,9 @@ curl -i -X POST http://127.0.0.1:18080/v1/chat/completions \ -H "Content-Type: application/json" \ -d '{"model":"fake-model","messages":[{"role":"user","content":"hi"}]}' -# Launch an agent against the gateway: the agent gets the gateway's address -# and the gateway's token, and nothing else. -OPENAI_API_KEY="$GATEWAY_TOKEN" spinloop harness ./client/Spinloop +# Launch an agent at the fleet's gateway: fleet.yaml's gateway section points +# the agent at the gateway, with the gateway's token, and nothing else. +spinloop fleet harness -O=./client/Spinloop ``` ## It is also the integration test @@ -126,14 +127,14 @@ an example that is exercised cannot quietly stop working. | File | What it is | | --- | --- | | `compose.yaml` | Two nodes, a gateway that wakes, a gateway that refuses to. Every service that listens on a non-loopback address needs a token, and the gateways need the fleet file's node tokens and engine keys in their environment. | -| `fleet.yaml` | The *operator's* view: the two nodes over their published ports, with `engine:` blocks because the engines are published on ports the daemons cannot know. | +| `fleet.yaml` | The *operator's* view: the two nodes over their published ports, with `engine:` blocks because the engines are published on ports the daemons cannot know — and a `gateway` section naming the waking gateway, so a launch through this file is pointed at it. | | `gateway/fleet.yaml` | What the `gateway` service serves: the same two nodes, addressed by compose service name — where the gateway can reach them, with no `engine:` override needed. | | `gateway/fleet-cold.yaml` | The same fleet with `wake: off`, served by `gateway-cold`. | | `Dockerfile` | Builds spinloop from this working tree, adds the Imposter engine and the shim, and bakes the gateway's files in. | | `shim/llama-server` | Stands in for the engine binary. Reads the key file the daemon passes and hands the mock its gate as an environment variable, so the value never rides on a command line. | | `engine/` | What the fake engine serves: `/health` and `/metrics` for the daemon, and the gated, stream-answering OpenAI routes. | | `node/Spinloop` | What a node runs when started: a model, and a `BASEURL` that binds the engine to every interface. | -| `client/Spinloop` | What an *agent's* machine wears: a model, and a `FLEET` that is the gateway's address. | +| `client/Spinloop` | What an *agent's* machine wears: the model. Its address comes from `fleet.yaml`'s gateway section, via `spinloop fleet harness`. | Two details that are easy to get wrong, and matter: diff --git a/examples/gateway-docker/client/Spinloop b/examples/gateway-docker/client/Spinloop index c56f8a21..918ef42f 100644 --- a/examples/gateway-docker/client/Spinloop +++ b/examples/gateway-docker/client/Spinloop @@ -1,13 +1,13 @@ -# What an agent's machine wears to use this stack: a model, and a FLEET that -# names the gateway's address rather than a fleet file. The gateway has done -# the choosing; the agent is only pointed at it. +# What an agent's machine wears to use this stack: a model, and nothing about +# where it is served — the fleet file's gateway section says that. `spinloop +# fleet harness` with this Spinloop beside that fleet file points the agent at +# the gateway; the Spinloop stays valid whatever address the gateway moves to. # # The agent holds exactly one credential — the gateway's token, as -# OPENAI_API_KEY (from the environment, or a .env beside this file). The node -# tokens and engine keys live with the gateway, which presents them to the -# nodes and the engines; none of them ever reach the agent. +# GATEWAY_TOKEN (from the environment, or a .env beside the fleet file). The +# node tokens and engine keys live with the gateway, which presents them to +# the nodes and the engines; none of them ever reach the agent. PROVIDER llamacpp MODEL org/fake-model ALIAS fake-model CONTEXT 4096 -FLEET http://127.0.0.1:4000 diff --git a/examples/gateway-docker/fleet.yaml b/examples/gateway-docker/fleet.yaml index 76d5dd75..1c47cb2c 100644 --- a/examples/gateway-docker/fleet.yaml +++ b/examples/gateway-docker/fleet.yaml @@ -2,9 +2,10 @@ # ports compose publishes. # # The gateways are not nodes in it — they serve the fleet files baked into -# the image — and the client's Spinloop names the gateway's published address, -# not this file. This is how you watch and drive the fleet itself: status, -# start, stop, metrics, logs. +# the image. The `gateway` section below names the waking gateway's published +# address instead: `spinloop fleet harness` from here points the agent at the +# gateway, with the gateway's token, and never at a node. This is also how you +# watch and drive the fleet itself: status, start, stop, metrics, logs. # # The engine: blocks here are not optional: each engine binds 8080 inside its # container and is published on 18080/18081 outside, which the daemons cannot @@ -16,6 +17,12 @@ # and pushes that. prefer: idle +# A launch routed through this file is pointed at the gateway: the address it +# serves on, and the variable holding the token for it (exported by .env). +gateway: + url: http://127.0.0.1:4000 + tokenEnv: GATEWAY_TOKEN + nodes: - name: node-a host: 127.0.0.1 diff --git a/examples/gateway-docker/run-tests.sh b/examples/gateway-docker/run-tests.sh index 622efcab..8cc9ce9b 100755 --- a/examples/gateway-docker/run-tests.sh +++ b/examples/gateway-docker/run-tests.sh @@ -492,13 +492,14 @@ test_cold_request_wakes_and_streams() { } ####################################### -# Assert a harness launch against the client's Spinloop points the agent at -# the gateway's address with the gateway's token as its key. +# Assert `spinloop fleet harness` against the client's Spinloop points the +# agent at the fleet file's gateway — the section's address, the section's +# token variable — and writes the harness config for it. # Globals: -# HERE, SPINLOOP_BIN +# HERE, SPINLOOP_BIN, GATEWAY_TOKEN ####################################### test_launch_points_agent_at_gateway() { - echo "A launch points the agent at the gateway" + echo "A launch points the agent at the fleet's gateway" local sandbox="${HERE}/.launch-sandbox" rm -rf "${sandbox}" mkdir -p "${sandbox}/bin" "${sandbox}/home" @@ -508,12 +509,14 @@ echo "HARNESS base_url=${OPENAI_BASE_URL:-} key=${OPENAI_API_KEY:- STUB chmod +x "${sandbox}/bin/opencode" + # From this directory: no -f, so the fleet.yaml beside the Spinloop is the + # one routed through, and its gateway section is where the agent goes. local launch launch="$(PATH="${sandbox}/bin:${PATH}" HOME="${sandbox}/home" \ XDG_CONFIG_HOME="${sandbox}/home/.config" \ OPENAI_BASE_URL="" \ - OPENAI_API_KEY="${GATEWAY_TOKEN}" \ - "${SPINLOOP_BIN}" harness -O="${HERE}/client/Spinloop" -H opencode 2>&1 || true)" + OPENAI_API_KEY="" \ + "${SPINLOOP_BIN}" fleet harness -O=client/Spinloop -H opencode 2>&1 || true)" assert_contains "the agent is pointed at the gateway with its prefix" \ "${launch}" "base_url=http://127.0.0.1:4000/v1" assert_contains "the agent is given the gateway's token as its key" \ @@ -531,7 +534,8 @@ STUB ####################################### # Assert a launch that cannot authenticate the gateway fails before the agent -# is started and before anything is written, naming the variable. +# is started and before anything is written, naming the variable the fleet +# file's gateway section holds. # Globals: # HERE, SPINLOOP_BIN ####################################### @@ -550,9 +554,9 @@ STUB launch="$(PATH="${sandbox}/bin:${PATH}" HOME="${sandbox}/home" \ XDG_CONFIG_HOME="${sandbox}/home/.config" \ OPENAI_BASE_URL="" \ - OPENAI_API_KEY="" \ - "${SPINLOOP_BIN}" harness -O="${HERE}/client/Spinloop" -H opencode 2>&1 || true)" - assert_contains "the failure names the variable to set" "${launch}" "OPENAI_API_KEY" + GATEWAY_TOKEN="" \ + "${SPINLOOP_BIN}" fleet harness -O=client/Spinloop -H opencode 2>&1 || true)" + assert_contains "the failure names the variable to set" "${launch}" "GATEWAY_TOKEN" assert_not_contains "the agent was not started" "${launch}" "HARNESS" if [[ ! -f "${sandbox}/home/.config/opencode/opencode.json" ]]; then pass "no harness config was written" diff --git a/internal/fleet/config.go b/internal/fleet/config.go index 1083c7cd..7f72294d 100644 --- a/internal/fleet/config.go +++ b/internal/fleet/config.go @@ -96,6 +96,13 @@ func (c *Config) Wakes() bool { // (the directory whose .env supplies token values). type Config struct { Nodes []NodeConfig `yaml:"nodes"` + // Gateway names the address the fleet is served under by its gateway. A + // launch routed through this file points the agent at the gateway rather + // than at a node: the gateway has done the choosing, so routing contacts + // no node and wakes none. It is a section rather than a node kind because + // it is not a machine the fleet drives — it has no control API, and no + // fleet operation but routing ever looks at it. + Gateway *GatewayConfig `yaml:"gateway"` // Prefer ranks nodes that could all serve a request. It belongs to the // file rather than to a node because it describes how this cluster // should be used — spread the work, or consolidate it. Empty means @@ -179,6 +186,38 @@ type EngineOverride struct { Path string `yaml:"path"` } +// DefaultGatewayTokenEnv is the variable a gateway's token is resolved under +// when the section names none: the same variable an endpoint FLEET resolves +// under, so a file that moves a launch from an endpoint to a section changes +// nothing the client has to export. +const DefaultGatewayTokenEnv = "OPENAI_API_KEY" + +// GatewayConfig is the fleet file's gateway section: the address the fleet is +// served under, and where the client's token for it lives. As with every other +// secret in this file, the token itself is never written here. +type GatewayConfig struct { + // URL is the gateway's address. It carries a scheme, the way an endpoint + // value does — the gateway is reached over HTTP, and a bare host names + // nothing spinloop could dial. + URL string `yaml:"url"` + // TokenEnv names the environment variable holding the gateway's token. + // Empty means DefaultGatewayTokenEnv. + TokenEnv string `yaml:"tokenEnv"` +} + +// GatewaySection returns the file's gateway section, with its token variable +// defaulted, and whether the file names a gateway at all. +func (c *Config) GatewaySection() (GatewayConfig, bool) { + if c.Gateway == nil { + return GatewayConfig{}, false + } + gw := *c.Gateway + if gw.TokenEnv == "" { + gw.TokenEnv = DefaultGatewayTokenEnv + } + return gw, true +} + // BaseURL is the root of this node's control API. func (n NodeConfig) BaseURL() string { return fmt.Sprintf("http://%s:%d", n.Host, n.port()) @@ -241,6 +280,19 @@ func (c *Config) validate() error { return err } } + if c.Gateway != nil { + if c.Gateway.URL == "" { + return fmt.Errorf("the gateway section names no url: name the gateway's address under `url:`") + } + // The section's url is an endpoint value wearing a section: it is + // dialed over HTTP, so it carries a scheme the way FLEET's endpoint + // values do. + if !strings.Contains(c.Gateway.URL, "://") { + return fmt.Errorf( + "the gateway section's url %q has no scheme: give the gateway's full address, the way an endpoint value does", + c.Gateway.URL) + } + } seen := map[string]bool{} for i := range c.Nodes { n := &c.Nodes[i] diff --git a/internal/fleet/config_test.go b/internal/fleet/config_test.go index c99c5862..0fedb7a6 100644 --- a/internal/fleet/config_test.go +++ b/internal/fleet/config_test.go @@ -663,3 +663,87 @@ func TestWakeRejectsUnknownValue(t *testing.T) { } } } + +func TestGatewaySection(t *testing.T) { + path := writeFleet(t, ` +nodes: + - name: studio + host: studio.local +gateway: + url: https://gw.example.com + tokenEnv: GW_TOKEN +`, "") + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + gw, ok := cfg.GatewaySection() + if !ok { + t.Fatal("a file with a gateway section should report one") + } + if gw.URL != "https://gw.example.com" || gw.TokenEnv != "GW_TOKEN" { + t.Errorf("gateway = %+v", gw) + } +} + +// A section that names no tokenEnv resolves under the endpoint FLEET's +// variable, so moving a launch from an endpoint to a section changes nothing +// the client has to export. +func TestGatewaySectionTokenDefaults(t *testing.T) { + path := writeFleet(t, ` +nodes: + - name: studio + host: studio.local +gateway: + url: https://gw.example.com +`, "") + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + gw, ok := cfg.GatewaySection() + if !ok { + t.Fatal("a file with a gateway section should report one") + } + if gw.TokenEnv != DefaultGatewayTokenEnv { + t.Errorf("tokenEnv = %q, want the default %q", gw.TokenEnv, DefaultGatewayTokenEnv) + } +} + +// No section: the accessor says so, and the file behaves as it always has. +func TestNoGatewaySection(t *testing.T) { + path := writeFleet(t, ` +nodes: + - name: studio + host: studio.local +`, "") + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if _, ok := cfg.GatewaySection(); ok { + t.Error("a file without a gateway section should report none") + } +} + +func TestLoadRefusesAGatewayWithoutAURL(t *testing.T) { + _, err := Load(writeFleet(t, + "nodes:\n - name: a\n host: a.local\ngateway:\n tokenEnv: GW_TOKEN\n", "")) + if err == nil { + t.Fatal("a gateway section naming no url should be refused") + } + if !strings.Contains(err.Error(), "url") { + t.Errorf("the refusal should name the missing field, got %q", err) + } +} + +func TestLoadRefusesAGatewayURLWithoutAScheme(t *testing.T) { + _, err := Load(writeFleet(t, + "nodes:\n - name: a\n host: a.local\ngateway:\n url: gw.example.com\n", "")) + if err == nil { + t.Fatal("a gateway url without a scheme should be refused") + } + if !strings.Contains(err.Error(), "scheme") { + t.Errorf("the refusal should say the url lacks a scheme, got %q", err) + } +} diff --git a/internal/fleet/select.go b/internal/fleet/select.go index 3cc62d0c..9c3ca580 100644 --- a/internal/fleet/select.go +++ b/internal/fleet/select.go @@ -120,8 +120,14 @@ type Choice struct { Woken bool // Gateway records that FLEET named an endpoint rather than a fleet file: // the endpoint has already done the choosing, Node is empty, and BaseURL - // is the endpoint's address rather than a node's engine. + // is the endpoint's address rather than a node's engine. A fleet file's + // gateway section yields the same shape. Gateway bool + // GatewayTokenEnv names the variable holding the token for a Gateway + // choice: the variable a fleet file's gateway section names, defaulted to + // the endpoint FLEET's variable when the section names none. The launch + // resolves it through its key chain, as with every other key. + GatewayTokenEnv string } // candidate pairs a node's file entry with what it answered, keeping the diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/.openspec.yaml b/openspec/changes/archive/2026-09-08-add-fleet-harness/.openspec.yaml new file mode 100644 index 00000000..7a8e2be6 --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-08 diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/design.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/design.md new file mode 100644 index 00000000..c396d875 --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/design.md @@ -0,0 +1,58 @@ +## Context + +See proposal.md. Two questions this design settles: where in the fleet file +the gateway lives, and what the new command does with it. + +## Goals / Non-Goals + +**Goals:** + +- The gateway is a property of the fleet: the fleet file names it, and the + command at the fleet's level uses it. +- `spinloop fleet harness` reuses the launch path's routing wholesale — no + second selector, no second wake path. + +**Non-Goals:** + +- Removing `FLEET` from Spinloops: the follow-up breaking change. +- `spinloop gateway` reading its own section for its token: a natural + follow-up, not part of this change. + +## Decisions + +### A section, not a node kind + +A node kind would put the gateway inside the fleet's machinery: every fleet +operation — the status fan-out, start/stop, selection, ranking, wake — would +special-case an entry it cannot act on. A gateway has no control API, nothing +in the fleet starts or stops it, and it must never be a selection candidate, +since it serves whatever its nodes serve. A top-level section costs the +machinery nothing: the fleet's commands ignore it, and only the paths that +need a gateway read it. Singular also reads correctly — a fleet has one +canonical front door, and extra gateways are operational choices, reached by +naming their URL. + +### The command's file wins, then the Spinloop's FLEET, then the default + +An explicit `-f` wins over a Spinloop's `FLEET`, generalising the existing +`--fleet`-overrides-instruction rule; with no `-f`, the Spinloop's `FLEET` +(file or endpoint) is used as today; with neither, `./fleet.yaml` is read. One +rule covers all four sources, and every existing launch keeps behaving exactly +as it does. + +### The section routes the way an endpoint routes + +A file naming a gateway produces the same answer an endpoint `FLEET` does — +the address as the base URL with the OpenAI-compatible prefix, the token +through the client's key chain, no node contacted — differing only in where +the address and the variable come from: the section rather than the +instruction. The token's default variable is `OPENAI_API_KEY`, the one an +endpoint already resolves under. + +### The command is a thin front on the launch path + +`spinloop fleet harness` resolves its two inputs (Spinloop, fleet file) and +then runs the launch path's routing — the gateway branch, selection and wake, +the `BASEURL` and environment rules, the stderr announcement — rather than +re-implementing any of it. Its one behaviour the launch path lacks: with no +Spinloop at all it fails, because routing needs a model to route. diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/proposal.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/proposal.md new file mode 100644 index 00000000..948e8268 --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/proposal.md @@ -0,0 +1,63 @@ +## Why + +A harness that runs against a fleet today learns the gateway's address from a +`FLEET` instruction in a Spinloop — but the Spinloop is the node-level +artifact (what a node runs), and the fleet's nodes already name their +Spinloops. Having a Spinloop name a fleet or a gateway closes that circle: one +artifact doing two jobs. The gateway is a property of the fleet, and the +command that points a harness at a fleet belongs at the fleet's level of the +hierarchy. + +## What Changes + +- The fleet file MAY name its gateway: an optional top-level `gateway` section + beside `wake:` and `prefer:` carrying the gateway's `url` and, optionally, + a `tokenEnv` naming the variable that holds its token. +- New `spinloop fleet harness` command: the fleet-level form of a + fleet-routed launch. It takes a Spinloop the way `spinloop harness` takes + one and a fleet file from `--fleet`/`-f` (defaulting to the `fleet.yaml` + beside it), routes the launch at the gateway the file names — or, where the + file names none, to a node with the existing selection and wake — applies + the Spinloop, and launches the harness. +- Routing precedence generalises the existing rule: an explicit `-f` wins + over a Spinloop's `FLEET`; with no `-f`, the Spinloop's `FLEET` (file or + endpoint) is used; with neither, `./fleet.yaml` is read. A pinned `BASEURL` + and an exported `OPENAI_BASE_URL` still win over any routing. +- `spinloop fleet route` answers a Spinloop whose fleet file names a gateway: + it reports the gateway's address, queries no node, and wakes nothing. +- `examples/gateway-docker/` moves its client onto the command: the example's + fleet file gains a `gateway` section, the client Spinloop loses its `FLEET` + URL, and the integration test exercises the new path. + +## Capabilities + +### New Capabilities + +(none) + +### Modified Capabilities + +- `fleet-config`: a fleet file MAY declare a top-level `gateway` section + naming the address the fleet is served under and the variable holding its + token. +- `fleet-routing`: a launch whose effective fleet file names a gateway is + routed at that gateway (the section's address as the base URL, its token + variable resolved through the client's key chain, no node contacted); + `fleet route` answers such a file by naming the gateway. +- `fleet-client`: a `spinloop fleet harness` command — the fleet-level way to + configure and launch the harness against a fleet. + +## Impact + +- The fleet file's config package: the `gateway` section's parsing. +- `cmd/spinloop`: the new `fleet harness` subcommand, built on the launch + path's routing rather than re-implementing it; `fleet route`'s endpoint + answer gains the file-with-a-section case. +- `examples/gateway-docker/`: the client Spinloop's `FLEET` URL becomes a + `gateway` section plus the command; `run-tests.sh` asserts the new flow. +- `docs/`: the section in the fleet file documentation; + `docs/commands/gateway.md` naming the command as the way to point a harness + at a fleet. +- Additive: a `FLEET` instruction keeps working exactly as it does today — + its removal is a separate, later change. No daemon endpoint changes, no new + dependencies. diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-client/spec.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-client/spec.md new file mode 100644 index 00000000..64030d1a --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-client/spec.md @@ -0,0 +1,54 @@ +## ADDED Requirements + +### Requirement: A fleet harness command + +`spinloop fleet harness` SHALL configure the active harness for a fleet and +launch it: the fleet-level form of a fleet-routed launch, in which the fleet +file comes from the command rather than from a Spinloop's `FLEET`. The command +SHALL take a Spinloop the way `spinloop harness` does — an +`-O`/`--spinloop` argument, a leading alias or path, or the `Spinloop` beside +it — and a fleet file from `--fleet`/`-f`, defaulting to the `fleet.yaml` +beside it. + +The command SHALL route the launch the way a fleet-routed launch routes: at +the gateway where the effective fleet file names one, otherwise by choosing a +node and, where the fleet's wake policy allows it, waking one — honouring +`--node`, `--prefer`, `--no-wake` and `--wake-timeout` as the launch does. The +choice SHALL be reported on stderr before the harness launches, as a +fleet-routed launch reports its choice. + +Where the command is given no `-f`, a Spinloop that names a `FLEET` — a file +or an endpoint — SHALL be used, exactly as `--fleet` overrides an instruction +on `spinloop harness`; an explicit `-f` SHALL win over the instruction. A +Spinloop that pins a `BASEURL` SHALL NOT be routed, and a variable already set +in spinloop's environment SHALL win, in each case as on the launch path. A +command with no Spinloop to route SHALL fail before launching, saying that a +launch needs a Spinloop to know which model to route. + +#### Scenario: A fleet's gateway is used + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding a + fleet file that names a gateway and a Spinloop, and the gateway's token is + set +- **THEN** the harness is applied for the Spinloop's model, the launched + agent's endpoint is the gateway's address, and no node is contacted + +#### Scenario: A fleet with no gateway routes to a node + +- **WHEN** the user runs `spinloop fleet harness` against a fleet file that + names no gateway, and a node is running the Spinloop's model +- **THEN** the harness is applied with that node's engine as the agent's + endpoint, as a fleet-routed launch would apply it + +#### Scenario: The command's file wins over the Spinloop's FLEET + +- **WHEN** the user runs `spinloop fleet harness -f ./a.yaml` with a Spinloop + whose `FLEET` names a different file +- **THEN** the fleet in `./a.yaml` is the one the launch routes through + +#### Scenario: No Spinloop, no route + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding no + Spinloop and gives none +- **THEN** the command fails saying a launch needs a Spinloop to know which + model to route, and no harness is launched diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-config/spec.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-config/spec.md new file mode 100644 index 00000000..cc3887dd --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-config/spec.md @@ -0,0 +1,34 @@ +## ADDED Requirements + +### Requirement: A fleet file MAY name a gateway + +A fleet file MAY declare a top-level `gateway` section, beside `wake:` and +`prefer:`, naming the address the fleet is served under by its gateway: a +`url`, and optionally a `tokenEnv` naming the variable that holds the +gateway's token. The section is how a machine that holds the fleet file +learns how to point a harness at the fleet without naming the address in a +Spinloop. + +The section's `url` SHALL carry a scheme, the way an endpoint value does, and +a `gateway` section without one SHALL be refused, naming the missing field. A +`tokenEnv` SHALL name a variable, and where the section names none, the token +SHALL be resolved under `OPENAI_API_KEY`, the variable an endpoint `FLEET` +already resolves under. A file without a `gateway` section SHALL behave +exactly as it does today. + +#### Scenario: A file names its gateway + +- **WHEN** a fleet file declares a `gateway` section naming a `url` and a + `tokenEnv`, and the file is read +- **THEN** the section's address and token variable are available to the + commands that route a launch at the fleet + +#### Scenario: A section without an address is refused + +- **WHEN** a fleet file declares a `gateway` section naming no `url` +- **THEN** the file is refused, naming the missing field + +#### Scenario: No section, no change + +- **WHEN** a fleet file declares no `gateway` section +- **THEN** nothing about the file's behaviour changes diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-routing/spec.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-routing/spec.md new file mode 100644 index 00000000..94b0994e --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/specs/fleet-routing/spec.md @@ -0,0 +1,55 @@ +## ADDED Requirements + +### Requirement: A fleet file naming a gateway routes the launch at it + +A launch whose effective fleet file — one given by `--fleet`/`-f`, or named +by the Spinloop's `FLEET` — declares a `gateway` section SHALL route at that +gateway the way a launch routes at an endpoint: the section's address SHALL be +written as the applied provider's base URL, with the OpenAI-compatible prefix +appended when it carries no path, and SHALL be placed in the launched +agent's environment as `OPENAI_BASE_URL`. No node SHALL be contacted and none +SHALL be woken: the gateway has already done the choosing. + +The gateway's token SHALL be resolved from the variable the section names — or +from `OPENAI_API_KEY` where the section names none — through the client's +existing key chain, with a variable already set in spinloop's environment +winning. When the variable is set nowhere, the launch SHALL fail before the +harness config is written, naming the variable to set. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed at the section: the +pinned address wins, as it wins over an endpoint. + +#### Scenario: A launch is pointed at the fleet's gateway + +- **WHEN** the user runs a launch against a fleet file whose `gateway` section + names an address, and the section's token variable is set +- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` at the + section's address with the OpenAI-compatible prefix, the applied provider's + base URL is the same address, and no node is contacted + +#### Scenario: A missing gateway token fails early + +- **WHEN** a launch routes at a fleet file's `gateway` section and the + variable the section names is set nowhere +- **THEN** the launch fails before the harness config is written, naming the + variable to set + +#### Scenario: A pinned BASEURL still wins over the section + +- **WHEN** a Spinloop pins a `BASEURL` and its fleet file names a gateway +- **THEN** the `BASEURL` is used, the gateway is not, and the launch says it + is not routing + +### Requirement: fleet route answers a fleet file naming a gateway + +`spinloop fleet route` on a Spinloop whose fleet file declares a `gateway` +section SHALL report that the gateway has already chosen: it SHALL name the +address the launch will be given, SHALL NOT query any node, and SHALL NOT wake +one. + +#### Scenario: A route against a file naming a gateway names it + +- **WHEN** the user runs `spinloop fleet route` on a Spinloop whose fleet file + declares a `gateway` section +- **THEN** the output names the section's address as the one the launch will + use, no node is queried, and nothing is started diff --git a/openspec/changes/archive/2026-09-08-add-fleet-harness/tasks.md b/openspec/changes/archive/2026-09-08-add-fleet-harness/tasks.md new file mode 100644 index 00000000..92857121 --- /dev/null +++ b/openspec/changes/archive/2026-09-08-add-fleet-harness/tasks.md @@ -0,0 +1,23 @@ +# Tasks + +## 1. The fleet file's gateway section + +- [x] 1.1 Parse an optional top-level `gateway` section in the fleet file — a `url` (required, carrying a scheme; a section without one refused naming the field) and a `tokenEnv` naming the token's variable, absent one defaulting to `OPENAI_API_KEY` — with a `fleet.Config` accessor for the section, verified by config tests (a section with both fields, a section with only a url, a section with no url refused naming the field, and no section leaving the file unchanged) + +## 2. Routing at a file's gateway + +- [x] 2.1 Route a launch whose effective fleet file names a gateway at that gateway: the section's address written as the applied provider's base URL with the OpenAI-compatible prefix and as `OPENAI_BASE_URL`, the token resolved from the section's variable through the client's key chain (an already-set variable winning, a variable set nowhere failing before anything is written and naming it), no node contacted and none woken, and a pinned `BASEURL` still winning and saying so, verified by launch tests for each +- [x] 2.2 Make `spinloop fleet route` answer a Spinloop whose fleet file names a gateway by naming the address the launch will be given, querying no node and starting nothing, verified by a command test + +## 3. The command + +- [x] 3.1 Add `spinloop fleet harness` beside the other fleet subcommands: a Spinloop taken the way `spinloop harness` takes one (`-O`, a leading alias or path, or the `Spinloop` beside it), a fleet file from `--fleet`/`-f` defaulting to the `fleet.yaml` beside it, and the launch's steering flags (`--node`, `--prefer`, `--no-wake`, `--wake-timeout`); with no `-f` a Spinloop's `FLEET` (file or endpoint) is used, and an explicit `-f` wins over it; it runs the launch path's routing — the gateway where the file names one, node selection and wake otherwise — announces the choice on stderr, applies the Spinloop, and launches the harness; a command with no Spinloop fails before launching, saying a launch needs a Spinloop to know which model to route, verified by command tests (the gateway is used when the file names one, a node is routed to when it names none, the command's file wins over the instruction, and no Spinloop fails) and by the completion dispatch-coverage scan passing with the new subcommand + +## 4. Example and documentation + +- [x] 4.1 Move `examples/gateway-docker/` onto the command: the example's fleet file gains a `gateway` section, the client Spinloop loses its `FLEET` URL, and `run-tests.sh` asserts the client is configured through `spinloop fleet harness` and reaches the fleet through the section's address, verified by `./run-tests.sh` passing +- [x] 4.2 Document the change where the rest of it is documented: the `gateway` section in the fleet file documentation, and `docs/commands/gateway.md` naming `spinloop fleet harness` as the way to point a harness at a fleet, verified by the docs links resolving and the new pages reading against the implemented flags + +## 5. Final verification + +- [x] 5.1 Run the full gate: `go test ./... -cover` at or above the 80% total, `go vet ./...`, `gofmt -l .` clean, the control plane's `pnpm build` and `pnpm test`, `openspec validate add-fleet-harness`, and `examples/gateway-docker/run-tests.sh`, fixing whatever each reports diff --git a/openspec/specs/fleet-client/spec.md b/openspec/specs/fleet-client/spec.md index 3d288e9c..0c6d137b 100644 --- a/openspec/specs/fleet-client/spec.md +++ b/openspec/specs/fleet-client/spec.md @@ -5,7 +5,9 @@ Define the `spinloop fleet` command family: the client that reads `fleet.yaml`, polls each node's daemon control API, and renders the cluster — observing every engine and driving individual ones, degrading gracefully when a node cannot be reached. + ## Requirements + ### Requirement: Fleet status `spinloop fleet status` SHALL query every node's daemon status endpoint and render one row per node: the node name, its engine state (`idle`/`running`/`stopped`/`crashed`), what it is serving (runner and model when known), the spinloop version of the daemon on that node, and its reachability. Nodes SHALL be queried concurrently so the command's latency is that of the slowest reachable node, not their sum. @@ -1538,3 +1540,56 @@ the sweep has already moved past as though the node were still held. - **WHEN** the operator opens the detail screen of a retained node - **THEN** the detail screen's deadline line is the same line, worded the same, the tile draws for the same read + +### Requirement: A fleet harness command + +`spinloop fleet harness` SHALL configure the active harness for a fleet and +launch it: the fleet-level form of a fleet-routed launch, in which the fleet +file comes from the command rather than from a Spinloop's `FLEET`. The command +SHALL take a Spinloop the way `spinloop harness` does — an +`-O`/`--spinloop` argument, a leading alias or path, or the `Spinloop` beside +it — and a fleet file from `--fleet`/`-f`, defaulting to the `fleet.yaml` +beside it. + +The command SHALL route the launch the way a fleet-routed launch routes: at +the gateway where the effective fleet file names one, otherwise by choosing a +node and, where the fleet's wake policy allows it, waking one — honouring +`--node`, `--prefer`, `--no-wake` and `--wake-timeout` as the launch does. The +choice SHALL be reported on stderr before the harness launches, as a +fleet-routed launch reports its choice. + +Where the command is given no `-f`, a Spinloop that names a `FLEET` — a file +or an endpoint — SHALL be used, exactly as `--fleet` overrides an instruction +on `spinloop harness`; an explicit `-f` SHALL win over the instruction. A +Spinloop that pins a `BASEURL` SHALL NOT be routed, and a variable already set +in spinloop's environment SHALL win, in each case as on the launch path. A +command with no Spinloop to route SHALL fail before launching, saying that a +launch needs a Spinloop to know which model to route. + +#### Scenario: A fleet's gateway is used + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding a + fleet file that names a gateway and a Spinloop, and the gateway's token is + set +- **THEN** the harness is applied for the Spinloop's model, the launched + agent's endpoint is the gateway's address, and no node is contacted + +#### Scenario: A fleet with no gateway routes to a node + +- **WHEN** the user runs `spinloop fleet harness` against a fleet file that + names no gateway, and a node is running the Spinloop's model +- **THEN** the harness is applied with that node's engine as the agent's + endpoint, as a fleet-routed launch would apply it + +#### Scenario: The command's file wins over the Spinloop's FLEET + +- **WHEN** the user runs `spinloop fleet harness -f ./a.yaml` with a Spinloop + whose `FLEET` names a different file +- **THEN** the fleet in `./a.yaml` is the one the launch routes through + +#### Scenario: No Spinloop, no route + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding no + Spinloop and gives none +- **THEN** the command fails saying a launch needs a Spinloop to know which + model to route, and no harness is launched diff --git a/openspec/specs/fleet-config/spec.md b/openspec/specs/fleet-config/spec.md index a5d0b398..8ad38956 100644 --- a/openspec/specs/fleet-config/spec.md +++ b/openspec/specs/fleet-config/spec.md @@ -360,3 +360,36 @@ source describes the wanted model and the command that would start it. - **WHEN** a fleet file declares `wake: sometimes` - **THEN** parsing fails naming `on` and `off` + +### Requirement: A fleet file MAY name a gateway + +A fleet file MAY declare a top-level `gateway` section, beside `wake:` and +`prefer:`, naming the address the fleet is served under by its gateway: a +`url`, and optionally a `tokenEnv` naming the variable that holds the +gateway's token. The section is how a machine that holds the fleet file +learns how to point a harness at the fleet without naming the address in a +Spinloop. + +The section's `url` SHALL carry a scheme, the way an endpoint value does, and +a `gateway` section without one SHALL be refused, naming the missing field. A +`tokenEnv` SHALL name a variable, and where the section names none, the token +SHALL be resolved under `OPENAI_API_KEY`, the variable an endpoint `FLEET` +already resolves under. A file without a `gateway` section SHALL behave +exactly as it does today. + +#### Scenario: A file names its gateway + +- **WHEN** a fleet file declares a `gateway` section naming a `url` and a + `tokenEnv`, and the file is read +- **THEN** the section's address and token variable are available to the + commands that route a launch at the fleet + +#### Scenario: A section without an address is refused + +- **WHEN** a fleet file declares a `gateway` section naming no `url` +- **THEN** the file is refused, naming the missing field + +#### Scenario: No section, no change + +- **WHEN** a fleet file declares no `gateway` section +- **THEN** nothing about the file's behaviour changes diff --git a/openspec/specs/fleet-routing/spec.md b/openspec/specs/fleet-routing/spec.md index 5414ad68..cfeedeaf 100644 --- a/openspec/specs/fleet-routing/spec.md +++ b/openspec/specs/fleet-routing/spec.md @@ -483,3 +483,57 @@ authenticate is worse than a message that says so. - **WHEN** `OPENAI_API_KEY` is already set in the user's environment and a fleet-routed launch runs - **THEN** the existing value reaches the agent unchanged + +### Requirement: A fleet file naming a gateway routes the launch at it + +A launch whose effective fleet file — one given by `--fleet`/`-f`, or named +by the Spinloop's `FLEET` — declares a `gateway` section SHALL route at that +gateway the way a launch routes at an endpoint: the section's address SHALL be +written as the applied provider's base URL, with the OpenAI-compatible prefix +appended when it carries no path, and SHALL be placed in the launched +agent's environment as `OPENAI_BASE_URL`. No node SHALL be contacted and none +SHALL be woken: the gateway has already done the choosing. + +The gateway's token SHALL be resolved from the variable the section names — or +from `OPENAI_API_KEY` where the section names none — through the client's +existing key chain, with a variable already set in spinloop's environment +winning. When the variable is set nowhere, the launch SHALL fail before the +harness config is written, naming the variable to set. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed at the section: the +pinned address wins, as it wins over an endpoint. + +#### Scenario: A launch is pointed at the fleet's gateway + +- **WHEN** the user runs a launch against a fleet file whose `gateway` section + names an address, and the section's token variable is set +- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` at the + section's address with the OpenAI-compatible prefix, the applied provider's + base URL is the same address, and no node is contacted + +#### Scenario: A missing gateway token fails early + +- **WHEN** a launch routes at a fleet file's `gateway` section and the + variable the section names is set nowhere +- **THEN** the launch fails before the harness config is written, naming the + variable to set + +#### Scenario: A pinned BASEURL still wins over the section + +- **WHEN** a Spinloop pins a `BASEURL` and its fleet file names a gateway +- **THEN** the `BASEURL` is used, the gateway is not, and the launch says it + is not routing + +### Requirement: fleet route answers a fleet file naming a gateway + +`spinloop fleet route` on a Spinloop whose fleet file declares a `gateway` +section SHALL report that the gateway has already chosen: it SHALL name the +address the launch will be given, SHALL NOT query any node, and SHALL NOT wake +one. + +#### Scenario: A route against a file naming a gateway names it + +- **WHEN** the user runs `spinloop fleet route` on a Spinloop whose fleet file + declares a `gateway` section +- **THEN** the output names the section's address as the one the launch will + use, no node is queried, and nothing is started From a68a03822f41c16b2d063ecfe19be3fdd982efa3 Mon Sep 17 00:00:00 2001 From: Pete Cornish Date: Sun, 13 Sep 2026 00:48:14 +0100 Subject: [PATCH 4/4] feat: drop the Spinloop's FLEET and route launches by fleet file --- README.md | 35 +- cmd/spinloop/commands.go | 8 +- cmd/spinloop/fleet.go | 58 +-- cmd/spinloop/fleet_harness_test.go | 37 +- cmd/spinloop/gateway.go | 16 +- cmd/spinloop/gateway_test.go | 8 +- cmd/spinloop/harness_test.go | 16 +- cmd/spinloop/main.go | 26 +- cmd/spinloop/route.go | 88 ++-- cmd/spinloop/route_test.go | 488 +++++++++--------- docs/commands/fleet.md | 27 +- docs/commands/gateway.md | 45 +- docs/commands/harness.md | 22 +- docs/spinloop-file.md | 62 +-- examples/fleet-docker/README.md | 6 +- examples/fleet-docker/client/Spinloop | 3 +- examples/fleet-docker/run-tests.sh | 28 +- examples/fleet-local/README.md | 24 +- examples/fleet-local/Spinloop | 7 +- examples/gateway-docker/README.md | 2 +- examples/gateway-docker/client/Spinloop | 4 +- internal/fleet/config.go | 11 +- internal/fleet/config_test.go | 4 +- internal/fleet/select.go | 11 +- internal/gateway/gateway.go | 8 +- internal/spinloop/spinloop.go | 32 +- internal/spinloop/spinloop_test.go | 55 +- .../.openspec.yaml | 2 + .../design.md | 102 ++++ .../proposal.md | 65 +++ .../specs/fleet-client/spec.md | 118 +++++ .../specs/fleet-config/spec.md | 33 ++ .../specs/fleet-routing/spec.md | 160 ++++++ .../specs/spinloop-files/spec.md | 98 ++++ .../tasks.md | 85 +++ openspec/specs/fleet-client/spec.md | 57 +- openspec/specs/fleet-config/spec.md | 65 ++- openspec/specs/fleet-routing/spec.md | 176 ++++--- openspec/specs/spinloop-files/spec.md | 61 +-- 39 files changed, 1345 insertions(+), 808 deletions(-) create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/.openspec.yaml create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/design.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/proposal.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-client/spec.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-config/spec.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-routing/spec.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/spinloop-files/spec.md create mode 100644 openspec/changes/archive/2026-09-12-remove-fleet-instruction/tasks.md diff --git a/README.md b/README.md index 95e4ffca..6abd90f2 100644 --- a/README.md +++ b/README.md @@ -111,9 +111,9 @@ spinloop harness # launch the agent, now running it ``` opencode, Pi and lucinate are all supported, chosen when you launch rather than -written into the file. A `Spinloop` naming a `FLEET` routes the launch to a node -that already has the model — or can load it — so the machine you are sitting at -needs no addresses of its own. +written into the file. A launch routed through a [fleet file](docs/commands/fleet.md) +goes to a node that already has the model — or can load it — so the machine you +are sitting at needs no addresses of its own. Not serving it yourself? The same commands point an agent at a hosted model: `spinloop add -p openrouter -m deepseek/deepseek-v4-flash`, then @@ -366,9 +366,8 @@ MODEL deepseek/deepseek-v4-pro # the provider-native model ref ALIAS deepseek # optional; friendly name for the model CONTEXT 128k # optional; context window OUTPUT 32k # optional; max output tokens -PARALLEL 2 # optional; concurrent slots when serving -BASEURL https://gateway/v1 # optional; API base URL override -FLEET ./fleet.yaml # optional; route the launch to a node + PARALLEL 2 # optional; concurrent slots when serving + BASEURL https://gateway/v1 # optional; API base URL override ``` ```sh @@ -382,11 +381,11 @@ spinloop export > Spinloop # capture your current setup as a Spinloop A `Spinloop` describes one provider selection and applies exactly like the equivalent `add`. The full keyword set is `PROVIDER`, `MODEL`, `ALIAS`, -`CONTEXT`, `OUTPUT`, `PARALLEL`, `BASEURL`, `PRESET`, `REMOTE`, `FLEET` and -`ENV` — `FLEET` and `REMOTE` are mutually exclusive, being two different answers -to where the model runs. Full syntax is in [`docs/spinloop-file.md`](docs/spinloop-file.md), -and ready-to-use examples live under [`examples/`](examples/), including -[fetching one from a URL](examples/remote-spinloop/). +`CONTEXT`, `OUTPUT`, `PARALLEL`, `BASEURL`, `PRESET`, `REMOTE` and `ENV`. +Routing a launch through a fleet is a launch concern, not a Spinloop field — see +the [fleet file](docs/commands/fleet.md). Full syntax is in +[`docs/spinloop-file.md`](docs/spinloop-file.md), and ready-to-use examples live +under [`examples/`](examples/), including [fetching one from a URL](examples/remote-spinloop/). ## Aliases @@ -566,14 +565,16 @@ thing for: which machine is doing nothing? #### Launching against the fleet -A fleet is also where `spinloop harness` sends the agent. A Spinloop naming a -`FLEET` picks a node and launches against its engine, so the machine you are -sitting at needs no addresses of its own: +A fleet is also where `spinloop harness` sends the agent: a launch routed +through a fleet file picks a node and launches against its engine, so the +machine you are sitting at needs no addresses of its own. The fleet file comes +from `--fleet`, or from the `./fleet.yaml` in the working directory when the +Spinloop is not named: ```sh -spinloop harness my-spinloop # picks a node, launches the agent against it -spinloop harness --fleet f.yaml # overrides the Spinloop's FLEET -spinloop fleet route my-spinloop # which node would I get? (launches nothing) +spinloop harness -O -f f.yaml # valueless -O wears ./Spinloop; routes through f.yaml +spinloop harness --fleet f.yaml # name the fleet file explicitly +spinloop fleet route -f f.yaml # which node would I get? (launches nothing) ``` The agent comes up talking to the node it picked — its address arrives as diff --git a/cmd/spinloop/commands.go b/cmd/spinloop/commands.go index 3b5e4954..522fc4d9 100644 --- a/cmd/spinloop/commands.go +++ b/cmd/spinloop/commands.go @@ -178,6 +178,12 @@ exits. Honours -H/--harness and SPINLOOP_HARNESS.`, } } + // A named Spinloop — the flag's value, a leading positional, or the + // alias SPINLOOP_ALIAS names — travels to its fleet only by flag, so + // a fleet.yaml in the working directory is not picked up for it. A + // valueless --spinloop wears the default Spinloop and is not named. + route.spinloopNamed = spinloopPath.path != "" || spinloopAliasInForce() + // A .env beside the applied Spinloop is where its keys live, so the // launched agent is given the same ones. Without a Spinloop there is // no such file and only the environment (plus any provider key @@ -210,7 +216,7 @@ exits. Honours -H/--harness and SPINLOOP_HARNESS.`, // path readSpinloop resolves as SPINLOOP_ALIAS > ./Spinloop. fs.Lookup("spinloop").NoOptDefVal = "true" fs.StringVar(&providers, "providers", "", "path to a providers.yaml override") - fs.StringVarP(&route.fleetPath, "fleet", "f", "", "route through this fleet file (overrides the Spinloop's FLEET)") + fs.StringVarP(&route.fleetPath, "fleet", "f", "", "route through this fleet file (default: ./fleet.yaml, when the Spinloop is not named)") fs.StringVar(&route.node, "node", "", "pin the launch to this fleet node") fs.StringVar(&route.prefer, "prefer", "", "rank fleet nodes by `idle` or `active` (overrides the fleet file)") fs.BoolVar(&route.noWake, "no-wake", false, "fail rather than starting an engine on an idle fleet node") diff --git a/cmd/spinloop/fleet.go b/cmd/spinloop/fleet.go index 64c98b41..9abf3382 100644 --- a/cmd/spinloop/fleet.go +++ b/cmd/spinloop/fleet.go @@ -770,29 +770,22 @@ func runFleetRoute(path, node, prefer string, args []string) error { return err } - target, fromFlag := path, true - if target == "" { - target, fromFlag = sel.Fleet, false + // The fleet file: --fleet/-f when given, otherwise the fleet.yaml in the + // working directory when the Spinloop was not named explicitly — no + // positional and no SPINLOOP_ALIAS, the default Spinloop standing in for + // the user. A named Spinloop routes only by flag, as a launch does. + target := path + if target == "" && spinloopPath == "" && !spinloopAliasInForce() { + if _, err := os.Stat(fleet.DefaultFile); err == nil { + target = fleet.DefaultFile + } } if target == "" { return fmt.Errorf( - "%s names no FLEET: add one, or pass --fleet to say which fleet to route through", + "no fleet file to route %s through: pass --fleet to say which fleet to route through", resolvedPath) } - if isEndpoint(target) { - // The endpoint has already done the choosing: no fleet file to read, - // no node to query, nothing to start. Say where a launch would point - // the agent. - if sel.BaseURL != "" { - fmt.Printf("This Spinloop pins BASEURL %s, so a launch would not route at all.\n", sel.BaseURL) - return nil - } - fmt.Printf("Spinloop: %s\nFleet: %s (an endpoint, not a fleet file)\n\n", resolvedPath, target) - fmt.Printf("The endpoint has already chosen: a launch would point the agent at %s.\n", endpointBaseURL(target)) - fmt.Println("No node is queried, and nothing is started.") - return nil - } - cfg, err := fleet.Resolve(resolveFleetPath(target, fromFlag, resolvedPath)) + cfg, err := fleet.Resolve(target) if err != nil { return err } @@ -814,8 +807,8 @@ func runFleetRoute(path, node, prefer string, args []string) error { return nil } if gw, ok := cfg.GatewaySection(); ok { - // The file names a gateway: as for an endpoint FLEET, the choosing is - // already done — no node is queried, and nothing is started. + // The file names a gateway: the choosing is already done — no node is + // queried, and nothing is started. fmt.Printf("The fleet file names a gateway: a launch would point the agent at %s.\n", endpointBaseURL(gw.URL)) fmt.Println("No node is queried, and nothing is started.") return nil @@ -855,8 +848,8 @@ func runFleetRoute(path, node, prefer string, args []string) error { func cmdFleetRoute(args []string) error { return execCmd(fleetRouteCmd(), args) } // fleetHarnessCmd configures the active harness for the fleet file and -// launches it: the fleet-level form of a fleet-routed launch, in which the -// fleet file comes from the command rather than from a Spinloop's FLEET. +// launches it: the fleet-level form of a launch routed through a fleet, in +// which the fleet file comes from the command, never from the Spinloop. func fleetHarnessCmd() *cobra.Command { var spinloopPath spinloopPathFlag var fleetPath, node, prefer, harnessName string @@ -866,12 +859,11 @@ func fleetHarnessCmd() *cobra.Command { Use: "harness", Short: "configure the active harness for this fleet and launch it", Long: `configures the active harness for the fleet in the fleet file and launches it: -the fleet-level form of a fleet-routed launch. The Spinloop is taken the way -spinloop harness takes one — a leading alias or path, -O/--spinloop, or the -Spinloop beside the fleet file — and the fleet file from -f/--fleet, defaulting -to the fleet.yaml beside it. Where no -f is given, a Spinloop's FLEET — a file -or an endpoint — is used, exactly as --fleet overrides an instruction on -spinloop harness; a Spinloop's pinned BASEURL is not routed, as on the launch.`, +the fleet-level form of a launch routed through a fleet. The Spinloop is taken +the way spinloop harness takes one — a leading alias or path, -O/--spinloop, or +the Spinloop beside the fleet file — and the fleet file from -f/--fleet, +defaulting to the fleet.yaml beside it: the fleet comes from the command, never +from the Spinloop. A Spinloop's pinned BASEURL is not routed, as on the launch.`, Args: cobra.MaximumNArgs(1), SilenceErrors: true, SilenceUsage: true, @@ -934,10 +926,10 @@ func runFleetHarness(sp spinloopPathFlag, fleetPath, node, prefer, harnessName s return err } - // The fleet file: -f when given, the Spinloop's FLEET when -f is not, and - // the fleet.yaml beside it when neither is — the launch's own precedence, - // with the default the launch does not have, because routing is what this - // command exists to do. + // The fleet file: -f when given, the fleet.yaml beside the command when + // not — the fleet comes from the command, never from the Spinloop, and + // routing is what this command exists to do, so there is no unrouted + // fall-through. route := routeOptions{ fleetPath: fleetPath, node: node, @@ -945,7 +937,7 @@ func runFleetHarness(sp spinloopPathFlag, fleetPath, node, prefer, harnessName s noWake: noWake, wakeTimeout: wakeTimeout, } - if route.fleetTarget(sel) == "" { + if route.fleetPath == "" { route.fleetPath = fleet.DefaultFile } diff --git a/cmd/spinloop/fleet_harness_test.go b/cmd/spinloop/fleet_harness_test.go index ff265c58..c18c4b4a 100644 --- a/cmd/spinloop/fleet_harness_test.go +++ b/cmd/spinloop/fleet_harness_test.go @@ -99,9 +99,9 @@ func TestCmdFleetHarnessRoutesToANode(t *testing.T) { } } -// -f beats the Spinloop's FLEET: the section in the named file is where the -// agent is pointed, not the endpoint the Spinloop names. -func TestCmdFleetHarnessFileFlagBeatsTheSpinloopsFleet(t *testing.T) { +// -f beats the fleet.yaml beside the command: the section in the named file is +// where the agent is pointed, not the one in the working directory. +func TestCmdFleetHarnessFileFlagBeatsTheCwdFleet(t *testing.T) { isolateConfig(t) t.Setenv("OPENAI_API_KEY", "gw-token") t.Setenv("OPENAI_BASE_URL", "") @@ -111,9 +111,11 @@ func TestCmdFleetHarnessFileFlagBeatsTheSpinloopsFleet(t *testing.T) { fleetPath := fleetFileIn(t, t.TempDir(), "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\ngateway:\n url: http://b.internal:4000\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://a.internal:4000") + fleetHarnessDir(t, + "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\ngateway:\n url: http://a.internal:4000\n", + "PROVIDER llamacpp\nMODEL qwen3-27b\n") captureStdout(t, func() { - if err := cmdFleetHarness([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { + if err := cmdFleetHarness([]string{"-f", fleetPath}); err != nil { t.Fatalf("cmdFleetHarness: %v", err) } }) @@ -126,31 +128,6 @@ func TestCmdFleetHarnessFileFlagBeatsTheSpinloopsFleet(t *testing.T) { } } -// With no -f, the Spinloop's FLEET — an endpoint here — is used, exactly as -// --fleet overrides an instruction on spinloop harness. -func TestCmdFleetHarnessSpinloopFleetIsUsedWithoutAFlag(t *testing.T) { - isolateConfig(t) - t.Setenv("OPENAI_API_KEY", "gw-token") - t.Setenv("OPENAI_BASE_URL", "") - argsFile := filepath.Join(t.TempDir(), "args") - envFile := filepath.Join(t.TempDir(), "env") - stubHarnessBinaryWithEnv(t, argsFile, envFile) - - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://a.internal:4000") - captureStdout(t, func() { - if err := cmdFleetHarness([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { - t.Fatalf("cmdFleetHarness: %v", err) - } - }) - data, err := os.ReadFile(envFile) - if err != nil { - t.Fatal(err) - } - if !strings.Contains(string(data), "BASE=http://a.internal:4000/v1") { - t.Errorf("without -f the Spinloop's FLEET should be routed through, got:\n%s", data) - } -} - // A directory holding no Spinloop, with none given: the command fails saying a // launch needs a Spinloop to know which model to route, and launches nothing. func TestCmdFleetHarnessNeedsASpinloop(t *testing.T) { diff --git a/cmd/spinloop/gateway.go b/cmd/spinloop/gateway.go index 8a1b551d..a974465b 100644 --- a/cmd/spinloop/gateway.go +++ b/cmd/spinloop/gateway.go @@ -39,9 +39,11 @@ asks for, holding the request until the engine answers. It needs the same environment a machine running spinloop fleet start would: the tokens the fleet file names, set here or in the .env beside it. -A Spinloop points an agent at it with a FLEET that names its address: +A fleet file points an agent at it with a gateway section that names its +address: - FLEET http://gateway.internal:4000 + gateway: + url: http://gateway.internal:4000 The agent then needs only the gateway's token, as OPENAI_API_KEY.`, Args: cobra.NoArgs, @@ -124,8 +126,8 @@ func gatewayListenAddr(listen string, listenExplicit, loopback bool) (string, er // newGatewayServer resolves the fleet file and the gateway's token, checks the // file's token references the way a startup must, opens the listener, and -// prints the address a Spinloop names in its FLEET. Everything that can fail -// without serving fails here, before a listener exists. +// prints the address a fleet file's gateway section names. Everything that can +// fail without serving fails here, before a listener exists. func newGatewayServer(fleetPath, listen, apiToken, apiTokenFile string) (*http.Server, net.Listener, error) { cfg, err := fleet.Resolve(fleetPath) if err != nil { @@ -181,12 +183,12 @@ func newGatewayServer(fleetPath, listen, apiToken, apiTokenFile string) (*http.S } fmt.Printf("Gateway for %s is listening on %s\n", cfg.Path, ln.Addr().String()) - fmt.Printf("Name %s in a Spinloop's FLEET\n\n", fleetURL(ln.Addr().String())) + fmt.Printf("Name %s in the fleet file's gateway section\n\n", fleetURL(ln.Addr().String())) return &http.Server{Handler: h}, ln, nil } -// fleetURL turns the address the gateway listens on into the value a Spinloop -// names in its FLEET: an http URL the agent's machine can reach. The host it +// fleetURL turns the address the gateway listens on into the URL a fleet file's +// gateway section names: an http URL the agent's machine can reach. The host it // can know is the one it was told to bind; for a wildcard bind the host is // whatever this machine is called from the other side, which only the operator // knows. diff --git a/cmd/spinloop/gateway_test.go b/cmd/spinloop/gateway_test.go index 8bf82891..a1efda1f 100644 --- a/cmd/spinloop/gateway_test.go +++ b/cmd/spinloop/gateway_test.go @@ -15,7 +15,7 @@ import ( ) // The gateway starts with the fleet file it serves and answers, and its -// banner names the address a Spinloop puts in its FLEET. +// banner names the address a fleet file's gateway section puts in. func TestGatewayStartsAndAnswers(t *testing.T) { isolateConfig(t) t.Setenv("SPINLOOP_API_TOKEN", "") @@ -38,7 +38,7 @@ func TestGatewayStartsAndAnswers(t *testing.T) { port := ln.Addr().(*net.TCPAddr).Port if !strings.Contains(out, "http://127.0.0.1:"+fmt.Sprint(port)) { - t.Errorf("the banner should name the address to put in a Spinloop's FLEET, got:\n%s", out) + t.Errorf("the banner should name the address to put in the fleet file's gateway section, got:\n%s", out) } client := &http.Client{Timeout: 2 * time.Second} @@ -50,8 +50,8 @@ func TestGatewayStartsAndAnswers(t *testing.T) { t.Errorf("health answered %d", resp.StatusCode) } - // The printed address is the one a Spinloop's FLEET names: asking it for - // the fleet's models gets the running node's. + // The printed address is the one the fleet file's gateway section names: + // asking it for the fleet's models gets the running node's. resp, err = client.Get(fmt.Sprintf("http://127.0.0.1:%d/v1/models", port)) if err != nil { t.Fatalf("models: %v", err) diff --git a/cmd/spinloop/harness_test.go b/cmd/spinloop/harness_test.go index 8bcbc865..fe449697 100644 --- a/cmd/spinloop/harness_test.go +++ b/cmd/spinloop/harness_test.go @@ -1127,9 +1127,10 @@ func TestHarness_LucinateInjectsKeyAtLaunch(t *testing.T) { } } -// A routed launch gives lucinate the fleet's key, not the provider's: the -// endpoint's token is the one the agent will authenticate with. -func TestHarness_LucinateCarriesTheFleetKeyAtLaunch(t *testing.T) { +// A launch routed at a fleet's gateway gives lucinate the gateway's key, not +// the provider's: the gateway's token is the one the agent will authenticate +// with. +func TestHarness_LucinateCarriesTheGatewayKeyAtLaunch(t *testing.T) { isolateConfig(t) t.Setenv("SPINLOOP_HARNESS", "lucinate") t.Setenv("OPENAI_API_KEY", "gw-token") @@ -1144,12 +1145,13 @@ func TestHarness_LucinateCarriesTheFleetKeyAtLaunch(t *testing.T) { } t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) + fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") spinloopDir := t.TempDir() mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), - "PROVIDER openrouter\nMODEL deepseek/deepseek-v4-flash\nFLEET http://gw.internal:4000\n") + "PROVIDER openrouter\nMODEL deepseek/deepseek-v4-flash\n") captureStdout(t, func() { - if err := cmdHarness([]string{"--spinloop=" + spinloopDir}); err != nil { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "-f", fleetPath}); err != nil { t.Fatalf("cmdHarness --spinloop: %v", err) } }) @@ -1159,10 +1161,10 @@ func TestHarness_LucinateCarriesTheFleetKeyAtLaunch(t *testing.T) { t.Fatalf("lucinate was not launched: %v", err) } if !strings.Contains(string(got), "LUCINATE_OPENAI_API_KEY=gw-token") { - t.Errorf("the fleet's key should reach lucinate in preference to the provider's:\n%s", got) + t.Errorf("the gateway's key should reach lucinate in preference to the provider's:\n%s", got) } if !strings.Contains(string(got), "OPENAI_BASE_URL=http://gw.internal:4000/v1") { - t.Errorf("the endpoint's address should reach lucinate:\n%s", got) + t.Errorf("the gateway's address should reach lucinate:\n%s", got) } } diff --git a/cmd/spinloop/main.go b/cmd/spinloop/main.go index e9c24e41..161bc8cc 100644 --- a/cmd/spinloop/main.go +++ b/cmd/spinloop/main.go @@ -363,6 +363,14 @@ func readSpinloop(usage, path string) (spinloop.Selection, string, error) { // Spinloop path but was given none. const spinloopAliasEnv = "SPINLOOP_ALIAS" +// spinloopAliasInForce reports whether SPINLOOP_ALIAS will decide the Spinloop +// a valueless --spinloop wears. The variable is the user's own naming of a +// Spinloop, so for routing it counts as one named explicitly: a fleet.yaml in +// the working directory is not picked up for it. +func spinloopAliasInForce() bool { + return cliViper.GetString("alias") != "" +} + // spinloopFromEnv resolves SPINLOOP_ALIAS, returning the name it holds alongside // the Spinloop it points at, or two empty strings when it is unset or empty. // @@ -1214,16 +1222,16 @@ func applyRoutedSpinloop(sel spinloop.Selection, path string, providers string, resolve = fleetLaunchResolver(resolve, choice.APIKey) } if choice != nil && choice.Gateway { - // A FLEET naming an endpoint — or a fleet file naming a gateway — - // authenticates with a token the client holds itself, resolved the way - // a key is resolved elsewhere: an ENV instruction, else the process - // environment, else the .env beside the Spinloop. The variable is the - // one the section names, or the endpoint's default where the section - // names none. Set nowhere, the launch cannot authenticate, and a - // gateway that refuses every request is not one to point an agent at. + // A fleet file naming a gateway authenticates with a token the client + // holds itself, resolved the way a key is resolved elsewhere: an ENV + // instruction, else the process environment, else the .env beside the + // Spinloop. The variable is the one the section names, or the + // default where the section names none. Set nowhere, the launch cannot + // authenticate, and a gateway that refuses every request is not one to + // point an agent at. env := choice.GatewayTokenEnv if env == "" { - env = remoteAPIKeyEnv + env = fleet.DefaultGatewayTokenEnv } key := localKeyUnder(sel, localResolve, env) if key == "" { @@ -1313,7 +1321,7 @@ func localKeyUnder(sel spinloop.Selection, resolve func(string) string, name str } // localKey resolves under the remote API key's variable — the one a REMOTE -// endpoint and a FLEET endpoint authenticate under. +// endpoint authenticates under. func localKey(sel spinloop.Selection, resolve func(string) string) string { return localKeyUnder(sel, resolve, remoteAPIKeyEnv) } diff --git a/cmd/spinloop/route.go b/cmd/spinloop/route.go index 4b4dd48f..c686f3fe 100644 --- a/cmd/spinloop/route.go +++ b/cmd/spinloop/route.go @@ -1,8 +1,8 @@ -// Routing a launch through the fleet: turning a Spinloop's FLEET into the node -// the agent talks to. It sits beside the remote path in main.go — both answer -// "where does this agent send its requests", one by asking a control plane and -// one by choosing a machine — and it runs before the apply for the same reason -// the remote fetch does: a failed route must leave the harness config alone. +// Routing a launch through the fleet: choosing the node the agent talks to. It +// sits beside the remote path in main.go — both answer "where does this agent +// send its requests", one by asking a control plane and one by choosing a +// machine — and it runs before the apply for the same reason the remote fetch +// does: a failed route must leave the harness config alone. package main @@ -12,7 +12,6 @@ import ( "fmt" "net/url" "os" - "path/filepath" "strings" "time" @@ -23,8 +22,12 @@ import ( // routeOptions is what the launch flags say about routing. They are inert // unless something names a fleet. type routeOptions struct { - // fleetPath overrides the Spinloop's FLEET. + // fleetPath is the fleet file to route through when given. fleetPath string + // spinloopNamed says the worn Spinloop was named by the user — a + // positional argument, a --spinloop value, or SPINLOOP_ALIAS — so a + // fleet.yaml in the working directory is not picked up for it. + spinloopNamed bool // node pins the selection to one node. node string // prefer overrides the fleet file's activity preference. @@ -35,13 +38,22 @@ type routeOptions struct { wakeTimeout time.Duration } -// fleetTarget is the fleet a launch routes through: the flag when given, -// otherwise the Spinloop's own FLEET. Empty means this launch does not route. -func (o routeOptions) fleetTarget(sel spinloop.Selection) string { +// fleetFile is the fleet file a launch routes through: the flag's path when +// given, otherwise the fleet.yaml in the working directory when the worn +// Spinloop was not named explicitly — a named Spinloop travels to its fleet +// only by flag, and a working directory with no fleet.yaml gives no fleet at +// all. Both sources are working-directory paths. +func (o routeOptions) fleetFile() string { if o.fleetPath != "" { return o.fleetPath } - return sel.Fleet + if o.spinloopNamed { + return "" + } + if _, err := os.Stat(fleet.DefaultFile); err == nil { + return fleet.DefaultFile + } + return "" } // routeThroughFleet chooses the node an agent will talk to, waking one when @@ -53,7 +65,7 @@ func (o routeOptions) fleetTarget(sel spinloop.Selection) string { // way. Saying so matters — silently selecting a node whose address is then // discarded would be a puzzle rather than a behaviour. func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOptions) (*fleet.Choice, error) { - target := opts.fleetTarget(sel) + target := opts.fleetFile() if target == "" { return nil, nil } @@ -62,28 +74,14 @@ func routeThroughFleet(sel spinloop.Selection, spinloopPath string, opts routeOp "Not routing through %s: this Spinloop pins BASEURL %s.\n", target, sel.BaseURL) return nil, nil } - // A FLEET naming a URL is the gateway shape: it has already done the - // choosing, so there is no fleet file to read and no node to contact. The - // token is not resolved here: applyBeforeLaunch resolves it through the - // same chain the launch uses, so a missing value fails before anything is - // written. - if isEndpoint(target) { - return &fleet.Choice{ - Gateway: true, - BaseURL: endpointBaseURL(target), - Reason: "FLEET names an endpoint", - }, nil - } - - cfg, err := fleet.Resolve(resolveFleetPath(target, opts.fleetPath != "", spinloopPath)) + cfg, err := fleet.Resolve(target) if err != nil { return nil, err } - // A fleet file that names a gateway routes the way an endpoint FLEET - // does: the gateway has already done the choosing, so there is no node to - // contact and nothing to wake. The token is not resolved here: the launch - // resolves it through the same chain, under the variable the section - // names. + // A fleet file that names a gateway routes at the gateway rather than a + // node: it has already done the choosing, so there is no node to contact + // and nothing to wake. The token is not resolved here: the launch resolves + // it through the same chain, under the variable the section names. if gw, ok := cfg.GatewaySection(); ok { choice := &fleet.Choice{ Gateway: true, @@ -170,10 +168,10 @@ func announceChoice(c *fleet.Choice) { fmt.Fprintf(os.Stderr, "Using %s at %s — %s\n", c.Node.Name, c.BaseURL, c.Reason) } -// endpointBaseURL is the address a launch gives an agent for a FLEET that names -// an endpoint: the value as given when it carries a path, and the OpenAI- -// compatible prefix added when it does not, so FLEET http://gw:4000 points the -// agent at http://gw:4000/v1. +// endpointBaseURL is the address a launch gives an agent for a gateway: the +// value as given when it carries a path, and the OpenAI-compatible prefix +// added when it does not, so a gateway at http://gw:4000 points the agent at +// http://gw:4000/v1. func endpointBaseURL(target string) string { u, err := url.Parse(target) if err != nil || (u.Path != "" && u.Path != "/") { @@ -181,23 +179,3 @@ func endpointBaseURL(target string) string { } return strings.TrimRight(target, "/") + "/v1" } - -// resolveFleetPath resolves a fleet file's path. A relative FLEET is resolved -// against the Spinloop that names it, the same rule PRESET and REMOTE follow — an -// Spinloop and the fleet beside it travel together, and resolving against the -// working directory would make the same Spinloop work from one directory and not -// another. A path given on the command line is the user's own, so it is -// resolved against the working directory as any other argument would be. -func resolveFleetPath(target string, fromFlag bool, spinloopPath string) string { - if fromFlag || spinloopPath == "" || filepath.IsAbs(target) { - return target - } - return filepath.Join(filepath.Dir(spinloopPath), target) -} - -// isEndpoint reports whether a FLEET value names an endpoint rather than a -// file. It mirrors spinloop.Selection.FleetIsEndpoint for a value that came from -// a flag rather than a Spinloop. -func isEndpoint(target string) bool { - return strings.Contains(target, "://") -} diff --git a/cmd/spinloop/route_test.go b/cmd/spinloop/route_test.go index 2310232f..1cae94b7 100644 --- a/cmd/spinloop/route_test.go +++ b/cmd/spinloop/route_test.go @@ -77,15 +77,13 @@ func fleetFileIn(t *testing.T, dir, body string) string { return path } -// routedSpinloop writes a Spinloop naming a fleet, and returns its directory. -func routedSpinloop(t *testing.T, model, fleetPath string) string { +// routedSpinloop writes a plain Spinloop and returns its directory. The fleet +// it routes through is a launch concern, not a Spinloop field: the tests name +// it with the --fleet flag or a routeOptions fleetPath. +func routedSpinloop(t *testing.T, model string) string { t.Helper() dir := t.TempDir() - body := "PROVIDER llamacpp\nMODEL " + model + "\n" - if fleetPath != "" { - body += "FLEET " + fleetPath + "\n" - } - mustWrite(t, filepath.Join(dir, "Spinloop"), body) + mustWrite(t, filepath.Join(dir, "Spinloop"), "PROVIDER llamacpp\nMODEL "+model+"\n") return dir } @@ -93,21 +91,19 @@ func TestRouteChoosesARunningNode(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 300) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } - var choice *struct{} - _ = choice stderr := captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatalf("routing failed: %v", err) } if c == nil { - t.Fatal("a Spinloop naming a FLEET should route") + t.Fatal("a launch naming a fleet should route") } want := "http://127.0.0.1:" + strconv.Itoa(node.enginePort) + "/v1" if c.BaseURL != want { @@ -125,26 +121,35 @@ func TestRouteChoosesARunningNode(t *testing.T) { } } -// A Spinloop naming no FLEET, with no --fleet, contacts nothing. +// No fleet by flag and no fleet.yaml in the working directory: the launch +// routes nowhere and contacts nothing. func TestNoFleetDoesNotRoute(t *testing.T) { - spinloopDir := routedSpinloop(t, "qwen3-27b", "") + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } + // An empty working directory holds no fleet.yaml, so an unnamed Spinloop + // has no fleet to find. + t.Chdir(t.TempDir()) choice, err := routeThroughFleet(sel, path, routeOptions{}) if err != nil || choice != nil { t.Errorf("choice = %+v, err = %v; want no routing at all", choice, err) } } -// The flag overrides the Spinloop's own FLEET. -func TestFleetFlagOverridesTheInstruction(t *testing.T) { - node := newRoutableNode(t, "qwen3-27b", true, 10) - dir := t.TempDir() - flagFleet := fleetFileIn(t, dir, "nodes:\n"+node.entry("from-flag")) - spinloopDir := routedSpinloop(t, "qwen3-27b", filepath.Join(dir, "nonexistent.yaml")) +// The --fleet flag beats the fleet.yaml in the working directory: when both +// are present the flag's file is the one read. +func TestFleetFlagBeatsTheCwdFleet(t *testing.T) { + flagNode := newRoutableNode(t, "qwen3-27b", true, 10) + cwdNode := newRoutableNode(t, "qwen3-27b", true, 10) + flagDir := t.TempDir() + flagFleet := fleetFileIn(t, flagDir, "nodes:\n"+flagNode.entry("from-flag")) + cwdDir := t.TempDir() + fleetFileIn(t, cwdDir, "nodes:\n"+cwdNode.entry("from-cwd")) + t.Chdir(cwdDir) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) @@ -160,16 +165,13 @@ func TestFleetFlagOverridesTheInstruction(t *testing.T) { }) } -// The short form is the flag: -f names the fleet a launch routes through, -// overriding the Spinloop's own FLEET. +// The short form is the flag: -f names the fleet a launch routes through. func TestHarnessFleetFlagShortForm(t *testing.T) { isolateConfig(t) node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() flagFleet := fleetFileIn(t, dir, "nodes:\n"+node.entry("from-flag")) - // The Spinloop names a fleet that does not exist, so a launch that - // succeeds has parsed -f as the fleet file. - spinloopDir := routedSpinloop(t, "qwen3-27b", filepath.Join(dir, "nonexistent.yaml")) + spinloopDir := routedSpinloop(t, "qwen3-27b") argsFile := filepath.Join(t.TempDir(), "args") stubHarnessBinary(t, "opencode", argsFile) @@ -188,31 +190,29 @@ func TestHarnessFleetFlagShortForm(t *testing.T) { } } -// A pinned BASEURL is the explicit answer, so nothing is selected. +// A pinned BASEURL is the explicit answer, so nothing is selected even though a +// fleet is in force. func TestPinnedBaseURLSkipsRouting(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) spinloopDir := t.TempDir() mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), - "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\nFLEET "+fleetPath+"\n") + "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\n") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } - var choice any stderr := captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatal(err) } - choice = c if c != nil { t.Errorf("a pinned BASEURL should not be routed, got %+v", c) } }) - _ = choice if !strings.Contains(stderr, "Not routing") || !strings.Contains(stderr, "pinned") { t.Errorf("spinloop should say it is not routing, got:\n%s", stderr) } @@ -223,14 +223,14 @@ func TestNoWakeFailsWithTheNodeTable(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - _, err := routeThroughFleet(sel, path, routeOptions{noWake: true}) + _, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath, noWake: true}) if err == nil { t.Fatal("--no-wake with nothing serving should fail") } @@ -245,17 +245,13 @@ func TestNoWakeFailsWithTheNodeTable(t *testing.T) { } } -// routedSpinloopWith writes a routed Spinloop carrying extra instructions, for the +// routedSpinloopWith writes a Spinloop carrying extra instructions, for the // cases where what the Spinloop says about the *engine* decides whether a wake // can happen at all. -func routedSpinloopWith(t *testing.T, model, fleetPath, extra string) string { +func routedSpinloopWith(t *testing.T, model, extra string) string { t.Helper() dir := t.TempDir() - body := "PROVIDER llamacpp\nMODEL " + model + "\n" + extra - if fleetPath != "" { - body += "FLEET " + fleetPath + "\n" - } - mustWrite(t, filepath.Join(dir, "Spinloop"), body) + mustWrite(t, filepath.Join(dir, "Spinloop"), "PROVIDER llamacpp\nMODEL "+model+"\n"+extra) return dir } @@ -266,14 +262,14 @@ func TestWakeRefusesAnUnusableParallel(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloopWith(t, "qwen3-27b", fleetPath, "PARALLEL 0\n") + spinloopDir := routedSpinloopWith(t, "qwen3-27b", "PARALLEL 0\n") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - _, err := routeThroughFleet(sel, path, routeOptions{}) + _, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err == nil { t.Fatal("a wake with an unusable PARALLEL should fail") } @@ -297,14 +293,14 @@ func TestRoutingToARunningNodeToleratesAnUnusableParallel(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 300) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloopWith(t, "qwen3-27b", fleetPath, "PARALLEL 0\n") + spinloopDir := routedSpinloopWith(t, "qwen3-27b", "PARALLEL 0\n") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatalf("a node already serving the model should still be chosen: %v", err) } @@ -314,70 +310,6 @@ func TestRoutingToARunningNodeToleratesAnUnusableParallel(t *testing.T) { }) } -// A FLEET naming a URL is the gateway shape: the endpoint has already done the -// choosing, so no fleet file is read, no node is contacted, and the node- -// steering flags are inert. A value with no path gets the OpenAI-compatible -// prefix. -func TestFleetURLYieldsTheEndpoint(t *testing.T) { - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") - sel, path, err := readSpinloop("test", spinloopDir) - if err != nil { - t.Fatal(err) - } - captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{node: "nobody", prefer: "sideways", noWake: true}) - if err != nil { - t.Fatalf("an endpoint FLEET should not consult any node: %v", err) - } - if !c.Gateway { - t.Fatalf("the choice should mark itself as an endpoint, got %+v", c) - } - if c.BaseURL != "http://gateway.internal:4000/v1" { - t.Errorf("an endpoint without a path gets the prefix, got %s", c.BaseURL) - } - }) -} - -func TestFleetURLWithAPathIsUsedAsGiven(t *testing.T) { - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000/proxy/v1") - sel, path, err := readSpinloop("test", spinloopDir) - if err != nil { - t.Fatal(err) - } - captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) - if err != nil { - t.Fatal(err) - } - if !c.Gateway || c.BaseURL != "http://gateway.internal:4000/proxy/v1" { - t.Errorf("an endpoint carrying a path is used as given, got %+v", c) - } - }) -} - -// A pinned BASEURL wins over an endpoint FLEET, as it wins over a fleet file. -func TestPinnedBaseURLBeatsAnEndpointFleet(t *testing.T) { - spinloopDir := t.TempDir() - mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), - "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\nFLEET http://gateway.internal:4000\n") - sel, path, err := readSpinloop("test", spinloopDir) - if err != nil { - t.Fatal(err) - } - stderr := captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) - if err != nil { - t.Fatal(err) - } - if c != nil { - t.Errorf("a pinned BASEURL is not routed, got %+v", c) - } - }) - if !strings.Contains(stderr, "Not routing") { - t.Errorf("spinloop should say it is not routing, got:\n%s", stderr) - } -} - // gatewayFleetFile writes a fleet file naming a gateway whose nodes point at a // port nothing listens on — so a route that consults them fails loudly rather // than passing quietly. @@ -387,18 +319,18 @@ func gatewayFleetFile(t *testing.T, section string) string { "nodes:\n - name: dead\n host: 127.0.0.1\n port: 1\n"+section) } -// A fleet file naming a gateway routes at it the way an endpoint FLEET does: -// no node is consulted — the dead node below would fail a route that tried. +// A fleet file naming a gateway routes at it: no node is consulted — the dead +// node below would fail a route that tried. func TestRouteToFileNamingAGateway(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n tokenEnv: GW_TOKEN\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } stderr := captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{node: "nobody", prefer: "sideways", noWake: true}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath, node: "nobody", prefer: "sideways", noWake: true}) if err != nil { t.Fatalf("a file naming a gateway should not consult any node: %v", err) } @@ -418,17 +350,17 @@ func TestRouteToFileNamingAGateway(t *testing.T) { } } -// A section url carrying a path is used as given, like an endpoint's. +// A section url carrying a path is used as given. func TestRouteToFileNamingAGatewayWithAPath(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000/proxy/v1\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatal(err) } @@ -438,18 +370,18 @@ func TestRouteToFileNamingAGatewayWithAPath(t *testing.T) { }) } -// A pinned BASEURL wins over a gateway section, as it wins over an endpoint. +// A pinned BASEURL wins over a gateway section, as it wins over a node fleet. func TestPinnedBaseURLBeatsAGatewaySection(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") spinloopDir := t.TempDir() mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), - "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\nFLEET "+fleetPath+"\n") + "PROVIDER llamacpp\nMODEL qwen3-27b\nBASEURL http://pinned:9999/v1\n") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } stderr := captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatal(err) } @@ -477,68 +409,9 @@ func stubHarnessBinaryWithEnv(t *testing.T, argsFile, envFile string) { t.Setenv("PATH", dir+string(os.PathListSeparator)+os.Getenv("PATH")) } -// A FLEET naming an endpoint points the agent at it: the address gets the -// OpenAI-compatible prefix, and the token is resolved from the client's -// environment the way a key is resolved elsewhere. -func TestLaunchWithEndpointFleetPointsTheAgentAtTheGateway(t *testing.T) { - isolateConfig(t) - t.Setenv("OPENAI_API_KEY", "gw-token") - t.Setenv("OPENAI_BASE_URL", "") - argsFile := filepath.Join(t.TempDir(), "args") - envFile := filepath.Join(t.TempDir(), "env") - stubHarnessBinaryWithEnv(t, argsFile, envFile) - - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") - captureStdout(t, func() { - if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { - t.Fatalf("cmdHarness: %v", err) - } - }) - if _, err := os.ReadFile(argsFile); err != nil { - t.Fatalf("harness was not launched: %v", err) - } - data, err := os.ReadFile(envFile) - if err != nil { - t.Fatal(err) - } - out := string(data) - if !strings.Contains(out, "BASE=http://gateway.internal:4000/v1") { - t.Errorf("the agent's base URL should be the endpoint with the prefix, got:\n%s", out) - } - if !strings.Contains(out, "KEY=gw-token") { - t.Errorf("the agent should carry the gateway's token as its key, got:\n%s", out) - } -} - -// A FLEET naming an endpoint with no token anywhere fails before the agent -// launches and before the harness config is written, naming the variable. -func TestLaunchWithEndpointFleetFailsWithoutAToken(t *testing.T) { - home := isolateConfig(t) - t.Setenv("OPENAI_API_KEY", "") - argsFile := filepath.Join(t.TempDir(), "args") - stubHarnessBinary(t, "opencode", argsFile) - - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gateway.internal:4000") - captureStdout(t, func() { - err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}) - if err == nil { - t.Fatal("a launch that cannot authenticate the endpoint should fail") - } - if !strings.Contains(err.Error(), "OPENAI_API_KEY") { - t.Errorf("the failure should name the variable to set, got:\n%v", err) - } - }) - if _, err := os.ReadFile(argsFile); err == nil { - t.Error("the agent launched without a token to reach the endpoint") - } - if _, err := os.Stat(filepath.Join(home, ".config", "opencode", "opencode.json")); err == nil { - t.Error("the harness config was written for a launch that could not authenticate") - } -} - // A fleet file naming a gateway points the agent at it: the address is the // applied provider's base URL, and the token is resolved under the variable -// the section names, not the endpoint's default. +// the section names, not the default. func TestLaunchWithAGatewaySectionPointsTheAgentAtTheGateway(t *testing.T) { home := isolateConfig(t) t.Setenv("GATEWAY_TOKEN", "gw-token") @@ -550,9 +423,9 @@ func TestLaunchWithAGatewaySectionPointsTheAgentAtTheGateway(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") captureStdout(t, func() { - if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "-f", fleetPath, "--", "run"}); err != nil { t.Fatalf("cmdHarness: %v", err) } }) @@ -579,9 +452,7 @@ func TestLaunchWithAGatewaySectionPointsTheAgentAtTheGateway(t *testing.T) { } } -// A section naming no tokenEnv resolves under the endpoint FLEET's variable, -// so moving a launch from an endpoint to a section changes nothing the client -// has to export. +// A section naming no tokenEnv resolves under the default variable. func TestLaunchWithAGatewaySectionDefaultsToOpenAIKey(t *testing.T) { isolateConfig(t) t.Setenv("OPENAI_API_KEY", "gw-token") @@ -591,9 +462,9 @@ func TestLaunchWithAGatewaySectionDefaultsToOpenAIKey(t *testing.T) { stubHarnessBinaryWithEnv(t, argsFile, envFile) fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") captureStdout(t, func() { - if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "-f", fleetPath, "--", "run"}); err != nil { t.Fatalf("cmdHarness: %v", err) } }) @@ -618,9 +489,9 @@ func TestLaunchWithAGatewaySectionFailsNamingItsVariable(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") captureStdout(t, func() { - err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}) + err := cmdHarness([]string{"--spinloop=" + spinloopDir, "-f", fleetPath, "--", "run"}) if err == nil { t.Fatal("a launch that cannot authenticate the gateway should fail") } @@ -650,10 +521,10 @@ func TestLaunchWithAGatewaySectionTokenFromDotEnv(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n tokenEnv: GATEWAY_TOKEN\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") mustWrite(t, filepath.Join(spinloopDir, ".env"), "GATEWAY_TOKEN=dotenv-token\n") captureStdout(t, func() { - if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "--", "run"}); err != nil { + if err := cmdHarness([]string{"--spinloop=" + spinloopDir, "-f", fleetPath, "--", "run"}); err != nil { t.Fatalf("cmdHarness: %v", err) } }) @@ -676,14 +547,14 @@ func TestWakeOffRefusesNamingTheNode(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "wake: off\nnodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - _, err := routeThroughFleet(sel, path, routeOptions{}) + _, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err == nil { t.Fatal("wake: off with nothing serving should fail") } @@ -702,14 +573,14 @@ func TestUnknownPreferenceIsRefused(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { t.Fatal(err) } captureStderr(t, func() { - if _, err := routeThroughFleet(sel, path, routeOptions{prefer: "sideways"}); err == nil { + if _, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath, prefer: "sideways"}); err == nil { t.Fatal("an unknown preference should fail") } else if !strings.Contains(err.Error(), "idle") || !strings.Contains(err.Error(), "active") { t.Errorf("error should name both values, got: %v", err) @@ -723,7 +594,7 @@ func TestRoutedLaunchEnvironment(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") sel, path, err := readSpinloop("test", spinloopDir) if err != nil { @@ -731,7 +602,7 @@ func TestRoutedLaunchEnvironment(t *testing.T) { } var baseURL string captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{fleetPath: fleetPath}) if err != nil { t.Fatal(err) } @@ -758,7 +629,7 @@ func TestFailedRouteLeavesTheConfigUntouched(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") h, _ := harness.Lookup("opencode") var err error @@ -766,7 +637,7 @@ func TestFailedRouteLeavesTheConfigUntouched(t *testing.T) { captureStdout(t, func() { _, _, _, _, err = applyBeforeLaunch( spinloopPathFlag{set: true, path: spinloopDir}, "", h, nil, - routeOptions{noWake: true}) + routeOptions{noWake: true, fleetPath: fleetPath}) }) }) if err == nil { @@ -784,7 +655,7 @@ func TestRoutedApplyWritesTheChosenBaseURL(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") h, _ := harness.Lookup("opencode") var sel spinloop.Selection @@ -792,7 +663,8 @@ func TestRoutedApplyWritesTheChosenBaseURL(t *testing.T) { captureStderr(t, func() { captureStdout(t, func() { sel, _, _, _, err = applyBeforeLaunch( - spinloopPathFlag{set: true, path: spinloopDir}, "", h, nil, routeOptions{}) + spinloopPathFlag{set: true, path: spinloopDir}, "", h, nil, + routeOptions{fleetPath: fleetPath}) }) }) if err != nil { @@ -809,10 +681,10 @@ func TestCmdFleetRouteExplainsTheChoice(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 42) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "prefer: active\nnodes:\n"+node.entry("gpu-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { t.Fatal(err) } }) @@ -823,35 +695,14 @@ func TestCmdFleetRouteExplainsTheChoice(t *testing.T) { } } -// A FLEET naming an endpoint has already chosen: the route says where a launch -// would point the agent, and queries nothing. -func TestCmdFleetRouteAgainstAnEndpoint(t *testing.T) { - spinloopDir := routedSpinloop(t, "qwen3-27b", "http://gw.internal:4000") - - out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { - t.Fatal(err) - } - }) - for _, want := range []string{ - "gw.internal:4000 (an endpoint, not a fleet file)", - "would point the agent at http://gw.internal:4000/v1", - "nothing is started", - } { - if !strings.Contains(out, want) { - t.Errorf("output should mention %q, got:\n%s", want, out) - } - } -} - -// A fleet file naming a gateway is answered the way an endpoint is: the +// A fleet file naming a gateway is answered without querying a node: the // address is named, and the dead node below proves none is queried. func TestCmdFleetRouteAgainstAFileNamingAGateway(t *testing.T) { fleetPath := gatewayFleetFile(t, "gateway:\n url: http://gw.internal:4000\n") - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { t.Fatal(err) } }) @@ -874,12 +725,12 @@ func TestCmdFleetRoutePreferenceFlagBeatsTheFile(t *testing.T) { dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "prefer: idle\nnodes:\n"+recent.entry("recent")+stale.entry("stale")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") spinloopFile := filepath.Join(spinloopDir, "Spinloop") // The file says idle, so the long-idle node wins. out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{spinloopFile}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, spinloopFile}); err != nil { t.Fatal(err) } }) @@ -889,7 +740,7 @@ func TestCmdFleetRoutePreferenceFlagBeatsTheFile(t *testing.T) { // The flag overrides it, and says so. out = captureStdout(t, func() { - if err := cmdFleetRoute([]string{"--prefer", "active", spinloopFile}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, "--prefer", "active", spinloopFile}); err != nil { t.Fatal(err) } }) @@ -913,10 +764,10 @@ func TestCmdFleetRouteStartsNothing(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "nodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { t.Fatal(err) } }) @@ -941,10 +792,10 @@ func TestCmdFleetRouteWakeOffRefusal(t *testing.T) { node := newRoutableNode(t, "", false, 0) dir := t.TempDir() fleetPath := fleetFileIn(t, dir, "wake: off\nnodes:\n"+node.entry("idle-box")) - spinloopDir := routedSpinloop(t, "qwen3-27b", fleetPath) + spinloopDir := routedSpinloop(t, "qwen3-27b") out := captureStdout(t, func() { - if err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}); err != nil { + if err := cmdFleetRoute([]string{"-f", fleetPath, filepath.Join(spinloopDir, "Spinloop")}); err != nil { t.Fatal(err) } }) @@ -958,9 +809,10 @@ func TestCmdFleetRouteWakeOffRefusal(t *testing.T) { } } -// A Spinloop naming no fleet, with no --fleet, has nothing to report. +// A named Spinloop, with no --fleet, has nothing to route through: the failure +// names the flag that supplies the fleet. func TestCmdFleetRouteNeedsAFleet(t *testing.T) { - spinloopDir := routedSpinloop(t, "qwen3-27b", "") + spinloopDir := routedSpinloop(t, "qwen3-27b") err := cmdFleetRoute([]string{filepath.Join(spinloopDir, "Spinloop")}) if err == nil { t.Fatal("expected a failure") @@ -970,29 +822,171 @@ func TestCmdFleetRouteNeedsAFleet(t *testing.T) { } } -// A relative FLEET is resolved against the Spinloop that names it, as PRESET and -// REMOTE are — otherwise the same Spinloop routes from one directory and not -// another. -func TestRelativeFleetResolvesAgainstTheSpinloop(t *testing.T) { +// Discovery: a fleet.yaml in the working directory is the fleet for a Spinloop +// the user did not name. A named Spinloop — a flag value, a positional, or the +// alias SPINLOOP_ALIAS names — travels to its fleet only by flag, so a +// fleet.yaml beside the working directory is not picked up for it. + +// An explicitly named Spinloop does not pick up the working directory's +// fleet.yaml. +func TestNamedSpinloopDoesNotPickUpCwdFleet(t *testing.T) { node := newRoutableNode(t, "qwen3-27b", true, 10) dir := t.TempDir() fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) - mustWrite(t, filepath.Join(dir, "Spinloop"), - "PROVIDER llamacpp\nMODEL qwen3-27b\nFLEET fleet.yaml\n") + t.Chdir(dir) - // Run from somewhere else entirely: the Spinloop still finds its fleet. - t.Chdir(t.TempDir()) - sel, path, err := readSpinloop("test", filepath.Join(dir, "Spinloop")) + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + sel, path, err := readSpinloop("test", filepath.Join(spinloopDir, "Spinloop")) + if err != nil { + t.Fatal(err) + } + choice, err := routeThroughFleet(sel, path, routeOptions{spinloopNamed: true}) + if err != nil || choice != nil { + t.Errorf("choice = %+v, err = %v; a named Spinloop should not pick up the working directory's fleet", choice, err) + } +} + +// The other side of the same rule: a Spinloop the user did not name finds the +// working directory's fleet.yaml. +func TestUnnamedSpinloopPicksUpCwdFleet(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", true, 10) + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + t.Chdir(dir) + + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + sel, path, err := readSpinloop("test", filepath.Join(spinloopDir, "Spinloop")) if err != nil { t.Fatal(err) } captureStderr(t, func() { - c, err := routeThroughFleet(sel, path, routeOptions{}) + c, err := routeThroughFleet(sel, path, routeOptions{spinloopNamed: false}) if err != nil { - t.Fatalf("a relative FLEET should resolve beside its Spinloop: %v", err) + t.Fatalf("an unnamed Spinloop should find the working directory's fleet: %v", err) } - if c.Node.Name != "gpu-box" { - t.Errorf("chose %q", c.Node.Name) + if c == nil || c.Node.Name != "gpu-box" { + t.Fatalf("expected the working directory's fleet, got %+v", c) } }) } + +// --fleet routes a Spinloop the user named, where the working directory would +// otherwise give no fleet at all. +func TestFleetFlagRoutesANamedSpinloop(t *testing.T) { + node := newRoutableNode(t, "qwen3-27b", true, 10) + dir := t.TempDir() + flagFleet := fleetFileIn(t, dir, "nodes:\n"+node.entry("from-flag")) + t.Chdir(t.TempDir()) // an empty working directory: no fleet.yaml to fall back on + + spinloopDir := t.TempDir() + mustWrite(t, filepath.Join(spinloopDir, "Spinloop"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + sel, path, err := readSpinloop("test", filepath.Join(spinloopDir, "Spinloop")) + if err != nil { + t.Fatal(err) + } + captureStderr(t, func() { + c, err := routeThroughFleet(sel, path, routeOptions{spinloopNamed: true, fleetPath: flagFleet}) + if err != nil { + t.Fatalf("routing failed: %v", err) + } + if c == nil || c.Node.Name != "from-flag" { + t.Fatalf("--fleet should route a named Spinloop, got %+v", c) + } + }) +} + +// The alias SPINLOOP_ALIAS names is a Spinloop the user named: a fleet.yaml in +// the working directory is not picked up for it, even though no path was typed. +func TestAliasNamedSpinloopDoesNotPickUpCwdFleet(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_BASE_URL", "") + node := newRoutableNode(t, "qwen3-27b", true, 10) + registerSpinloop(t, "PROVIDER llamacpp\nMODEL qwen3-27b\nALIAS q3\n") + + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + t.Chdir(dir) // the fleet.yaml, but no Spinloop: the alias supplies the Spinloop + t.Setenv("SPINLOOP_ALIAS", "q3") + + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + stderr := captureStderr(t, func() { + captureStdout(t, func() { + if err := cmdHarness([]string{"-O", "--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + }) + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("harness was not launched: %v", err) + } + if strings.Contains(stderr, "Routing through") { + t.Errorf("a Spinloop named by SPINLOOP_ALIAS should not pick up the working directory's fleet, got:\n%s", stderr) + } +} + +// The valueless --spinloop wears the default Spinloop and is not named: a +// launch run from a directory holding both a Spinloop and a fleet.yaml routes +// through that fleet. +func TestValuelessSpinloopFlagRoutesThroughCwdFleet(t *testing.T) { + home := isolateConfig(t) + t.Setenv("OPENAI_BASE_URL", "") + node := newRoutableNode(t, "qwen3-27b", true, 10) + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + mustWrite(t, filepath.Join(dir, "Spinloop"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + t.Chdir(dir) + + argsFile := filepath.Join(t.TempDir(), "args") + envFile := filepath.Join(t.TempDir(), "env") + stubHarnessBinaryWithEnv(t, argsFile, envFile) + stderr := captureStderr(t, func() { + captureStdout(t, func() { + if err := cmdHarness([]string{"-O", "--", "run"}); err != nil { + t.Fatalf("cmdHarness -O: %v", err) + } + }) + }) + if !strings.Contains(stderr, "gpu-box") { + t.Errorf("the valueless --spinloop should route through the working directory's fleet, got:\n%s", stderr) + } + data, err := os.ReadFile(envFile) + if err != nil { + t.Fatal(err) + } + want := "BASE=http://127.0.0.1:" + strconv.Itoa(node.enginePort) + "/v1" + if !strings.Contains(string(data), want) { + t.Errorf("the agent should point at the chosen node, got:\n%s", data) + } + _ = home +} + +// A bare `spinloop harness` wears no Spinloop at all, so it routes nowhere even +// from a directory holding both a Spinloop and a fleet.yaml. +func TestBareHarnessWearsNothingAndRoutesNothing(t *testing.T) { + isolateConfig(t) + t.Setenv("OPENAI_BASE_URL", "") + node := newRoutableNode(t, "qwen3-27b", true, 10) + dir := t.TempDir() + fleetFileIn(t, dir, "nodes:\n"+node.entry("gpu-box")) + mustWrite(t, filepath.Join(dir, "Spinloop"), "PROVIDER llamacpp\nMODEL qwen3-27b\n") + t.Chdir(dir) + + argsFile := filepath.Join(t.TempDir(), "args") + stubHarnessBinary(t, "opencode", argsFile) + stderr := captureStderr(t, func() { + captureStdout(t, func() { + if err := cmdHarness([]string{"--", "run"}); err != nil { + t.Fatalf("cmdHarness: %v", err) + } + }) + }) + if _, err := os.ReadFile(argsFile); err != nil { + t.Fatalf("harness was not launched: %v", err) + } + if strings.Contains(stderr, "Routing through") { + t.Errorf("a bare harness wears nothing and routes nowhere, got:\n%s", stderr) + } +} diff --git a/docs/commands/fleet.md b/docs/commands/fleet.md index 5c73cc20..0113a246 100644 --- a/docs/commands/fleet.md +++ b/docs/commands/fleet.md @@ -17,8 +17,8 @@ spinloop fleet deploy --all # create every kind: remote node's AWS environmen ``` A fleet is also where [`spinloop harness`](harness.md#launching-against-your-fleet) -sends an agent: a Spinloop naming a `FLEET` picks a node and launches against it, -so the machine you are sitting at needs no engine of its own. +sends an agent: a launch routed through a fleet file picks a node and launches +against it, so the machine you are sitting at needs no engine of its own. ## Try it without any hardware @@ -224,16 +224,17 @@ the launch queries no node and wakes none: ```yaml nodes: … gateway: - url: http://gateway.internal:4000 # required, with a scheme, like a FLEET endpoint + url: http://gateway.internal:4000 # required, with a scheme tokenEnv: GATEWAY_TOKEN # optional; OPENAI_API_KEY when absent ``` -A launch through such a file is dressed exactly as a launch whose `FLEET` -names an endpoint: the section's address is the agent's base URL, and the -token is resolved from the variable the section names — `OPENAI_API_KEY` when -it names none — the way a key is resolved elsewhere: an `ENV` instruction, -then the process environment, then the `.env` beside the Spinloop. A variable -set nowhere fails the launch before anything is written, naming the variable. +A launch through such a file is pointed at the section's address — the agent's +base URL, with the OpenAI-compatible `/v1` prefix added when it carries no path +— and the token is resolved from the variable the section names — +`OPENAI_API_KEY` when it names none — the way a key is resolved elsewhere: an +`ENV` instruction, then the process environment, then the `.env` beside the +Spinloop. A variable set nowhere fails the launch before anything is written, +naming the variable. As with a node's choice, the launch reports the address on stderr before the agent starts. @@ -556,10 +557,10 @@ Use it to check a route before an agent depends on it, to see what the other `spinloop fleet harness` is the fleet-level form of a [harness launch](harness.md#launching-against-your-fleet): the fleet file comes -from the command — `--fleet`, or the `fleet.yaml` beside it — rather than from -the Spinloop's `FLEET`, which stands in when `--fleet` is not given. A fleet -file that names a [gateway](#gateway) points the agent there, so the address -lives in the file, not in every Spinloop: +from the command — `--fleet`, or the `fleet.yaml` in the working directory — +never from the Spinloop. A fleet file that names a +[gateway](#gateway) points the agent there, so the address lives in the file, +not in every Spinloop: ```sh spinloop fleet harness # the Spinloop and fleet.yaml beside it diff --git a/docs/commands/gateway.md b/docs/commands/gateway.md index 31836f98..c680eb3b 100644 --- a/docs/commands/gateway.md +++ b/docs/commands/gateway.md @@ -18,11 +18,11 @@ the fleet file it serves, and a signal shuts it down cleanly. On startup it resolves the fleet file and its own token, and checks the token references the file names — a `tokenEnv` or `engineTokenEnv` variable set nowhere fails here, naming the node, rather than surfacing later as a per-request authentication -failure. It prints the address a Spinloop names in its `FLEET`: +failure. It prints the address to name in the fleet file's `gateway` section: ``` Gateway for fleet.yaml is listening on [::]:4000 -Name http://:4000 in a Spinloop's FLEET +Name http://:4000 in the fleet file's gateway section ``` The host it can know is the one it was told to bind; for a wildcard bind the @@ -32,28 +32,17 @@ side. ## Pointing an agent at it -A Spinloop names the gateway's address in its `FLEET` — a URL, not a file: - -```dockerfile -PROVIDER llamacpp -MODEL qwen3-27b -FLEET http://gateway.internal:4000 -``` - -The launch reads no fleet file and contacts no node — the endpoint has already -done the choosing — and the agent it launches authenticates with the -gateway's token, as `OPENAI_API_KEY`, resolved the way a key is resolved +A [`gateway` section](fleet.md#gateway) in the fleet file names the address and +the variable holding the token. A launch routed through that file — `spinloop +harness -f` or `spinloop fleet harness` — is pointed at the gateway rather than +a node: the section's address is the agent's base URL (with the OpenAI- +compatible `/v1` prefix added when it carries no path), and the agent +authenticates with the gateway's token, resolved the way a key is resolved elsewhere: an `ENV` instruction, then the process environment, then the `.env` -beside the Spinloop. An agent pointed at the gateway holds exactly that one -credential; the node tokens and engine keys live with the gateway, which -presents them to the nodes and the engines. See -[The `Spinloop` file](../spinloop-file.md#running-the-model-on-another-machine-you-own). - -The same pointing can live in the fleet file instead of the Spinloop: a -[`gateway` section](fleet.md#gateway) beside its `nodes` names the address and -the variable holding the token, and `spinloop fleet harness` — the fleet-level -form of a launch — reads it. A Spinloop beside that file then needs only the -model, and the address travels with the file when the gateway moves: +beside the Spinloop. `OPENAI_API_KEY` stands in where the section names no +variable. An agent pointed at the gateway holds exactly that one credential; +the node tokens and engine keys live with the gateway, which presents them to +the nodes and the engines: ```yaml # fleet.yaml @@ -66,6 +55,10 @@ gateway: spinloop fleet harness -O=./Spinloop # from the fleet file's directory ``` +A Spinloop beside that file then needs only the model, and the address travels +with the file when the gateway moves. See +[The `Spinloop` file](../spinloop-file.md#running-the-model-on-another-machine-you-own). + ## What it answers | Path | Meaning | @@ -160,7 +153,5 @@ on a shared machine wants, and the reason the token is not optional there. drives - [`spinloop daemon`](serve.md#the-control-api---api-and-spinloop-daemon) — what each node runs -- [`examples/gateway-docker/`](../../examples/gateway-docker/) — a gateway and - its fleet in containers, with the test suite that asserts all of this -- [The `Spinloop` file](../spinloop-file.md) — the `FLEET` that points an agent - here + - [`examples/gateway-docker/`](../../examples/gateway-docker/) — a gateway and + its fleet in containers, with the test suite that asserts all of this diff --git a/docs/commands/harness.md b/docs/commands/harness.md index 767fdd1c..076309ef 100644 --- a/docs/commands/harness.md +++ b/docs/commands/harness.md @@ -61,7 +61,7 @@ variable chooses which Spinloop, never whether you are configured. See | `--set` | Store the default harness and exit | | `--get` | Print the active harness instead of launching | | `--providers` | Path to a custom catalogue, for the applied Spinloop | -| `-f`, `--fleet` | Route through this fleet file (overrides the Spinloop's `FLEET`) | +| `-f`, `--fleet` | Route through this fleet file (default: `./fleet.yaml`, when the Spinloop is not named) | | `--node` | Pin the launch to one fleet node | | `--prefer` | Rank fleet nodes by `idle` or `active` (overrides the fleet file) | | `--no-wake` | Fail rather than starting an engine on an idle fleet node | @@ -69,14 +69,18 @@ variable chooses which Spinloop, never whether you are configured. See ## Launching against your fleet -A Spinloop with a [`FLEET`](../spinloop-file.md#running-the-model-on-another-machine-you-own) -instruction sends the agent to a machine on your network instead of a local -engine: +A fleet file sends the agent to a machine on your network instead of a local +engine. Which one a launch routes through is a launch concern, not a Spinloop +field: `--fleet ` names it explicitly, and without the flag a Spinloop you +did not name — the default `./Spinloop`, worn by a valueless `-O` — takes the +`fleet.yaml` in the working directory. A Spinloop you did name — a path, a `-O` +value, or the alias `SPINLOOP_ALIAS` names — routes only by flag. See +[fleet files](../spinloop-file.md#running-the-model-on-another-machine-you-own). ```sh -spinloop harness my-spinloop # picks a node, launches the agent at it -spinloop harness --node gpu-box my-spinloop -spinloop harness --prefer active my-spinloop +spinloop harness -O -f fleet.yaml # valueless -O wears ./Spinloop; routes through fleet.yaml +spinloop harness --node gpu-box -f fleet.yaml +spinloop harness --prefer active -f fleet.yaml ``` spinloop queries the fleet, prefers a node already serving the Spinloop's model, @@ -97,8 +101,8 @@ takes the machine that has been quiet longest, keeping a second agent off an engine that is mid-request; `active` consolidates onto the busy one instead. [`spinloop fleet harness`](fleet.md#launching-the-harness) is the fleet-level -form of this launch: the fleet file comes from the command rather than from the -Spinloop's `FLEET`, and a fleet file that names a +form of this launch: the fleet file comes from the command — `-f`, or the +`fleet.yaml` beside it — and a fleet file that names a [gateway](fleet.md#gateway) points the agent there, so the address lives in the file, not in every Spinloop. diff --git a/docs/spinloop-file.md b/docs/spinloop-file.md index e015a85b..f76b02f6 100644 --- a/docs/spinloop-file.md +++ b/docs/spinloop-file.md @@ -122,51 +122,44 @@ machine. ## Running the model on another machine you own -`FLEET` names a [fleet file](commands/fleet.md#fleetyaml) — the machines on your -network running `spinloop daemon` — and lets `spinloop harness` pick one for you: +A [fleet file](commands/fleet.md#fleetyaml) names the machines on your network +running `spinloop daemon`, and `spinloop harness` can pick one for you. Which +fleet file a launch routes through is a launch concern, not a Spinloop field: + +- `spinloop harness --fleet ` names it explicitly. +- Without the flag, a Spinloop you did not name — the default `./Spinloop`, + worn by a valueless `--spinloop` — takes the `fleet.yaml` in the working + directory. A Spinloop you did name — a path, a `--spinloop` value, or the + alias `SPINLOOP_ALIAS` names — routes only by flag. ```dockerfile # Spinloop PROVIDER llamacpp MODEL qwen3-27b -FLEET ./fleet.yaml ``` -Launching against it queries the fleet, picks a node already serving that model, +Launching against a fleet queries it, picks a node already serving that model, and points the agent at that node's engine. When nothing is serving it, spinloop starts one and waits for it to load — so the machine you sat down at needs -nothing but a path to the fleet file. `spinloop harness --fleet=` overrides -the instruction, `--node ` pins one machine, and `--no-wake` refuses to -start anything. - -`FLEET` and `REMOTE` are mutually exclusive: each is a different answer to where -the model is served from, and a Spinloop stating both fails to parse rather than -picking one. As with `REMOTE`, note the missing `BASEURL` — the address is -whichever node gets chosen. Writing one pins the address and turns routing off, -and spinloop says so rather than choosing a node and discarding it. - -A `FLEET` may also name a URL rather than a file: a single endpoint that has -already done the choosing, the shape -[`spinloop gateway`](commands/gateway.md) serves: - -```dockerfile -PROVIDER llamacpp -MODEL qwen3-27b -FLEET http://gateway.internal:4000 -``` - -Naming one reads no fleet file and contacts no node. The launch is pointed at -the address as given — with the OpenAI-compatible `/v1` prefix added when it -carries no path, and a value that already carries one used as given — and the -agent it launches authenticates with the endpoint's token, resolved the way a -key is resolved elsewhere: an `ENV` instruction, then the process environment, -then the `.env` beside the Spinloop. A variable already set wins, as on the -remote path. Set nowhere, the launch fails before it writes anything, naming -`OPENAI_API_KEY`. +nothing but the fleet file. `--node ` pins one machine, and `--no-wake` +refuses to start anything. + +A fleet file may name a [gateway](commands/gateway.md) instead of nodes: a +single endpoint that has already done the choosing. A launch through such a file +is pointed at the gateway's address — with the OpenAI-compatible `/v1` prefix +added when it carries no path — and the agent it launches authenticates with the +gateway's token, resolved the way a key is resolved elsewhere: an `ENV` +instruction, then the process environment, then the `.env` beside the Spinloop. +A variable already set wins. Set nowhere, the launch fails before it writes +anything, naming the variable the section names — `OPENAI_API_KEY` where the +section names none. + +Note the missing `BASEURL` — the address is whichever node gets chosen. Writing +one pins the address and turns routing off, and spinloop says so rather than +choosing a node and discarding it. See [`spinloop fleet route`](commands/fleet.md#which-node-would-i-get) to check -which node you would get before launching anything — a route against an -endpoint just names it, without querying a node or starting one. +which node you would get before launching anything. ## Syntax @@ -183,7 +176,6 @@ One instruction per line: a keyword followed by a single value. | `BASEURL` | no | `--base-url` | `BASEURL https://gateway/v1` | | `PRESET` | no | `spinloop serve` | `PRESET ./preset.ini` | | `REMOTE` | no | `spinloop remote` | `REMOTE ./remote.json` | -| `FLEET` | no | `spinloop harness`, `spinloop fleet` | `FLEET ./fleet.yaml` | | `ENV` | no (repeatable) | `spinloop remote`, `spinloop harness` | `ENV AWS_PROFILE=prod` | Rules: diff --git a/examples/fleet-docker/README.md b/examples/fleet-docker/README.md index dd7a24ab..bb975c58 100644 --- a/examples/fleet-docker/README.md +++ b/examples/fleet-docker/README.md @@ -68,10 +68,10 @@ path — `docker compose exec studio ps ax` shows `--api-key-file`, not the key. ```sh # Which node would a harness launch pick? (Changes nothing.) -spinloop fleet route ./client/Spinloop +spinloop fleet route ./client/Spinloop --fleet ./fleet.yaml # Actually launch an agent against the fleet, waking a node if none is serving. -spinloop harness ./client/Spinloop +spinloop harness ./client/Spinloop --fleet ./fleet.yaml # A node that goes away: the row degrades, the rest keep reporting, exit 0. docker compose stop gpu-box @@ -107,7 +107,7 @@ cannot quietly stop working. | `Dockerfile` | Builds spinloop from this working tree, adds the Imposter engine and the shim. | | `shim/llama-server` | Stands in for the engine binary. Execs the Imposter engine **directly**, so the daemon supervises it as its own child. | | `engine/` | What the fake engine serves: `/health`, and a `/metrics` spinloop can parse. | -| `client/Spinloop` | What a *client* wears to use the fleet: a model, and a `FLEET`. The nodes hold no Spinloop at all. | +| `client/Spinloop` | What a *client* wears to use the fleet: a model. The fleet file is named at launch with `--fleet`. The nodes hold no Spinloop at all. | Two details that are easy to get wrong, and matter: diff --git a/examples/fleet-docker/client/Spinloop b/examples/fleet-docker/client/Spinloop index 6fb54028..9b8e1b94 100644 --- a/examples/fleet-docker/client/Spinloop +++ b/examples/fleet-docker/client/Spinloop @@ -4,9 +4,8 @@ # # Note the missing BASEURL. The whole point is that spinloop picks a node and # fills the address in; a Spinloop that states one is taken at its word and -# never routed. +# never routed. The fleet file is named at launch (--fleet), not here. PROVIDER llamacpp MODEL org/fake-model ALIAS fake-model CONTEXT 4096 -FLEET ../fleet.yaml diff --git a/examples/fleet-docker/run-tests.sh b/examples/fleet-docker/run-tests.sh index c66f5b74..bdb60f9d 100755 --- a/examples/fleet-docker/run-tests.sh +++ b/examples/fleet-docker/run-tests.sh @@ -258,23 +258,27 @@ restart_node() { } ####################################### -# Wait until the engine's token counters reach the fleet view. A node reads -# `running` as soon as the engine process is alive, which is earlier than the -# counters exist: the engine needs a moment to answer at all, and the daemon -# reports the reading its background sampler took rather than scraping when -# asked. The sampler retries about once a second until the first reading -# lands, so the counters trail the state by a second or two and a state check -# returns inside that window. +# Wait until a complete reading reaches the fleet view: the engine's token +# counters and the host's figures. A node reads `running` as soon as the engine +# process is alive, which is earlier than either exists: the engine needs a +# moment to answer at all, and the daemon reports the reading its background +# sampler took rather than scraping when asked. The sampler records the two in +# one tick, but the token scrape runs before the slow `vmstat` host read, so a +# just-woken node can report counters before its first host reading lands. +# Waiting for both means the assertions read a complete reading, not the +# counters half a tick early. # Arguments: # Timeout in seconds. # Returns: -# 0 once the counters appear, 1 on timeout. +# 0 once a complete reading appears, 1 on timeout. ####################################### -wait_for_tokens() { +wait_for_metrics() { local timeout="$1" local deadline=$((SECONDS + timeout)) + local out while (( SECONDS < deadline )); do - if [[ "$(fleet metrics)" == *"prompt tokens"* ]]; then + out="$(fleet metrics)" + if [[ "${out}" == *"prompt tokens"* && "${out}" == *"RAM"* ]]; then return 0 fi sleep 1 @@ -446,7 +450,7 @@ STUB local launch launch="$(PATH="${sandbox}/bin:${PATH}" HOME="${sandbox}/home" \ XDG_CONFIG_HOME="${sandbox}/home/.config" \ - "${SPINLOOP_BIN}" harness -O="${spinloop_file}" -H opencode 2>&1 || true)" + "${SPINLOOP_BIN}" harness -O="${spinloop_file}" --fleet "${HERE}/fleet.yaml" -H opencode 2>&1 || true)" assert_contains "a routed launch wakes the node" "${launch}" "Waking studio" assert_contains "the agent is pointed at the published engine port" "${launch}" "18080" assert_contains "the agent is given the key the client set" \ @@ -511,7 +515,7 @@ test_metrics() { # Running is the process, not a sample; let the counters land before reading # them. A timeout is not fatal here — the assertions below say what was # missing, which is more use than an abort. - wait_for_tokens 30 || true + wait_for_metrics 30 || true local out out="$(fleet metrics)" diff --git a/examples/fleet-local/README.md b/examples/fleet-local/README.md index f1591d26..0a10111e 100644 --- a/examples/fleet-local/README.md +++ b/examples/fleet-local/README.md @@ -63,15 +63,13 @@ real network: would be unreachable, and routing says so rather than handing you an address that refuses connections. -[`Spinloop`](Spinloop) is `examples/llamacpp/gemma4`'s with one line added: - -```dockerfile -FLEET ./fleet.yaml -``` - -and one line deliberately absent — there is no `BASEURL`. The address is -whichever node gets chosen; pinning one turns routing off, and spinloop says so -rather than choosing a node it would then ignore. +[`Spinloop`](Spinloop) is `examples/llamacpp/gemma4`'s, beside a +[`fleet.yaml`](fleet.yaml). The fleet file is what turns the launch into a +routed one: a valueless `-O` wears the default `./Spinloop`, which — being +unnamed — takes the `fleet.yaml` in the working directory. One line is +deliberately absent — there is no `BASEURL`. The address is whichever node gets +chosen; pinning one turns routing off, and spinloop says so rather than choosing +a node it would then ignore. [`preset.ini`](preset.ini) is unchanged from the non-fleet example. It matters more here than it looks: when routing wakes a node it sends the preset's flags @@ -128,10 +126,10 @@ Using local at http://127.0.0.1:8080/v1 — woken to serve gemma-4-12b-it The wait is the model loading — minutes for a cold 12B, then seconds forever after, because the engine stays up between sessions. -**`-O` is not optional.** A bare `spinloop harness` launches unconfigured: it applies -no Spinloop, so there is no `FLEET` to act on and nothing routes. Wear the Spinloop -(`-O` for `./Spinloop`, a path, or a [registered alias](../../docs/commands/alias.md)) -and routing follows from it. +**`-O` is not optional.** A bare `spinloop harness` launches unconfigured: it +applies no Spinloop, so there is no model to route and no fleet is picked up. +Wear the Spinloop (`-O` for `./Spinloop`, a path, or a +[registered alias](../../docs/commands/alias.md)) and routing follows from it. ## Prerequisites diff --git a/examples/fleet-local/Spinloop b/examples/fleet-local/Spinloop index f847ab70..51c273f4 100644 --- a/examples/fleet-local/Spinloop +++ b/examples/fleet-local/Spinloop @@ -1,11 +1,10 @@ # Gemma-4-12B-IT, served by a daemon on this machine and reached through the -# fleet. Same model and preset as examples/llamacpp/gemma4 — the only addition -# is FLEET, which is what turns `spinloop harness` from "assume something is -# listening on :8080" into "find a node, start one if need be, then launch". +# fleet. Same model and preset as examples/llamacpp/gemma4 — the difference is +# that the fleet.yaml beside it turns `spinloop harness` from "assume something +# is listening on :8080" into "find a node, start one if need be, then launch". PROVIDER llamacpp ALIAS gemma-4-12b-it CONTEXT 32768 # match the preset's ctx-size PRESET ./preset.ini # what a woken node is told to run -FLEET ./fleet.yaml # the fleet to route through (this machine) # No BASEURL: the address is whichever node gets chosen. Pinning one here # would be taken at its word and turn routing off. diff --git a/examples/gateway-docker/README.md b/examples/gateway-docker/README.md index dec54fc0..3732129c 100644 --- a/examples/gateway-docker/README.md +++ b/examples/gateway-docker/README.md @@ -150,5 +150,5 @@ Two details that are easy to get wrong, and matter: - [`examples/fleet-docker/`](../fleet-docker/) — a plain fleet, no gateway - [`docs/commands/gateway.md`](../../docs/commands/gateway.md) -- [`docs/spinloop-file.md`](../../docs/spinloop-file.md) — the endpoint form of `FLEET` +- [`docs/commands/fleet.md`](../../docs/commands/fleet.md#the-gateway-section) — the `gateway` section - [HTTP Control API](../../docs/http-api.md) diff --git a/examples/gateway-docker/client/Spinloop b/examples/gateway-docker/client/Spinloop index 918ef42f..edfb46aa 100644 --- a/examples/gateway-docker/client/Spinloop +++ b/examples/gateway-docker/client/Spinloop @@ -1,7 +1,7 @@ # What an agent's machine wears to use this stack: a model, and nothing about # where it is served — the fleet file's gateway section says that. `spinloop -# fleet harness` with this Spinloop beside that fleet file points the agent at -# the gateway; the Spinloop stays valid whatever address the gateway moves to. +# fleet harness` run from the directory holding the fleet file points the agent +# at the gateway; the Spinloop stays valid whatever address the gateway moves to. # # The agent holds exactly one credential — the gateway's token, as # GATEWAY_TOKEN (from the environment, or a .env beside the fleet file). The diff --git a/internal/fleet/config.go b/internal/fleet/config.go index 7f72294d..924040a4 100644 --- a/internal/fleet/config.go +++ b/internal/fleet/config.go @@ -187,9 +187,7 @@ type EngineOverride struct { } // DefaultGatewayTokenEnv is the variable a gateway's token is resolved under -// when the section names none: the same variable an endpoint FLEET resolves -// under, so a file that moves a launch from an endpoint to a section changes -// nothing the client has to export. +// when the section names none — the standard OpenAI-compatible key variable. const DefaultGatewayTokenEnv = "OPENAI_API_KEY" // GatewayConfig is the fleet file's gateway section: the address the fleet is @@ -284,12 +282,11 @@ func (c *Config) validate() error { if c.Gateway.URL == "" { return fmt.Errorf("the gateway section names no url: name the gateway's address under `url:`") } - // The section's url is an endpoint value wearing a section: it is - // dialed over HTTP, so it carries a scheme the way FLEET's endpoint - // values do. + // The section's url is an address spinloop dials over HTTP, so it + // carries a scheme: a bare host names nothing it could reach. if !strings.Contains(c.Gateway.URL, "://") { return fmt.Errorf( - "the gateway section's url %q has no scheme: give the gateway's full address, the way an endpoint value does", + "the gateway section's url %q has no scheme: give the gateway's full address, including http:// or https://", c.Gateway.URL) } } diff --git a/internal/fleet/config_test.go b/internal/fleet/config_test.go index 0fedb7a6..b3deff5a 100644 --- a/internal/fleet/config_test.go +++ b/internal/fleet/config_test.go @@ -686,9 +686,7 @@ gateway: } } -// A section that names no tokenEnv resolves under the endpoint FLEET's -// variable, so moving a launch from an endpoint to a section changes nothing -// the client has to export. +// A section that names no tokenEnv resolves under the default variable. func TestGatewaySectionTokenDefaults(t *testing.T) { path := writeFleet(t, ` nodes: diff --git a/internal/fleet/select.go b/internal/fleet/select.go index 9c3ca580..1687bd3f 100644 --- a/internal/fleet/select.go +++ b/internal/fleet/select.go @@ -118,15 +118,14 @@ type Choice struct { Reason string // Woken records that this node was started to satisfy the launch. Woken bool - // Gateway records that FLEET named an endpoint rather than a fleet file: - // the endpoint has already done the choosing, Node is empty, and BaseURL - // is the endpoint's address rather than a node's engine. A fleet file's - // gateway section yields the same shape. + // Gateway records that the fleet's gateway answered rather than a node: + // it has already done the choosing, Node is empty, and BaseURL is the + // gateway's address rather than a node's engine. Gateway bool // GatewayTokenEnv names the variable holding the token for a Gateway // choice: the variable a fleet file's gateway section names, defaulted to - // the endpoint FLEET's variable when the section names none. The launch - // resolves it through its key chain, as with every other key. + // DefaultGatewayTokenEnv when the section names none. The launch resolves + // it through its key chain, as with every other key. GatewayTokenEnv string } diff --git a/internal/gateway/gateway.go b/internal/gateway/gateway.go index 1020d387..c5485e0a 100644 --- a/internal/gateway/gateway.go +++ b/internal/gateway/gateway.go @@ -4,8 +4,8 @@ // for — so a machine running an agent needs nothing but a URL and one token. // // It is a foreground process, the way `spinloop serve` is: it holds the fleet -// file it serves, and a machine that hosts agents points its Spinloop's FLEET -// at the address it prints. +// file it serves, and a machine that hosts agents names the address it prints +// in that file's gateway section. package gateway @@ -32,8 +32,8 @@ import ( ) // DefaultListen is where the gateway answers when --listen is not given: a -// fixed port on every interface, so a Spinloop's FLEET can name one address -// without knowing the machine it lands on. +// fixed port on every interface, so a fleet file's gateway section can name +// one address without knowing the machine it lands on. const DefaultListen = ":4000" // LoopbackListen is where `--loopback` binds the gateway: the default port on diff --git a/internal/spinloop/spinloop.go b/internal/spinloop/spinloop.go index 0e72e7be..c0c41cc0 100644 --- a/internal/spinloop/spinloop.go +++ b/internal/spinloop/spinloop.go @@ -15,7 +15,6 @@ // BASEURL https://gateway/v1 # optional; API base URL override // PRESET ./preset.ini # optional; llama.cpp preset for `serve` // REMOTE ./remote.json # optional; remote-instance config for `remote` -// FLEET ./fleet.yaml # optional; the fleet a harness launch routes through // ENV AWS_PROFILE=dev # optional, repeatable; local env var // // MODEL is the reference the provider itself understands: an OpenRouter/Bedrock @@ -59,10 +58,6 @@ type Selection struct { // local-serving capability for how it and Context translate into each // engine's own flags. Parallel string - // Fleet is the FLEET instruction's value: a path to a fleet file whose - // nodes a launch chooses between, or a URL naming an endpoint that has - // already chosen (see FleetIsEndpoint). - Fleet string // DisplayName is the harness provider's display name, derived at apply time // rather than parsed from a Spinloop — like BaseURL, it may be filled from the // remote environment. It is set only when a REMOTE renames the provider, to @@ -92,7 +87,6 @@ const ( kwBaseURL = "baseurl" kwPreset = "preset" kwRemote = "remote" - kwFleet = "fleet" kwEnv = "env" ) @@ -101,7 +95,7 @@ const ( // "" for an unrecognised keyword. func canonicalKeyword(kw string) string { switch kw { - case kwProvider, kwModel, kwAlias, kwContext, kwOutput, kwParallel, kwPreset, kwRemote, kwFleet, kwEnv: + case kwProvider, kwModel, kwAlias, kwContext, kwOutput, kwParallel, kwPreset, kwRemote, kwEnv: return kw case kwBaseURL, "base-url", "base_url", "url": return kwBaseURL @@ -129,7 +123,7 @@ func Parse(data []byte) (Selection, error) { fields := strings.Fields(text) canon := canonicalKeyword(strings.ToLower(fields[0])) if canon == "" { - return Selection{}, fmt.Errorf("line %d: unknown keyword %q (expected PROVIDER, MODEL, ALIAS, CONTEXT, OUTPUT, PARALLEL, BASEURL, PRESET, REMOTE, FLEET, or ENV)", line, fields[0]) + return Selection{}, fmt.Errorf("line %d: unknown keyword %q (expected PROVIDER, MODEL, ALIAS, CONTEXT, OUTPUT, PARALLEL, BASEURL, PRESET, REMOTE, or ENV)", line, fields[0]) } switch { case len(fields) < 2: @@ -175,8 +169,6 @@ func Parse(data []byte) (Selection, error) { sel.Preset = value case kwRemote: sel.Remote = value - case kwFleet: - sel.Fleet = value } } if err := scanner.Err(); err != nil { @@ -186,28 +178,9 @@ func Parse(data []byte) (Selection, error) { if sel.Provider == "" { return Selection{}, fmt.Errorf("Spinloop is missing a PROVIDER instruction") } - // REMOTE and FLEET are two different answers to where the model is served - // from — a deployed endpoint, or a machine on your network. A Spinloop - // stating both is a mistake rather than a precedence to resolve. BASEURL is - // not in conflict: it is the pinned address that already wins over REMOTE, - // and it wins over FLEET the same way. - if sel.Remote != "" && sel.Fleet != "" { - return Selection{}, fmt.Errorf( - "Spinloop sets both REMOTE (line %d) and FLEET (line %d): each names where the model is served from, so state one", - seen[kwRemote], seen[kwFleet]) - } return sel, nil } -// FleetIsEndpoint reports whether a FLEET value names an endpoint that has -// already chosen a node — a gateway — rather than a fleet file to choose from. -// A value carrying a scheme is an endpoint; anything else is a path. Keeping -// both behind one instruction is what lets a gateway slot in later without a -// second keyword. -func (s Selection) FleetIsEndpoint() bool { - return strings.Contains(s.Fleet, "://") -} - // stripComment removes a comment from a Spinloop line. A line whose first // non-blank character is `#` is dropped entirely; otherwise a trailing ` #` // (or tab-`#`) comment is removed. Provider, family, and model identifiers @@ -242,7 +215,6 @@ func Format(sel Selection) string { line("BASEURL", sel.BaseURL) line("PRESET", sel.Preset) line("REMOTE", sel.Remote) - line("FLEET", sel.Fleet) for _, e := range sel.Env { line("ENV", e.Key+"="+e.Value) } diff --git a/internal/spinloop/spinloop_test.go b/internal/spinloop/spinloop_test.go index 7a67a01d..728bc5ed 100644 --- a/internal/spinloop/spinloop_test.go +++ b/internal/spinloop/spinloop_test.go @@ -180,56 +180,17 @@ func TestParse_Remote(t *testing.T) { } } -func TestParse_Fleet(t *testing.T) { - sel, err := Parse([]byte("PROVIDER llamacpp\nMODEL qwen3-27b\nFLEET ./fleet.yaml\n")) - if err != nil { - t.Fatal(err) - } - if sel.Fleet != "./fleet.yaml" { - t.Errorf("Fleet = %q, want ./fleet.yaml", sel.Fleet) - } - if sel.FleetIsEndpoint() { - t.Error("a path should not read as an endpoint") - } - if out := Format(sel); !strings.Contains(out, "FLEET ./fleet.yaml") { - t.Errorf("Format should emit FLEET, got:\n%s", out) - } - if _, err := Parse([]byte("PROVIDER x\nFLEET a\nFLEET b\n")); err == nil { - t.Error("duplicate FLEET should error") - } -} - -// A FLEET naming a URL is the gateway shape: it parses, so the eventual -// gateway needs no new keyword, and it is distinguishable from a path. -func TestParse_FleetEndpoint(t *testing.T) { - sel, err := Parse([]byte("PROVIDER llamacpp\nFLEET http://gateway.internal:4000\n")) - if err != nil { - t.Fatal(err) - } - if !sel.FleetIsEndpoint() { - t.Errorf("FLEET %q should read as an endpoint", sel.Fleet) - } -} - -// REMOTE and FLEET each name where the model is served from, so a Spinloop -// stating both is a mistake. BASEURL is the pinned address that already wins -// over REMOTE, so pairing it with FLEET is not a conflict. -func TestParse_FleetAndRemoteConflict(t *testing.T) { - _, err := Parse([]byte("PROVIDER x\nFLEET ./fleet.yaml\nREMOTE ./remote.json\n")) +// FLEET is no longer a keyword: a Spinloop still carrying one fails with the +// standard unknown-keyword error, which names the offending line and lists the +// accepted keywords. There is no migration message of its own. +func TestParse_FleetIsAnUnknownKeyword(t *testing.T) { + _, err := Parse([]byte("PROVIDER llamacpp\nMODEL qwen3-27b\nFLEET ./fleet.yaml\n")) if err == nil { - t.Fatal("FLEET with REMOTE should error") + t.Fatal("a Spinloop carrying FLEET should not parse") } - for _, want := range []string{"REMOTE", "FLEET"} { + for _, want := range []string{"line 3", "FLEET", "expected PROVIDER"} { if !strings.Contains(err.Error(), want) { - t.Errorf("error should name %s, got %q", want, err) + t.Errorf("error should contain %q, got %q", want, err) } } - - sel, err := Parse([]byte("PROVIDER x\nFLEET ./fleet.yaml\nBASEURL http://pinned/v1\n")) - if err != nil { - t.Fatalf("FLEET with BASEURL should parse: %v", err) - } - if sel.BaseURL != "http://pinned/v1" || sel.Fleet != "./fleet.yaml" { - t.Errorf("both should survive parsing, got %+v", sel) - } } diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/.openspec.yaml b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/.openspec.yaml new file mode 100644 index 00000000..7a8e2be6 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-08 diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/design.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/design.md new file mode 100644 index 00000000..291e62c4 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/design.md @@ -0,0 +1,102 @@ +# Design + +## What is removed, and what stays + +The `FLEET` instruction goes away in both forms — a fleet-file path and a +gateway URL (the latter added by the still-open `add-fleet-gateway` change, +whose code lands on this same branch). That deletes: + +- `Selection.Fleet`, the `kwFleet` parse case, the REMOTE/FLEET exclusivity + check, `FleetIsEndpoint`, and the `FLEET` line in `Format`. +- The endpoint branch of `routeThroughFleet` and `fleet route`'s endpoint + handling. +- Every FLEET scenario in the four affected requirements. + +What stays: + +- The `gateway` section of the fleet file (`add-fleet-harness`, already + archived) — the fleet's address now lives with the fleet. +- `endpointBaseURL` and the gateway-section routing branch of + `routeThroughFleet`: the section branch points a launch at the fleet's + gateway exactly as the old endpoint branch did, and both need the same + base-URL normalisation. +- `REMOTE` and everything around it. Its removal is a separate follow-up; + only the exclusivity check that paired it with `FLEET` goes. + +## The discovery rule, per command + +"Explicitly named" means the user supplied the identity of the Spinloop the +launch wears: a leading positional argument, a `--spinloop`/`-O` value, or +`SPINLOOP_ALIAS` set. A valueless `--spinloop` wears the default Spinloop and +counts as not named — the user asked for whatever is there, which is exactly +the case where a `fleet.yaml` beside it is the local convention. A bare +`spinloop harness` wears no Spinloop at all (the documented "still applies +nothing"), so there is nothing to route and the discovery rules never engage. + +| command | fleet file in force | +| --- | --- | +| `spinloop harness` | `--fleet`/`-f`; else `./fleet.yaml` when the Spinloop was not explicitly named; else none | +| `spinloop fleet route` | `--fleet`/`-f`; else `./fleet.yaml` when the Spinloop was not explicitly named; else fail naming `--fleet` | +| `spinloop fleet harness` | `--fleet`/`-f`; else the `fleet.yaml` beside the command — the Spinloop's explicitness does not gate this one, because its fleet file has always come from the command, never from beside the Spinloop | + +A pinned `BASEURL` still beats routing on every path, and the stderr +announcement ("Using … at … — …") is unchanged in shape. + +`fleet route` failing with no fleet file in force is the one behaviour that +gets stricter: it used to error naming the missing `FLEET`, and now errors +naming `--fleet` — same shape, different fix. + +## Restated requirements, not modified ones + +OpenSpec refuses to drop a scenario from a `MODIFIED` requirement, and two of +the requirements being changed carry scenarios whose names reference the +`FLEET` ("The flag overrides the instruction", "A Spinloop with no FLEET is +unaffected", "The command's file wins over the Spinloop's FLEET"). Leaving +those names in the main spec would keep the removed keyword on the page, so the +two requirements are removed and restated under clean names, the same +remove+add-under-a-new-name pattern `retire-model-families` used: + +- `fleet-routing`: `A fleet-routed launch` → `A launch routed through a fleet`. +- `fleet-client`: `A fleet harness command` → `The fleet harness command`. + +Prose elsewhere that speaks of "a fleet-routed launch" as a concept rather +than naming the requirement is left alone; the only formal cross-references are +updated to the new names. + +## No fail-loud migration + +A Spinloop carrying `FLEET` now fails with the standard unknown-keyword error, +which lists the accepted keywords. A dedicated "FLEET is gone, do X instead" +message would be a migration branch in the parser for a keyword that never +comes back, and the standard error plus the docs carry the same information. + +## Why the `add-fleet-gateway` delta is trimmed here + +`add-fleet-gateway` is unarchived on this branch, and its delta adds the +endpoint-`FLEET` behaviour this change deletes. If the delta stayed intact, +archiving it after this one would re-add requirements whose code no longer +exists. The trim is part of this change so the two archive in a consistent +order on main: + +- `specs/spinloop-files/spec.md` — deleted outright; its only requirement is + the endpoint form of `FLEET`. +- `specs/fleet-routing/spec.md` — the two endpoint-`FLEET` ADDED + requirements go; the MODIFIED `Choosing a node` and `Waking a node` stay + untouched (neither mentions FLEET). +- `specs/fleet-gateway/spec.md` — the gateway command's printed address is + named in a fleet file's `gateway` section rather than a Spinloop's `FLEET`, + in the requirement text and its scenario. + +## Examples + +`examples/fleet-local`: the Spinloop sits beside its `fleet.yaml`, so dropping +the `FLEET` line keeps `spinloop harness` routing exactly as the example +describes — the README's story sharpens rather than changes. + +`examples/fleet-docker`: the client's Spinloop lives in a subdirectory, so an +explicit `-O` path would no longer pick up the fleet. The Spinloop drops its +`FLEET` line and the test run passes `--fleet` explicitly — which is also a +live demonstration of the flag overriding the directory convention. + +`examples/gateway-docker` needs no change to its Spinloop (it carries no +`FLEET`); its test run is checked against the new `fleet harness` resolution. diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/proposal.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/proposal.md new file mode 100644 index 00000000..17663b13 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/proposal.md @@ -0,0 +1,65 @@ +# Remove the FLEET instruction from the Spinloop file + +## Why + +The `FLEET` instruction duplicates a fact the directory already holds. A +Spinloop that sits beside its `fleet.yaml` does not need to name it, and a +Spinloop named from elsewhere is being pointed at deliberately — an incidental +`fleet.yaml` in the working directory should not hijack that launch. The +gateway feature (a `gateway` section in the fleet file) already gives a fleet +an address its members do not have to carry, so the instruction's remaining +job — naming where the model is served from — is covered by the directory +convention plus the `--fleet` flag. + +Support goes away entirely: a Spinloop that still carries `FLEET` fails to +parse with the standard unknown-keyword error, which lists the accepted +keywords. There is no migration message and no compat path. + +## What changes + +- **Spinloop grammar** (`spinloop-files`): `FLEET` is no longer a keyword; the + parser rejects it as unknown. The `FLEET and REMOTE are exclusive` and + `FLEET names a file or an endpoint` requirements are removed, along with the + parse-time exclusivity check and the `FLEET` line `Format` emits. `REMOTE` + stays — its removal is a separate follow-up. +- **Routing discovery** (`fleet-routing`): a launch routes through the fleet + file given by `--fleet`/`-f` when that is passed. Otherwise, when the + Spinloop is not named explicitly — no positional path, no `--spinloop` value, + no `SPINLOOP_ALIAS` — a `./fleet.yaml` in the working directory is used. An + explicitly named Spinloop routes only when `--fleet` is given. A pinned + `BASEURL` still wins over routing, as today. +- **`spinloop fleet route`** (`fleet-client`): the Spinloop's `FLEET` is no + longer a fleet source; the same discovery rules apply, and a route with no + fleet file in force fails naming `--fleet`. +- **`spinloop fleet harness`** (`fleet-client`): the fleet file comes from + `--fleet`/`-f` or `./fleet.yaml` beside the command; the Spinloop's `FLEET` + is no longer consulted. +- **FLEET-as-endpoint removed**: the endpoint form of `FLEET` (a URL value, + added by the still-open `add-fleet-gateway` change) is removed with the rest + of the instruction — its code lands on this same branch, unarchived. +- **`spinloop gateway` output**: the address it prints is named in a fleet + file's `gateway` section rather than in a Spinloop's `FLEET`. +- **`add-fleet-gateway` delta trimmed**: that open change's delta specs are + edited so archiving it after this one cannot resurrect FLEET requirements — + its `spinloop-files` delta goes, its two endpoint-`FLEET` `fleet-routing` + requirements go, and its `fleet-gateway` output wording points at the + `gateway` section. +- **Docs and examples**: the `FLEET` sections of the Spinloop file and command + docs are replaced with the discovery rules; `examples/fleet-local` drops + `FLEET` from its Spinloop (it sits beside its `fleet.yaml`), and + `examples/fleet-docker`'s client Spinloop drops `FLEET` with the test run + passing `--fleet`. + +## Capabilities + +- `spinloop-files`: `Spinloop file format` modified (keyword list, scenarios); + `FLEET and REMOTE are exclusive` removed; `FLEET names a file or an endpoint` + removed. +- `fleet-routing`: `A fleet-routed launch` removed and restated as `A launch + routed through a fleet` (discovery rules; its FLEET scenarios cannot stand + under them); `A fleet file naming a gateway routes the launch at it` modified + (its fleet file no longer comes from the Spinloop's `FLEET`). +- `fleet-client`: `Explaining a route` modified; `A fleet harness command` + removed and restated as `The fleet harness command` (fleet sources). +- `fleet-config`: `A fleet file MAY name a gateway` modified (wording: the + `FLEET` it referenced is gone). diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-client/spec.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-client/spec.md new file mode 100644 index 00000000..9cb93b9d --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-client/spec.md @@ -0,0 +1,118 @@ +## ADDED Requirements + +### Requirement: The fleet harness command + +`spinloop fleet harness` SHALL configure the active harness for a fleet and +launch it: the fleet-level form of a launch routed through a fleet, in which +the fleet file comes from the command rather than from the Spinloop. The +command SHALL take a Spinloop the way `spinloop harness` does — an +`-O`/`--spinloop` argument, a leading alias or path, or the `Spinloop` beside +it — and a fleet file from `--fleet`/`-f`, defaulting to the `fleet.yaml` +beside it. + +The command SHALL route the launch the way a launch routed through a fleet +routes: at the gateway where the effective fleet file names one, otherwise by +choosing a node and, where the fleet's wake policy allows it, waking one — +honouring `--node`, `--prefer`, `--no-wake` and `--wake-timeout` as the launch +does. The choice SHALL be reported on stderr before the harness launches, as a +launch routed through a fleet reports its choice. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed, and a variable already +set in spinloop's environment SHALL win, in each case as on the launch path. A +command with no Spinloop to route SHALL fail before launching, saying that a +launch needs a Spinloop to know which model to route. + +#### Scenario: A fleet's gateway is used + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding a + fleet file that names a gateway and a Spinloop, and the gateway's token is + set +- **THEN** the harness is applied for the Spinloop's model, the launched + agent's endpoint is the gateway's address, and no node is contacted + +#### Scenario: A fleet with no gateway routes to a node + +- **WHEN** the user runs `spinloop fleet harness` against a fleet file that + names no gateway, and a node is running the Spinloop's model +- **THEN** the harness is applied with that node's engine as the agent's + endpoint, as a launch routed through a fleet would apply it + +#### Scenario: The command's file wins over the fleet file beside it + +- **WHEN** the user runs `spinloop fleet harness -f ./a.yaml` in a directory + holding a `./fleet.yaml` +- **THEN** the fleet in `./a.yaml` is the one the launch routes through + +#### Scenario: No Spinloop, no route + +- **WHEN** the user runs `spinloop fleet harness` in a directory holding no + Spinloop and gives none +- **THEN** the command fails saying a launch needs a Spinloop to know which + model to route, and no harness is launched + +## MODIFIED Requirements + +### Requirement: Explaining a route + +`spinloop fleet route` SHALL report the node a harness launch would choose for a +given Spinloop, the endpoint that node resolves to, and why it was chosen — and +SHALL change nothing: it SHALL never push a config, start an engine, or write a +harness config. It is how a routing decision is checked before an agent depends +on it, and how an unexpected choice is diagnosed after one. + +When no node would be chosen, it SHALL report each node's state and the reason +it was passed over, and SHALL name what would happen on a real launch: which +node would be woken, or that none could serve it. + +The Spinloop and the fleet file SHALL resolve as they do for a launch: the +Spinloop path defaults to `./Spinloop`, the fleet file comes from +`--fleet`/`-f`, and a Spinloop the user did not name explicitly picks up a +`./fleet.yaml` in the working directory. It SHALL accept `--prefer` and +`--node` as a launch does, and SHALL name the activity preference in force — +comparing the two preferences on a live fleet is the cheapest way to decide +which one a fleet should be run with. With no fleet file in force, the command +SHALL fail naming `--fleet`. + +#### Scenario: The chosen node is explained + +- **WHEN** `spinloop fleet route` runs against a fleet with a node serving the + Spinloop's model +- **THEN** it prints that node, its resolved engine endpoint, and why it was + chosen + +#### Scenario: Routing changes nothing + +- **WHEN** `spinloop fleet route` runs against a fleet where no node is serving + the Spinloop's model +- **THEN** no engine is started, no config is pushed, and no harness config is + written + +#### Scenario: The two preferences can be compared + +- **WHEN** `spinloop fleet route --prefer active` runs against a fleet whose file + declares `prefer: idle` +- **THEN** it reports the node `active` would choose and names that preference, + without changing the fleet file + +#### Scenario: A launch that would wake a node says so + +- **WHEN** `spinloop fleet route` runs and no node is serving the model but one + could +- **THEN** it names the node a launch would wake, and does not wake it + +#### Scenario: No fleet file names the flag + +- **WHEN** `spinloop fleet route` runs on an explicitly named Spinloop in a + directory holding no `fleet.yaml`, passing no `--fleet` +- **THEN** it fails saying there is no fleet file to route through, and names + `--fleet` + +## REMOVED Requirements + +### Requirement: A fleet harness command + +**Reason**: It treated the Spinloop's `FLEET` as a fleet source, and one of its +scenarios names that instruction; both are removed. The command itself stays, +restated without the `FLEET` tier. + +**Migration**: See `The fleet harness command` in this capability. diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-config/spec.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-config/spec.md new file mode 100644 index 00000000..e852a86e --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-config/spec.md @@ -0,0 +1,33 @@ +## MODIFIED Requirements + +### Requirement: A fleet file MAY name a gateway + +A fleet file MAY declare a top-level `gateway` section, beside `wake:` and +`prefer:`, naming the address the fleet is served under by its gateway: a +`url`, and optionally a `tokenEnv` naming the variable that holds the +gateway's token. The section is how a machine that holds the fleet file +learns how to point a harness at the fleet without naming the address in a +Spinloop. + +The section's `url` SHALL carry a scheme, the way an endpoint value does, and +a `gateway` section without one SHALL be refused, naming the missing field. A +`tokenEnv` SHALL name a variable, and where the section names none, the token +SHALL be resolved under `OPENAI_API_KEY`. A file without a `gateway` section +SHALL behave exactly as it does today. + +#### Scenario: A file names its gateway + +- **WHEN** a fleet file declares a `gateway` section naming a `url` and a + `tokenEnv`, and the file is read +- **THEN** the section's address and token variable are available to the + commands that route a launch at the fleet + +#### Scenario: A section without an address is refused + +- **WHEN** a fleet file declares a `gateway` section naming no `url` +- **THEN** the file is refused, naming the missing field + +#### Scenario: No section, no change + +- **WHEN** a fleet file declares no `gateway` section +- **THEN** nothing about the file's behaviour changes diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-routing/spec.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-routing/spec.md new file mode 100644 index 00000000..cb26a01f --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/fleet-routing/spec.md @@ -0,0 +1,160 @@ +## ADDED Requirements + +### Requirement: A launch routed through a fleet + +`spinloop harness` SHALL route a worn Spinloop through a fleet when +`--fleet ` — or its `-f ` short form — is given, and, when the +worn Spinloop was not named explicitly, through a `./fleet.yaml` in the +working directory. A worn Spinloop is named explicitly when the user gives its +path — a leading positional argument or a `--spinloop`/`-O` value — or when +`SPINLOOP_ALIAS` is set; a valueless `--spinloop` wears the default Spinloop +and is not named. An explicitly named Spinloop SHALL NOT pick up a fleet file +from the working directory on its own: the user pointed at that file +deliberately, and routing happens only when `--fleet` is also given. A launch +that wears no Spinloop applies nothing and routes nothing, in a directory +holding a `fleet.yaml` or not. Where a Spinloop is worn and neither a flag nor +a directory `./fleet.yaml` is in force, the launch SHALL behave exactly as it +does without routing. + +Routing SHALL choose one node and give the launched agent that +node's engine as its OpenAI-compatible endpoint: the chosen base URL SHALL be +written as the applied provider's base URL, in the same place a `REMOTE` +endpoint's address is written, and SHALL also be placed in the launched agent's +environment as `OPENAI_BASE_URL`. + +A variable already set in spinloop's environment SHALL win, as it does on the +remote path — routing fills what is unset, it does not override an explicit +choice. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed: the pinned address wins +and spinloop SHALL say it is not routing through the fleet, rather than +silently selecting a node whose address it then discards. + +The chosen node and the reason it was chosen SHALL be reported on stderr before +the agent launches, so a launch that lands somewhere unexpected says so at the +time rather than at the first request. + +#### Scenario: A running node becomes the agent's endpoint + +- **WHEN** the user runs `spinloop harness -O` in a directory holding a + `Spinloop` and a `fleet.yaml`, and a node in that fleet is running the model + the Spinloop names +- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` pointing + at that node's engine, and the applied provider's base URL is the same + address + +#### Scenario: The flag overrides the directory's fleet file + +- **WHEN** the user runs `spinloop harness -O --fleet=./cluster.yaml` in a + directory holding a `Spinloop` and a `fleet.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates, not the ones in + the directory's `fleet.yaml` + +#### Scenario: The short form overrides the directory's fleet file + +- **WHEN** the user runs `spinloop harness -O -f ./cluster.yaml` in a directory + holding a `Spinloop` and a `fleet.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates, not the ones in + the directory's `fleet.yaml` + +#### Scenario: No fleet file in force leaves the launch local + +- **WHEN** the user runs `spinloop harness -O` in a directory holding a + `Spinloop` and no `fleet.yaml`, passing no `--fleet` +- **THEN** no fleet file is read, no node is contacted, and the launch behaves + as it did before + +#### Scenario: An explicitly named Spinloop does not route + +- **WHEN** the user runs `spinloop harness ./elsewhere/Spinloop` in a directory + holding a `fleet.yaml`, and passes no `--fleet` +- **THEN** no fleet file is read, no node is contacted, and the launch applies + the Spinloop locally + +#### Scenario: The flag routes an explicitly named Spinloop + +- **WHEN** the user runs `spinloop harness ./elsewhere/Spinloop --fleet + ./cluster.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates + +#### Scenario: An aliased Spinloop does not route + +- **WHEN** `SPINLOOP_ALIAS` is set and the user runs `spinloop harness -O` in + a directory holding a `fleet.yaml`, passing no `--fleet` +- **THEN** no fleet file is read, and the launch applies the aliased Spinloop + locally + +#### Scenario: A pinned BASEURL is not routed + +- **WHEN** a `Spinloop` beside a `fleet.yaml` names a `BASEURL`, and the user + runs `spinloop harness -O` +- **THEN** the `BASEURL` is used, no node is selected, and spinloop reports + that it is not routing through the fleet + +#### Scenario: An exported base URL wins + +- **WHEN** `OPENAI_BASE_URL` is already set in the user's environment and a + fleet-routed launch runs +- **THEN** the existing value reaches the agent unchanged + +#### Scenario: The choice is announced + +- **WHEN** a fleet-routed launch selects a node +- **THEN** the node's name, the resolved endpoint, and why it was chosen are + written to stderr before the harness is launched + +## MODIFIED Requirements + +### Requirement: A fleet file naming a gateway routes the launch at it + +A launch whose effective fleet file — one given by `--fleet`/`-f`, or the +`./fleet.yaml` in the working directory where the worn Spinloop was not named +explicitly — declares a `gateway` section SHALL route at that +gateway the way a launch routes at an endpoint: the section's address SHALL be +written as the applied provider's base URL, with the OpenAI-compatible prefix +appended when it carries no path, and SHALL be placed in the launched +agent's environment as `OPENAI_BASE_URL`. No node SHALL be contacted and none +SHALL be woken: the gateway has already done the choosing. + +The gateway's token SHALL be resolved from the variable the section names — or +from `OPENAI_API_KEY` where the section names none — through the client's +existing key chain, with a variable already set in spinloop's environment +winning. When the variable is set nowhere, the launch SHALL fail before the +harness config is written, naming the variable to set. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed at the section: the +pinned address wins, as it wins over an endpoint. + +#### Scenario: A launch is pointed at the fleet's gateway + +- **WHEN** the user runs a launch against a fleet file whose `gateway` section + names an address, and the section's token variable is set +- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` at the + section's address with the OpenAI-compatible prefix, the applied provider's + base URL is the same address, and no node is contacted + +#### Scenario: A missing gateway token fails early + +- **WHEN** a launch routes at a fleet file's `gateway` section and the + variable the section names is set nowhere +- **THEN** the launch fails before the harness config is written, naming the + variable to set + +#### Scenario: A pinned BASEURL still wins over the section + +- **WHEN** a Spinloop pins a `BASEURL` and its fleet file names a gateway +- **THEN** the `BASEURL` is used, the gateway is not, and the launch says it + is not routing + +## REMOVED Requirements + +### Requirement: A fleet-routed launch + +**Reason**: Its fleet sources were the Spinloop's `FLEET` and the flag; the +`FLEET` is removed, and the remaining sources are the flag and a `./fleet.yaml` +in the working directory for a Spinloop the user did not name. The scenarios +tied to the `FLEET` cannot stand under the new rules, so the requirement is +restated wholesale. + +**Migration**: See `A launch routed through a fleet` in this capability, which +carries the same behaviour with the new discovery rules. diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/spinloop-files/spec.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/spinloop-files/spec.md new file mode 100644 index 00000000..87229196 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/specs/spinloop-files/spec.md @@ -0,0 +1,98 @@ +## MODIFIED Requirements + +### Requirement: Spinloop file format + +A Spinloop SHALL be a flat, line-oriented text file of `KEYWORD value` +instructions. The keywords are `PROVIDER`, `MODEL`, `ALIAS`, `CONTEXT`, +`OUTPUT`, `PARALLEL`, `BASEURL` (also accepted as `BASE-URL`, `BASE_URL`, or +`URL`), `PRESET`, `REMOTE`, and `ENV`. Keywords SHALL match +case-insensitively, with UPPERCASE as the canonical form. Blank lines, full-line +`#` comments, and trailing comments introduced by whitespace-then-`#` SHALL be +ignored. Each instruction SHALL take exactly one value; every instruction SHALL +appear at most once, except `ENV`, which MAY be repeated. An `ENV` instruction's +value SHALL be a single `KEY=VALUE` token with a non-empty key and no +whitespace. `PARALLEL`'s value SHALL name a count of concurrent request slots; +like `CONTEXT`, its numeric validity (a positive integer) is enforced by the +commands that consume it rather than by parsing itself, and what it does to +the served engine's command is defined by the `local-serving` capability. +`PROVIDER` is required. Parse errors SHALL name the offending line. + +#### Scenario: A minimal Spinloop + +- **WHEN** a file containing only `PROVIDER openrouter` and + `MODEL deepseek/deepseek-v4-pro` is parsed +- **THEN** it yields a selection of that provider and model + +#### Scenario: Duplicate instruction + +- **WHEN** a Spinloop sets `MODEL` on two lines +- **THEN** parsing fails, citing both line numbers + +#### Scenario: Unknown keyword + +- **WHEN** a Spinloop contains `HARNESS pi` +- **THEN** parsing fails listing the accepted keywords + +#### Scenario: Naming a fleet + +- **WHEN** a Spinloop contains `FLEET ./fleet.yaml` +- **THEN** parsing fails as an unknown keyword, listing the accepted keywords, + and no special migration message is printed + +#### Scenario: Missing provider + +- **WHEN** a Spinloop has no `PROVIDER` instruction +- **THEN** parsing fails saying the PROVIDER instruction is missing + +#### Scenario: Naming a remote endpoint + +- **WHEN** a Spinloop contains `REMOTE ./remote.json` +- **THEN** it parses, and the value is available to the `remote` command group + +#### Scenario: Declaring local environment variables + +- **WHEN** a Spinloop contains `ENV AWS_PROFILE=dev` and `ENV AWS_REGION=eu-west-2` + on separate lines +- **THEN** it parses, yielding both key/value pairs in the selection, and the + repetition is not treated as a duplicate-instruction error + +#### Scenario: Malformed ENV value + +- **WHEN** a Spinloop contains an `ENV` instruction whose value has no `=` or an + empty key +- **THEN** parsing fails, naming the offending line + +#### Scenario: Setting the number of parallel slots + +- **WHEN** a Spinloop contains `PARALLEL 2` +- **THEN** it parses, yielding a parallel count of 2 in the selection + +#### Scenario: A non-numeric or non-positive PARALLEL is caught on use + +- **WHEN** a Spinloop contains `PARALLEL 0`, `PARALLEL -1`, or `PARALLEL abc` +- **THEN** parsing accepts the raw value, exactly as it does for `CONTEXT`, and + the command that goes on to use it (`serve`, `remote deploy`, a fleet wake) + fails naming the value, rather than silently treating it as a slot count + +## REMOVED Requirements + +### Requirement: FLEET and REMOTE are exclusive + +**Reason**: There is no `FLEET` to conflict with a `REMOTE`. + +**Migration**: Nothing to migrate; a Spinloop naming both now fails to parse on +the `FLEET` line as an unknown keyword. The statement that a pinned `BASEURL` +is not in conflict is carried by `A launch routed through a fleet` in the +`fleet-routing` capability, which covers the routing sources left. + +### Requirement: FLEET names a file or an endpoint + +**Reason**: The `FLEET` instruction is removed. Where a launch routes through +is decided by the `--fleet`/`-f` flag and, for a Spinloop the user did not name +explicitly, a `fleet.yaml` in the working directory — see `A fleet-routed +launch` in the `fleet-routing` capability. A fleet's gateway address is named +in the fleet file itself, not in a Spinloop. + +**Migration**: Delete the `FLEET` line. Put the fleet file beside the Spinloop +and run from there, or pass `--fleet `; a gateway's address goes in the +fleet file's `gateway` section. diff --git a/openspec/changes/archive/2026-09-12-remove-fleet-instruction/tasks.md b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/tasks.md new file mode 100644 index 00000000..c99daaf0 --- /dev/null +++ b/openspec/changes/archive/2026-09-12-remove-fleet-instruction/tasks.md @@ -0,0 +1,85 @@ +# Tasks + +## 1. Spinloop grammar + +- [x] 1.1 `internal/spinloop`: drop the `Fleet` field, the `kwFleet` parse + case, the REMOTE/FLEET exclusivity check, `FleetIsEndpoint`, and the + `FLEET` line in `Format`; `FLEET` now falls into the unknown-keyword + error +- [x] 1.2 `internal/spinloop` tests: remove the FLEET parse/format cases and + the REMOTE+FLEET conflict case; add a case that a `FLEET` line fails as + an unknown keyword naming the accepted keywords + +## 2. Routing discovery + +- [x] 2.1 `cmd/spinloop`: compute whether the Spinloop was named explicitly + (positional, `--spinloop` value, or `SPINLOOP_ALIAS`) and thread it + through `applyBeforeLaunch`, `routeThroughFleet`, and the `fleet route` + and `fleet harness` paths +- [x] 2.2 `routeThroughFleet`: the fleet file comes from `--fleet`/`-f`, else + `./fleet.yaml` in the working directory when the Spinloop was not + explicitly named, else no routing; delete the endpoint branch and + `fleetTarget`'s `sel.Fleet` +- [x] 2.3 `spinloop fleet route`: fleet file from `--fleet`/`-f`, else + `./fleet.yaml` for a not-explicitly-named Spinloop; with none in force, + fail naming `--fleet`; delete the endpoint reporting branch +- [x] 2.4 `spinloop fleet harness`: drop the Spinloop-`FLEET` tier — the + fleet file is `--fleet`/`-f` or the `fleet.yaml` beside the command + +## 3. Gateway command output + +- [x] 3.1 `cmd/spinloop/gateway.go`: the printed address is named in a fleet + file's `gateway` section, not a Spinloop's `FLEET` (usage text included) + +## 4. Tests + +- [x] 4.1 Update the FLEET-based launch tests in `cmd/spinloop` to the + directory convention or `--fleet`; remove the endpoint-`FLEET` tests +- [x] 4.2 New tests: an explicitly named Spinloop in a directory holding a + `fleet.yaml` does not route; `SPINLOOP_ALIAS` does not trigger the + lookup; a valueless `--spinloop` beside a `fleet.yaml` routes, while a + bare `spinloop harness` (which wears nothing) does not; `--fleet` routes + an explicit Spinloop +- [x] 4.3 `fleet route` and `fleet harness` tests: no-`FLEET` resolution and + the no-fleet-file failure naming `--fleet` + +## 5. `add-fleet-gateway` delta trim + +- [x] 5.1 Delete `openspec/changes/add-fleet-gateway/specs/spinloop-files/` +- [x] 5.2 Remove the two endpoint-`FLEET` requirements from + `openspec/changes/add-fleet-gateway/specs/fleet-routing/spec.md` +- [x] 5.3 Reword the `fleet-gateway` requirement and scenario so the printed + address is named in a fleet file's `gateway` section +- [x] 5.4 `openspec validate` passes for both changes + +## 6. Docs + +- [x] 6.1 `docs/spinloop-file.md`: drop the `FLEET` section and the + REMOTE/FLEET exclusivity; the keyword table loses `FLEET`; point readers + at the discovery rules +- [x] 6.2 `docs/commands/harness.md` and `docs/commands/fleet.md`: replace + "overrides the Spinloop's `FLEET`" with the discovery rules; the + `fleet route` no-fleet error now names `--fleet` +- [x] 6.3 `docs/commands/gateway.md`: the address goes in a fleet file's + `gateway` section +- [x] 6.4 Root `README.md`: the Spinloop example and keyword list lose + `FLEET`; the routing paragraph describes the flag and the directory + convention + +## 7. Examples + +- [x] 7.1 `examples/fleet-local`: drop the `FLEET` line from the Spinloop; + README wording for why routing happens +- [x] 7.2 `examples/fleet-docker`: drop `FLEET` from `client/Spinloop`; the + harness invocation in `run-tests.sh` passes `--fleet`; README table and + commands updated +- [x] 7.3 `examples/gateway-docker`: verify its run against the new + resolution and adjust only if it breaks + +## 8. Verify + +- [x] 8.1 `go build ./... && go vet ./... && gofmt -l internal/ cmd/` clean, + `go test ./... -cover` green with the mean at or above the bar, and the + `remote/` pnpm suite green +- [x] 8.2 `examples/fleet-docker/run-tests.sh` and + `examples/gateway-docker/run-tests.sh` green end to end diff --git a/openspec/specs/fleet-client/spec.md b/openspec/specs/fleet-client/spec.md index 0c6d137b..f4e52a26 100644 --- a/openspec/specs/fleet-client/spec.md +++ b/openspec/specs/fleet-client/spec.md @@ -547,11 +547,14 @@ When no node would be chosen, it SHALL report each node's state and the reason it was passed over, and SHALL name what would happen on a real launch: which node would be woken, or that none could serve it. -The Spinloop and the fleet file SHALL resolve as they do for a launch: the Spinloop -path defaults to `./Spinloop`, and `--fleet` overrides the Spinloop's `FLEET`. It -SHALL accept `--prefer` and `--node` as a launch does, and SHALL name the -activity preference in force — comparing the two preferences on a live fleet is -the cheapest way to decide which one a fleet should be run with. +The Spinloop and the fleet file SHALL resolve as they do for a launch: the +Spinloop path defaults to `./Spinloop`, the fleet file comes from +`--fleet`/`-f`, and a Spinloop the user did not name explicitly picks up a +`./fleet.yaml` in the working directory. It SHALL accept `--prefer` and +`--node` as a launch does, and SHALL name the activity preference in force — +comparing the two preferences on a live fleet is the cheapest way to decide +which one a fleet should be run with. With no fleet file in force, the command +SHALL fail naming `--fleet`. #### Scenario: The chosen node is explained @@ -580,6 +583,13 @@ the cheapest way to decide which one a fleet should be run with. could - **THEN** it names the node a launch would wake, and does not wake it +#### Scenario: No fleet file names the flag + +- **WHEN** `spinloop fleet route` runs on an explicitly named Spinloop in a + directory holding no `fleet.yaml`, passing no `--fleet` +- **THEN** it fails saying there is no fleet file to route through, and names + `--fleet` + ### Requirement: Fleet dashboard `spinloop fleet dashboard` SHALL open an interactive, full-screen view of the fleet: @@ -1541,28 +1551,25 @@ the sweep has already moved past as though the node were still held. - **THEN** the detail screen's deadline line is the same line, worded the same, the tile draws for the same read -### Requirement: A fleet harness command +### Requirement: The fleet harness command `spinloop fleet harness` SHALL configure the active harness for a fleet and -launch it: the fleet-level form of a fleet-routed launch, in which the fleet -file comes from the command rather than from a Spinloop's `FLEET`. The command -SHALL take a Spinloop the way `spinloop harness` does — an +launch it: the fleet-level form of a launch routed through a fleet, in which +the fleet file comes from the command rather than from the Spinloop. The +command SHALL take a Spinloop the way `spinloop harness` does — an `-O`/`--spinloop` argument, a leading alias or path, or the `Spinloop` beside it — and a fleet file from `--fleet`/`-f`, defaulting to the `fleet.yaml` beside it. -The command SHALL route the launch the way a fleet-routed launch routes: at -the gateway where the effective fleet file names one, otherwise by choosing a -node and, where the fleet's wake policy allows it, waking one — honouring -`--node`, `--prefer`, `--no-wake` and `--wake-timeout` as the launch does. The -choice SHALL be reported on stderr before the harness launches, as a -fleet-routed launch reports its choice. - -Where the command is given no `-f`, a Spinloop that names a `FLEET` — a file -or an endpoint — SHALL be used, exactly as `--fleet` overrides an instruction -on `spinloop harness`; an explicit `-f` SHALL win over the instruction. A -Spinloop that pins a `BASEURL` SHALL NOT be routed, and a variable already set -in spinloop's environment SHALL win, in each case as on the launch path. A +The command SHALL route the launch the way a launch routed through a fleet +routes: at the gateway where the effective fleet file names one, otherwise by +choosing a node and, where the fleet's wake policy allows it, waking one — +honouring `--node`, `--prefer`, `--no-wake` and `--wake-timeout` as the launch +does. The choice SHALL be reported on stderr before the harness launches, as a +launch routed through a fleet reports its choice. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed, and a variable already +set in spinloop's environment SHALL win, in each case as on the launch path. A command with no Spinloop to route SHALL fail before launching, saying that a launch needs a Spinloop to know which model to route. @@ -1579,12 +1586,12 @@ launch needs a Spinloop to know which model to route. - **WHEN** the user runs `spinloop fleet harness` against a fleet file that names no gateway, and a node is running the Spinloop's model - **THEN** the harness is applied with that node's engine as the agent's - endpoint, as a fleet-routed launch would apply it + endpoint, as a launch routed through a fleet would apply it -#### Scenario: The command's file wins over the Spinloop's FLEET +#### Scenario: The command's file wins over the fleet file beside it -- **WHEN** the user runs `spinloop fleet harness -f ./a.yaml` with a Spinloop - whose `FLEET` names a different file +- **WHEN** the user runs `spinloop fleet harness -f ./a.yaml` in a directory + holding a `./fleet.yaml` - **THEN** the fleet in `./a.yaml` is the one the launch routes through #### Scenario: No Spinloop, no route diff --git a/openspec/specs/fleet-config/spec.md b/openspec/specs/fleet-config/spec.md index 8ad38956..57dc69d9 100644 --- a/openspec/specs/fleet-config/spec.md +++ b/openspec/specs/fleet-config/spec.md @@ -321,6 +321,38 @@ given: the `file` field, a `spinloop alias` named after the node, or a alias registry, and the subdirectory convention as the three ways a source could have been given +### Requirement: A fleet file MAY name a gateway + +A fleet file MAY declare a top-level `gateway` section, beside `wake:` and +`prefer:`, naming the address the fleet is served under by its gateway: a +`url`, and optionally a `tokenEnv` naming the variable that holds the +gateway's token. The section is how a machine that holds the fleet file +learns how to point a harness at the fleet without naming the address in a +Spinloop. + +The section's `url` SHALL carry a scheme, the way an endpoint value does, and +a `gateway` section without one SHALL be refused, naming the missing field. A +`tokenEnv` SHALL name a variable, and where the section names none, the token +SHALL be resolved under `OPENAI_API_KEY`. A file without a `gateway` section +SHALL behave exactly as it does today. + +#### Scenario: A file names its gateway + +- **WHEN** a fleet file declares a `gateway` section naming a `url` and a + `tokenEnv`, and the file is read +- **THEN** the section's address and token variable are available to the + commands that route a launch at the fleet + +#### Scenario: A section without an address is refused + +- **WHEN** a fleet file declares a `gateway` section naming no `url` +- **THEN** the file is refused, naming the missing field + +#### Scenario: No section, no change + +- **WHEN** a fleet file declares no `gateway` section +- **THEN** nothing about the file's behaviour changes + ### Requirement: Fleet-wide wake policy A fleet file MAY declare a top-level `wake` value of `on` or `off`, deciding @@ -360,36 +392,3 @@ source describes the wanted model and the command that would start it. - **WHEN** a fleet file declares `wake: sometimes` - **THEN** parsing fails naming `on` and `off` - -### Requirement: A fleet file MAY name a gateway - -A fleet file MAY declare a top-level `gateway` section, beside `wake:` and -`prefer:`, naming the address the fleet is served under by its gateway: a -`url`, and optionally a `tokenEnv` naming the variable that holds the -gateway's token. The section is how a machine that holds the fleet file -learns how to point a harness at the fleet without naming the address in a -Spinloop. - -The section's `url` SHALL carry a scheme, the way an endpoint value does, and -a `gateway` section without one SHALL be refused, naming the missing field. A -`tokenEnv` SHALL name a variable, and where the section names none, the token -SHALL be resolved under `OPENAI_API_KEY`, the variable an endpoint `FLEET` -already resolves under. A file without a `gateway` section SHALL behave -exactly as it does today. - -#### Scenario: A file names its gateway - -- **WHEN** a fleet file declares a `gateway` section naming a `url` and a - `tokenEnv`, and the file is read -- **THEN** the section's address and token variable are available to the - commands that route a launch at the fleet - -#### Scenario: A section without an address is refused - -- **WHEN** a fleet file declares a `gateway` section naming no `url` -- **THEN** the file is refused, naming the missing field - -#### Scenario: No section, no change - -- **WHEN** a fleet file declares no `gateway` section -- **THEN** nothing about the file's behaviour changes diff --git a/openspec/specs/fleet-routing/spec.md b/openspec/specs/fleet-routing/spec.md index cfeedeaf..fc0fd48e 100644 --- a/openspec/specs/fleet-routing/spec.md +++ b/openspec/specs/fleet-routing/spec.md @@ -8,74 +8,6 @@ with — so a machine that can reach the fleet needs no addresses of its own. ## Requirements -### Requirement: A fleet-routed launch - -`spinloop harness` SHALL route through a fleet when the Spinloop it wears names one -with a `FLEET` instruction, or when `--fleet ` — or its `-f ` -short form — is given; the flag SHALL -override the instruction, and a launch with neither SHALL behave exactly as it -does today. Routing SHALL choose one node and give the launched agent that -node's engine as its OpenAI-compatible endpoint: the chosen base URL SHALL be -written as the applied provider's base URL, in the same place a `REMOTE` -endpoint's address is written, and SHALL also be placed in the launched agent's -environment as `OPENAI_BASE_URL`. - -A variable already set in spinloop's environment SHALL win, as it does on the -remote path — routing fills what is unset, it does not override an explicit -choice. - -A Spinloop that pins a `BASEURL` SHALL NOT be routed: the pinned address wins and -spinloop SHALL say it is not routing through the fleet, rather than silently -selecting a node whose address it then discards. - -The chosen node and the reason it was chosen SHALL be reported on stderr before -the agent launches, so a launch that lands somewhere unexpected says so at the -time rather than at the first request. - -#### Scenario: A running node becomes the agent's endpoint - -- **WHEN** the user runs `spinloop harness` with a Spinloop naming a `FLEET`, and a - node in that fleet is running the model the Spinloop names -- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` pointing - at that node's engine, and the applied provider's base URL is the same address - -#### Scenario: The flag overrides the instruction - -- **WHEN** the user runs `spinloop harness --fleet=./cluster.yaml` with a Spinloop - whose `FLEET` names a different file -- **THEN** the nodes in `./cluster.yaml` are the candidates - -#### Scenario: The short form overrides the instruction - -- **WHEN** the user runs `spinloop harness -f ./cluster.yaml` with a Spinloop - whose `FLEET` names a different file -- **THEN** the nodes in `./cluster.yaml` are the candidates - -#### Scenario: A Spinloop with no FLEET is unaffected - -- **WHEN** the user runs `spinloop harness` with a Spinloop naming no `FLEET` and - passes no `--fleet` -- **THEN** no fleet file is read, no node is contacted, and the launch behaves - as it did before - -#### Scenario: A pinned BASEURL is not routed - -- **WHEN** a Spinloop names both a `FLEET` and a `BASEURL` -- **THEN** the `BASEURL` is used, no node is selected, and spinloop reports that - it is not routing through the fleet - -#### Scenario: An exported base URL wins - -- **WHEN** `OPENAI_BASE_URL` is already set in the user's environment and a - fleet-routed launch runs -- **THEN** the existing value reaches the agent unchanged - -#### Scenario: The choice is announced - -- **WHEN** a fleet-routed launch selects a node -- **THEN** the node's name, the resolved endpoint, and why it was chosen are - written to stderr before the harness is launched - ### Requirement: Choosing a node Selection SHALL query every candidate node concurrently, as `spinloop fleet @@ -486,8 +418,9 @@ authenticate is worse than a message that says so. ### Requirement: A fleet file naming a gateway routes the launch at it -A launch whose effective fleet file — one given by `--fleet`/`-f`, or named -by the Spinloop's `FLEET` — declares a `gateway` section SHALL route at that +A launch whose effective fleet file — one given by `--fleet`/`-f`, or the +`./fleet.yaml` in the working directory where the worn Spinloop was not named +explicitly — declares a `gateway` section SHALL route at that gateway the way a launch routes at an endpoint: the section's address SHALL be written as the applied provider's base URL, with the OpenAI-compatible prefix appended when it carries no path, and SHALL be placed in the launched @@ -537,3 +470,106 @@ one. declares a `gateway` section - **THEN** the output names the section's address as the one the launch will use, no node is queried, and nothing is started + +### Requirement: A launch routed through a fleet + +`spinloop harness` SHALL route a worn Spinloop through a fleet when +`--fleet ` — or its `-f ` short form — is given, and, when the +worn Spinloop was not named explicitly, through a `./fleet.yaml` in the +working directory. A worn Spinloop is named explicitly when the user gives its +path — a leading positional argument or a `--spinloop`/`-O` value — or when +`SPINLOOP_ALIAS` is set; a valueless `--spinloop` wears the default Spinloop +and is not named. An explicitly named Spinloop SHALL NOT pick up a fleet file +from the working directory on its own: the user pointed at that file +deliberately, and routing happens only when `--fleet` is also given. A launch +that wears no Spinloop applies nothing and routes nothing, in a directory +holding a `fleet.yaml` or not. Where a Spinloop is worn and neither a flag nor +a directory `./fleet.yaml` is in force, the launch SHALL behave exactly as it +does without routing. + +Routing SHALL choose one node and give the launched agent that +node's engine as its OpenAI-compatible endpoint: the chosen base URL SHALL be +written as the applied provider's base URL, in the same place a `REMOTE` +endpoint's address is written, and SHALL also be placed in the launched agent's +environment as `OPENAI_BASE_URL`. + +A variable already set in spinloop's environment SHALL win, as it does on the +remote path — routing fills what is unset, it does not override an explicit +choice. + +A Spinloop that pins a `BASEURL` SHALL NOT be routed: the pinned address wins +and spinloop SHALL say it is not routing through the fleet, rather than +silently selecting a node whose address it then discards. + +The chosen node and the reason it was chosen SHALL be reported on stderr before +the agent launches, so a launch that lands somewhere unexpected says so at the +time rather than at the first request. + +#### Scenario: A running node becomes the agent's endpoint + +- **WHEN** the user runs `spinloop harness -O` in a directory holding a + `Spinloop` and a `fleet.yaml`, and a node in that fleet is running the model + the Spinloop names +- **THEN** the launched agent's environment carries `OPENAI_BASE_URL` pointing + at that node's engine, and the applied provider's base URL is the same + address + +#### Scenario: The flag overrides the directory's fleet file + +- **WHEN** the user runs `spinloop harness -O --fleet=./cluster.yaml` in a + directory holding a `Spinloop` and a `fleet.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates, not the ones in + the directory's `fleet.yaml` + +#### Scenario: The short form overrides the directory's fleet file + +- **WHEN** the user runs `spinloop harness -O -f ./cluster.yaml` in a directory + holding a `Spinloop` and a `fleet.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates, not the ones in + the directory's `fleet.yaml` + +#### Scenario: No fleet file in force leaves the launch local + +- **WHEN** the user runs `spinloop harness -O` in a directory holding a + `Spinloop` and no `fleet.yaml`, passing no `--fleet` +- **THEN** no fleet file is read, no node is contacted, and the launch behaves + as it did before + +#### Scenario: An explicitly named Spinloop does not route + +- **WHEN** the user runs `spinloop harness ./elsewhere/Spinloop` in a directory + holding a `fleet.yaml`, and passes no `--fleet` +- **THEN** no fleet file is read, no node is contacted, and the launch applies + the Spinloop locally + +#### Scenario: The flag routes an explicitly named Spinloop + +- **WHEN** the user runs `spinloop harness ./elsewhere/Spinloop --fleet + ./cluster.yaml` +- **THEN** the nodes in `./cluster.yaml` are the candidates + +#### Scenario: An aliased Spinloop does not route + +- **WHEN** `SPINLOOP_ALIAS` is set and the user runs `spinloop harness -O` in + a directory holding a `fleet.yaml`, passing no `--fleet` +- **THEN** no fleet file is read, and the launch applies the aliased Spinloop + locally + +#### Scenario: A pinned BASEURL is not routed + +- **WHEN** a `Spinloop` beside a `fleet.yaml` names a `BASEURL`, and the user + runs `spinloop harness -O` +- **THEN** the `BASEURL` is used, no node is selected, and spinloop reports + that it is not routing through the fleet + +#### Scenario: An exported base URL wins + +- **WHEN** `OPENAI_BASE_URL` is already set in the user's environment and a + fleet-routed launch runs +- **THEN** the existing value reaches the agent unchanged + +#### Scenario: The choice is announced + +- **WHEN** a fleet-routed launch selects a node +- **THEN** the node's name, the resolved endpoint, and why it was chosen are + written to stderr before the harness is launched diff --git a/openspec/specs/spinloop-files/spec.md b/openspec/specs/spinloop-files/spec.md index 12f154a9..73fe9099 100644 --- a/openspec/specs/spinloop-files/spec.md +++ b/openspec/specs/spinloop-files/spec.md @@ -5,13 +5,15 @@ Define the `Spinloop` file — a declarative, Dockerfile-style description of one provider selection — and the commands that consume and produce it: `spinloop apply`, `spinloop unapply`, and `spinloop export`. + ## Requirements + ### Requirement: Spinloop file format A Spinloop SHALL be a flat, line-oriented text file of `KEYWORD value` instructions. The keywords are `PROVIDER`, `MODEL`, `ALIAS`, `CONTEXT`, `OUTPUT`, `PARALLEL`, `BASEURL` (also accepted as `BASE-URL`, `BASE_URL`, or -`URL`), `PRESET`, `REMOTE`, `FLEET`, and `ENV`. Keywords SHALL match +`URL`), `PRESET`, `REMOTE`, and `ENV`. Keywords SHALL match case-insensitively, with UPPERCASE as the canonical form. Blank lines, full-line `#` comments, and trailing comments introduced by whitespace-then-`#` SHALL be ignored. Each instruction SHALL take exactly one value; every instruction SHALL @@ -39,6 +41,12 @@ the served engine's command is defined by the `local-serving` capability. - **WHEN** a Spinloop contains `HARNESS pi` - **THEN** parsing fails listing the accepted keywords +#### Scenario: Naming a fleet + +- **WHEN** a Spinloop contains `FLEET ./fleet.yaml` +- **THEN** parsing fails as an unknown keyword, listing the accepted keywords, + and no special migration message is printed + #### Scenario: Missing provider - **WHEN** a Spinloop has no `PROVIDER` instruction @@ -49,12 +57,6 @@ the served engine's command is defined by the `local-serving` capability. - **WHEN** a Spinloop contains `REMOTE ./remote.json` - **THEN** it parses, and the value is available to the `remote` command group -#### Scenario: Naming a fleet - -- **WHEN** a Spinloop contains `FLEET ./fleet.yaml` -- **THEN** it parses, and the value is available to the launch as the fleet to - route through - #### Scenario: Declaring local environment variables - **WHEN** a Spinloop contains `ENV AWS_PROFILE=dev` and `ENV AWS_REGION=eu-west-2` @@ -215,48 +217,3 @@ UPPERCASE keywords with aligned values, so `spinloop export > Spinloop` round-tr - **WHEN** the harness config has no providers - **THEN** the command fails naming the config file it read - -### Requirement: FLEET and REMOTE are exclusive - -A Spinloop SHALL NOT name both a `FLEET` and a `REMOTE`: each is a different -answer to where the model is served from — one chooses a machine on your -network, the other a deployed endpoint — and a Spinloop stating both is a mistake -rather than a precedence to resolve. Parsing SHALL fail naming both -instructions. - -An explicit `BASEURL` is not in conflict: it is the pinned address that already -takes precedence over a `REMOTE`, and it takes precedence over a `FLEET` the -same way. - -#### Scenario: Both fail to parse - -- **WHEN** a Spinloop contains both `FLEET ./fleet.yaml` and `REMOTE ./remote.json` -- **THEN** parsing fails naming both instructions - -#### Scenario: A pinned address is allowed - -- **WHEN** a Spinloop contains both `FLEET ./fleet.yaml` and a `BASEURL` -- **THEN** it parses, and the pinned base URL is what applies - -### Requirement: FLEET names a file or an endpoint - -A `FLEET` value SHALL be either a path to a fleet file, which routing reads to -choose a node, or a URL, which names an endpoint that has already done the -choosing. A value carrying a scheme SHALL be read as the latter; anything else -as a path. Both SHALL parse, so the two ways of routing are one instruction -rather than two. - -Routing through a URL SHALL fail with a message saying it is not implemented -yet, rather than being silently ignored or treated as a filename. - -#### Scenario: A path names a fleet file - -- **WHEN** a Spinloop contains `FLEET ./fleet.yaml` -- **THEN** it parses as a fleet file to route through - -#### Scenario: A URL names an endpoint - -- **WHEN** a Spinloop contains `FLEET http://gateway.internal:4000` -- **THEN** it parses as an endpoint, and a launch against it fails saying that - routing through a gateway is not implemented yet -