diff --git a/bun.lock b/bun.lock index 990ccad9e7..edcdb10734 100644 --- a/bun.lock +++ b/bun.lock @@ -200,11 +200,11 @@ "packages": { "@ai-sdk/anthropic": ["@ai-sdk/anthropic@2.0.50", "", { "dependencies": { "@ai-sdk/provider": "2.0.0", "@ai-sdk/provider-utils": "3.0.18" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-21PaHfoLmouOXXNINTsZJsMw+wE5oLR2He/1kq/sKokTVKyq7ObGT1LDk6ahwxaz/GoaNaGankMh+EgVcdv2Cw=="], - "@ai-sdk/gateway": ["@ai-sdk/gateway@4.0.70", "", { "dependencies": { "@ai-sdk/provider": "4.0.9", "@ai-sdk/provider-utils": "5.0.34", "@vercel/oidc": "3.2.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-0tzAH2vwXOs/kVktAZRS04dATEQJk1hf1QR+VuVfvo9QmW3UPgcjhhJD9QFgP8HZLxkrEGDImwLIQ7sUfQTIsA=="], + "@ai-sdk/gateway": ["@ai-sdk/gateway@4.0.69", "", { "dependencies": { "@ai-sdk/provider": "4.0.9", "@ai-sdk/provider-utils": "5.0.34", "@vercel/oidc": "3.2.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-W5MMdyqsaziQy/A4kxlK74iEQ+NuO6OaszH32cEQpUgBW0o15S2fAdP0aYSH2/5lrVZSMXLQLCzuMkRGHBua3A=="], "@ai-sdk/provider": ["@ai-sdk/provider@2.0.3", "", { "dependencies": { "json-schema": "^0.4.0" } }, "sha512-h88OPkavHTiN9tMn2l5awAznGB0lXzjcLhgR1/rvjB2zlLprsNxbM2tt6OJsHUxduLC3klq0/eqaSf6fX5XVww=="], - "@ai-sdk/provider-utils": ["@ai-sdk/provider-utils@3.0.36", "", { "dependencies": { "@ai-sdk/provider": "2.0.3", "@standard-schema/spec": "^1.0.0", "eventsource-parser": "^3.0.6", "undici": "^5.29.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-2eSw90hn32Je6n2a8Gf4dJ2EoecPJuOCWqwZCw+BkhPq2LOS01HX3s6ljgOm0iIkZiD5aAuMdpOw17rYKQF/Zg=="], + "@ai-sdk/provider-utils": ["@ai-sdk/provider-utils@3.0.35", "", { "dependencies": { "@ai-sdk/provider": "2.0.3", "@standard-schema/spec": "^1.0.0", "eventsource-parser": "^3.0.6", "undici": "^5.29.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-/5z8tRGuYXwFy0ID+WtiWiECJzH5x/rI/g/3H8x3GQvE4i4etnZfKWAiU23VZEymInh+4l7uMNQJyaNN/54QFw=="], "@auth/core": ["@auth/core@0.41.3", "", { "dependencies": { "@panva/hkdf": "^1.2.1", "jose": "^6.0.6", "oauth4webapi": "^3.3.0", "preact": "10.24.3", "preact-render-to-string": "6.5.11" }, "peerDependencies": { "@simplewebauthn/browser": "^9.0.1", "@simplewebauthn/server": "^9.0.2", "nodemailer": "^7.0.7 || ^8.0.5" }, "optionalPeers": ["@simplewebauthn/browser", "@simplewebauthn/server", "nodemailer"] }, "sha512-sJ3JMHHkXMD3aOjopv7mOBTO1Ocw4b0fAEXJBz6k7YHLpYQI6C40jCUPc5fNvUKxXRXNE1/sRISA15UrwWJBTw=="], @@ -390,23 +390,23 @@ "@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.30.0", "", { "dependencies": { "@hono/node-server": "^1.19.9 || ^2.0.5", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-xKd8OIzlqNzcqcNumGAa6g+PW2kjD5vrpcKOnfldAUPP3j7lnqMPwlTXQm8gF+UwH72z0lqaRbjr9hqGz0eITA=="], - "@next/env": ["@next/env@16.3.4", "", {}, "sha512-cjWZnUUa6jZq2kFaNe/ZyJdZonOZ/QoN0Zka2nz/FLOrfx14pQuM9c5RaSVkWMqgdt4ksgPAMWPyHSs/CyV48Q=="], + "@next/env": ["@next/env@16.3.3", "", {}, "sha512-U2eYQRwXj+dsqxV79zFqExDdatnNY/ZWc2nsJU1p/OgT7fd3dXwlF6OjYaFQCfMoeTA19PWq+wVmYgimVA+V+g=="], - "@next/swc-darwin-arm64": ["@next/swc-darwin-arm64@16.3.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-iBr3I5LZNk5/bgl5//iTgD2tcym14MX0Xo7fD//u9dYAEgGzza1y9oywluPtf74YnOswVdH1908aK9xVz7zQTw=="], + "@next/swc-darwin-arm64": ["@next/swc-darwin-arm64@16.3.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-8Hiv32QJPwdV6KYJ8meR9SBA061tQqnIKTJDocvOXlEQqib0xMFpzArosuffFUUc0sslbh7QQ8a3Yey1QV8EIw=="], - "@next/swc-darwin-x64": ["@next/swc-darwin-x64@16.3.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-2dpiSyl2Jw/NrBPaU2MAKGSa+2MR82pJIn4Sm5Rjr+gxAeuh0z158Su3Z2O8zn7UNNq+ej4bToed6RcRN/Lydg=="], + "@next/swc-darwin-x64": ["@next/swc-darwin-x64@16.3.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-A1lgKgwVchRYmSe467zdwhxT9040dd8lH+o65sL5Jet8fjB4kegw/rDyPIpYVRb6jAqwXFOJpjIXJLxQKLiE3A=="], - "@next/swc-linux-arm64-gnu": ["@next/swc-linux-arm64-gnu@16.3.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-+t+U8HZT+fApePCS5h89CSH3datz29MkzyfCn+6fpsZBG/oiEOhINcb9rtkv6sdpToLGFn2e6146NzaKCXkqrA=="], + "@next/swc-linux-arm64-gnu": ["@next/swc-linux-arm64-gnu@16.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-bf0FIssMFueU2dm7vQEWWxk0c8UjKTdW0yzuh0sQsD8pf1+KCLDdaqhYZNMYGmXwEOiHAUzgBKudovIlcvvBjg=="], - "@next/swc-linux-arm64-musl": ["@next/swc-linux-arm64-musl@16.3.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-mx03GNs1ocQA5JQ4FxDMmIsNkdrZh8cuezKCrId28e5/gIPU/l7Kcy2+vmCCzdjnnmXJy+iOAu+7K0QppO6Urg=="], + "@next/swc-linux-arm64-musl": ["@next/swc-linux-arm64-musl@16.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-W7viwCk9JY/cAkdz/A273rd5bb3RgT/IHwR7Upv90tunjBWNtAAhGhoecHh+teRNRSinuAFmE+l7fwZ4YKkrXg=="], - "@next/swc-linux-x64-gnu": ["@next/swc-linux-x64-gnu@16.3.4", "", { "os": "linux", "cpu": "x64" }, "sha512-YIhGY6fSMfha52bnVxnzc9zaVBzJg+cqQTOD8tXIBSx4fuv0pVMxQTE0PaS59YhnMOiYiG09IMwxJAf/CFm/Dw=="], + "@next/swc-linux-x64-gnu": ["@next/swc-linux-x64-gnu@16.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-0W46zw1N3ODpI6n0GeivHvvob1pooozgZVqy65k0mh4/7vr+FbY9+WpHzNVXjHipJf/A3FDheBG19H1s5A25rA=="], - "@next/swc-linux-x64-musl": ["@next/swc-linux-x64-musl@16.3.4", "", { "os": "linux", "cpu": "x64" }, "sha512-+eaaX6axpDb0yF1GCpiERe6njplvdC+nks/fKfcHu3XPGRrald8P3/X7yv7QLdjA51knnxwl9pxdIJsg+w1L+Q=="], + "@next/swc-linux-x64-musl": ["@next/swc-linux-x64-musl@16.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-H4mBso8ZTMBPtdT0PN0pBx2ayTvQuTuvS6qT13d77yVFJXAPCxkyIhLTmdMaGTJs0krQYI/qpzdHijCeihXhbg=="], - "@next/swc-win32-arm64-msvc": ["@next/swc-win32-arm64-msvc@16.3.4", "", { "os": "win32", "cpu": "arm64" }, "sha512-0jcXW7Xs/uzICrmgV3MhDYDeRy++1CqnpDIerlPIqYO4bhzB4WNbX/aRnQclustsAyTkFKB0z6rbcjmNg5tR8A=="], + "@next/swc-win32-arm64-msvc": ["@next/swc-win32-arm64-msvc@16.3.3", "", { "os": "win32", "cpu": "arm64" }, "sha512-cTMUJpcEGmeywofCUfhR+rSsoE33+rVPnPEYNTNdLNlsOeEg/vktOsKUSTb28vUGqD2jkm4Zaskcwn7OCI6FQg=="], - "@next/swc-win32-x64-msvc": ["@next/swc-win32-x64-msvc@16.3.4", "", { "os": "win32", "cpu": "x64" }, "sha512-vvBzwu1pYQCp92maZCFCIw/XgOTMR5tur9GjakwIo2cmwRTMKajRZZDS9+e4KsUZWKu1E007WUeAFXRRjZeuzw=="], + "@next/swc-win32-x64-msvc": ["@next/swc-win32-x64-msvc@16.3.3", "", { "os": "win32", "cpu": "x64" }, "sha512-2VR4cTBzHXaBjnGsuH6GyJjENzQOmHeAh11uY1iUhjm3j5dEUrVJuUj+VL78jaGi/Dik8xS76zEj18BsFhlVZQ=="], "@nodelib/fs.scandir": ["@nodelib/fs.scandir@2.1.5", "", { "dependencies": { "@nodelib/fs.stat": "2.0.5", "run-parallel": "^1.1.9" } }, "sha512-vq24Bq3ym5HEQm2NKCr3yXDwjc7vTsEThRDnkp2DK9p1uqLR+DHurm/NOTo0KG7HYHU7eppKZj3MyqYuMBf62g=="], @@ -550,7 +550,7 @@ "agent-base": ["agent-base@6.0.2", "", { "dependencies": { "debug": "4" } }, "sha512-RZNwNclF7+MS/8bDg70amg32dyeZGZxiDuQmZxKLAlQjr3jGyLx+4Kkk58UO7D2QdgFIQCovuSuZESne6RG6XQ=="], - "ai": ["ai@7.0.86", "", { "dependencies": { "@ai-sdk/gateway": "4.0.70", "@ai-sdk/provider": "4.0.9", "@ai-sdk/provider-utils": "5.0.34" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-11Hovs3BI98tPJiOuA85Be+ktxbZ2QUIqqLJqfHJ55zz4106pjRkEP9OQ95glyjBXPEtTdr6/z4ISsk6G13rvw=="], + "ai": ["ai@7.0.85", "", { "dependencies": { "@ai-sdk/gateway": "4.0.69", "@ai-sdk/provider": "4.0.9", "@ai-sdk/provider-utils": "5.0.34" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-HVtPz0qLbTUad+QBnWWReIUmwk+U4PcRENMx+9PsHGVoinoc5CLDiVjNR+VBXTKOSaNgce8kSU/Rtbb4kjZsSw=="], "ajv": ["ajv@8.20.0", "", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA=="], @@ -1186,7 +1186,7 @@ "negotiator": ["negotiator@1.1.0", "", { "dependencies": { "content-type": "^2.1.0" } }, "sha512-NMPBRMJgiQHjbd8phG3Vebdx4kZ1H121rbl5IkMqeOsahptB9BKo/d7oJ3zTXqTgagn2bWlNSXkh0QUGM31RYg=="], - "next": ["next@16.3.4", "", { "dependencies": { "@next/env": "16.3.4", "@swc/helpers": "0.5.23", "baseline-browser-mapping": "^2.9.19", "caniuse-lite": "^1.0.30001579", "postcss": "8.5.23", "styled-jsx": "5.1.6" }, "optionalDependencies": { "@next/swc-darwin-arm64": "16.3.4", "@next/swc-darwin-x64": "16.3.4", "@next/swc-linux-arm64-gnu": "16.3.4", "@next/swc-linux-arm64-musl": "16.3.4", "@next/swc-linux-x64-gnu": "16.3.4", "@next/swc-linux-x64-musl": "16.3.4", "@next/swc-win32-arm64-msvc": "16.3.4", "@next/swc-win32-x64-msvc": "16.3.4", "sharp": "^0.35.4" }, "peerDependencies": { "@opentelemetry/api": "^1.1.0", "@playwright/test": "^1.51.1", "babel-plugin-react-compiler": "*", "react": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "react-dom": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "sass": "^1.3.0" }, "optionalPeers": ["@opentelemetry/api", "@playwright/test", "babel-plugin-react-compiler", "sass"], "bin": { "next": "dist/bin/next" } }, "sha512-/Ztf6CeRH+ejEXUrYtqI4gkS66eFIHuSwqi60RgcpWKodxFZx2/dqVCMKBwILfAHXQ+F1b1vAudgj3mnxqtoIA=="], + "next": ["next@16.3.3", "", { "dependencies": { "@next/env": "16.3.3", "@swc/helpers": "0.5.23", "baseline-browser-mapping": "^2.9.19", "caniuse-lite": "^1.0.30001579", "postcss": "8.5.23", "styled-jsx": "5.1.6" }, "optionalDependencies": { "@next/swc-darwin-arm64": "16.3.3", "@next/swc-darwin-x64": "16.3.3", "@next/swc-linux-arm64-gnu": "16.3.3", "@next/swc-linux-arm64-musl": "16.3.3", "@next/swc-linux-x64-gnu": "16.3.3", "@next/swc-linux-x64-musl": "16.3.3", "@next/swc-win32-arm64-msvc": "16.3.3", "@next/swc-win32-x64-msvc": "16.3.3", "sharp": "^0.35.3" }, "peerDependencies": { "@opentelemetry/api": "^1.1.0", "@playwright/test": "^1.51.1", "babel-plugin-react-compiler": "*", "react": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "react-dom": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "sass": "^1.3.0" }, "optionalPeers": ["@opentelemetry/api", "@playwright/test", "babel-plugin-react-compiler", "sass"], "bin": { "next": "dist/bin/next" } }, "sha512-tuRTx1nQ/yVw83cwJBo9F+njGUgMn3UHQycreWHB8XsStvvAh1AthbI8/4IpKnFaF58F+iSiHejYOlMQ/eq83g=="], "next-auth": ["next-auth@4.24.15", "", { "dependencies": { "@babel/runtime": "^7.20.13", "@panva/hkdf": "^1.0.2", "cookie": "^0.7.0", "jose": "^4.15.5", "oauth": "^0.9.15", "openid-client": "^5.4.0", "preact": "^10.6.3", "preact-render-to-string": "^5.1.19", "uuid": "^11.1.1" }, "peerDependencies": { "@auth/core": "0.34.3", "next": "^12.2.5 || ^13 || ^14 || ^15 || ^16", "nodemailer": "^7.0.7", "react": "^17.0.2 || ^18 || ^19", "react-dom": "^17.0.2 || ^18 || ^19" }, "optionalPeers": ["@auth/core", "nodemailer"] }, "sha512-NnjYtjrSOAx/TIVFGTX4IfI/9yHnNpi4B7FuLUwuV20v2Zxgr2OGP/YN0ynJuI7y8QOnTBPitfOdEXZrVvhIuA=="], @@ -1542,7 +1542,7 @@ "ts-pattern": ["ts-pattern@5.9.0", "", {}, "sha512-6s5V71mX8qBUmlgbrfL33xDUwO0fq48rxAu2LBE11WBeGdpCPOsXksQbZJHvHwhrd3QjUusd3mAOM5Gg0mFBLg=="], - "tsc-alias": ["tsc-alias@1.9.3", "", { "dependencies": { "chokidar": "^3.5.3", "commander": "^9.0.0", "get-tsconfig": "^4.10.0", "globby": "^11.0.4", "mylas": "^2.1.9", "normalize-path": "^3.0.0", "plimit-lit": "^1.2.6" }, "bin": { "tsc-alias": "dist/bin/index.js" } }, "sha512-GKrkA/K5hwae80rlfJRazukMMMIUsIHRyb75lbEp+qaUP57sYmur2Z05dosZNBspByX3ZrxbLHkMRgLfVuUcYg=="], + "tsc-alias": ["tsc-alias@1.9.2", "", { "dependencies": { "chokidar": "^3.5.3", "commander": "^9.0.0", "get-tsconfig": "^4.10.0", "globby": "^11.0.4", "mylas": "^2.1.9", "normalize-path": "^3.0.0", "plimit-lit": "^1.2.6" }, "bin": { "tsc-alias": "dist/bin/index.js" } }, "sha512-VTWQGMv0xXCEyHDLpmV2DEvGYHMxwsyx87dZeou2ynkM0+WOFdHe+KWiRucavMPUEdQysr7xSu60Y/WY0R4YKA=="], "tsconfig-paths": ["tsconfig-paths@4.2.0", "", { "dependencies": { "json5": "^2.2.2", "minimist": "^1.2.6", "strip-bom": "^3.0.0" } }, "sha512-NoZ4roiN7LnbKn9QqE1amc9DJfzvZXxF4xDavcOWt1BPkdx+m+0gJuPM+S0vCe7zTJMYUP0R8pO2XMr+Y8oLIg=="], @@ -1688,8 +1688,6 @@ "@opentui/react/react-reconciler": ["react-reconciler@0.33.0", "", { "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { "react": "^19.2.0" } }, "sha512-KetWRytFv1epdpJc3J4G75I4WrplZE5jOL7Yq0p34+OVOKF4Se7WrdIdVC45XsSSmUTlht2FM/fM1FZb1mfQeA=="], - "@types/diff/diff": ["diff@9.0.0", "", {}, "sha512-svtcdpS8CgJyqAjEQIXdb3OjhFVVYjzGAPO8WGCmRbrml64SPw/jJD4GoE98aR7r25A0XcgrK3F02yw9R/vhQw=="], - "@typescript-eslint/eslint-plugin/ignore": ["ignore@5.3.2", "", {}, "sha512-hsBTNUqQTDwkWtcdYI2i06Y/nUBEsNEDJKjWdigLvegy8kDuJAS8uRlpkkcQpyEXL0Z/pjDy5HBmMjRCJ2gq+g=="], "@typescript-eslint/parser/@typescript-eslint/scope-manager": ["@typescript-eslint/scope-manager@7.18.0", "", { "dependencies": { "@typescript-eslint/types": "7.18.0", "@typescript-eslint/visitor-keys": "7.18.0" } }, "sha512-jjhdIE/FPF2B7Z1uzc6i3oWKbGcHb87Qw7AWj6jmEqNOfDFbJWtjt/XfwCpvNkpGWlcJaog5vTR+VV8+w9JflA=="], diff --git a/cli/src/chat.tsx b/cli/src/chat.tsx index 8a2770b575..88f6662fe8 100644 --- a/cli/src/chat.tsx +++ b/cli/src/chat.tsx @@ -535,7 +535,6 @@ export const Chat = ({ logoutMutation, streamMessageIdRef, addToQueue, - hasQueuedMessages: () => queuedCount > 0, clearMessages, saveToHistory, scrollToLatest, diff --git a/cli/src/commands/__tests__/router-steering.test.ts b/cli/src/commands/__tests__/router-steering.test.ts deleted file mode 100644 index cb15dcf92b..0000000000 --- a/cli/src/commands/__tests__/router-steering.test.ts +++ /dev/null @@ -1,128 +0,0 @@ -import { afterEach, beforeEach, describe, expect, mock, test } from 'bun:test' - -import { useChatStore } from '../../state/chat-store' -import { - __resetSteeringForTests, - activateSteering, - drainSteeringMessages, -} from '../../utils/steering-buffer' -import { routeUserPrompt } from '../router' - -import type { RouterParams } from '../command-registry' - -const createMockParams = (overrides: Partial = {}): RouterParams => - ({ - agentMode: 'DEFAULT', - inputRef: { current: null }, - inputValue: '', - isChainInProgressRef: { current: false }, - isStreaming: false, - logoutMutation: {} as RouterParams['logoutMutation'], - streamMessageIdRef: { current: null }, - addToQueue: mock(() => {}), - hasQueuedMessages: () => false, - clearMessages: mock(() => {}), - saveToHistory: mock(() => {}), - scrollToLatest: mock(() => {}), - sendMessage: mock(async () => {}), - setCanProcessQueue: mock(() => {}), - setInputFocused: mock(() => {}), - setInputValue: mock(() => {}), - setIsAuthenticated: mock(() => {}), - setMessages: mock(() => {}), - setUser: mock(() => {}), - ...overrides, - }) as RouterParams - -beforeEach(() => { - useChatStore.getState().clearPendingBashMessages() -}) - -afterEach(() => { - __resetSteeringForTests() - useChatStore.getState().clearPendingBashMessages() -}) - -describe('mid-turn routing', () => { - test('plain text steers the active run and echoes a bubble immediately', async () => { - activateSteering('run-1') - const params = createMockParams({ - inputValue: 'actually use zod for validation', - isStreaming: true, - }) - await routeUserPrompt(params) - - expect(params.addToQueue).not.toHaveBeenCalled() - expect(params.sendMessage).not.toHaveBeenCalled() - // Bubble echoed at push time so the submit is visible right away. - expect(params.setMessages).toHaveBeenCalledTimes(1) - const drained = drainSteeringMessages('run-1') - expect(drained.map((entry) => entry.text)).toEqual([ - 'actually use zod for validation', - ]) - expect(drained[0]!.messageId).toStartWith('user-') - }) - - test('falls back to the queue when no run is accepting steering', async () => { - const params = createMockParams({ - inputValue: 'between chained runs', - isStreaming: true, - }) - await routeUserPrompt(params) - - expect(params.addToQueue).toHaveBeenCalledTimes(1) - const [queued] = (params.addToQueue as ReturnType).mock - .calls[0] as [string] - expect(queued).toBe('between chained runs') - }) - - test('queues instead of steering when earlier messages are already queued', async () => { - activateSteering('run-1') - const params = createMockParams({ - inputValue: 'this must not overtake the queue', - isStreaming: true, - hasQueuedMessages: () => true, - }) - await routeUserPrompt(params) - - expect(drainSteeringMessages('run-1')).toEqual([]) - expect(params.addToQueue).toHaveBeenCalledTimes(1) - }) - - test('queues instead of steering while bash output is pending', async () => { - activateSteering('run-1') - useChatStore.getState().addPendingBashMessage({ - command: 'bun test', - output: '3 fail', - } as never) - const params = createMockParams({ - inputValue: 'fix those failures', - isStreaming: true, - }) - await routeUserPrompt(params) - - expect(drainSteeringMessages('run-1')).toEqual([]) - expect(params.addToQueue).toHaveBeenCalledTimes(1) - }) - - test('slash commands never steer', async () => { - activateSteering('run-1') - const params = createMockParams({ - inputValue: '/definitely-not-a-command', - isStreaming: true, - }) - await routeUserPrompt(params) - - expect(drainSteeringMessages('run-1')).toEqual([]) - expect(params.addToQueue).toHaveBeenCalledTimes(1) - }) - - test('idle submits are unaffected and send normally', async () => { - activateSteering('run-1') - const params = createMockParams({ inputValue: 'a fresh task' }) - await routeUserPrompt(params) - - expect(params.sendMessage).toHaveBeenCalledTimes(1) - expect(drainSteeringMessages('run-1')).toEqual([]) - }) -}) diff --git a/cli/src/commands/__tests__/skill-command.test.ts b/cli/src/commands/__tests__/skill-command.test.ts deleted file mode 100644 index 9ff609622a..0000000000 --- a/cli/src/commands/__tests__/skill-command.test.ts +++ /dev/null @@ -1,156 +0,0 @@ -import { afterEach, beforeEach, describe, expect, mock, test } from 'bun:test' - -import { useChatStore } from '../../state/chat-store' -import { - __resetSkillRegistryForTests, - __setSkillsForTests, -} from '../../utils/skill-registry' -import { findCommand } from '../command-registry' -import { buildSkillPrompt } from '../prompt-builders' -import { routeUserPrompt } from '../router' - -import type { RouterParams } from '../command-registry' -import type { SkillDefinition } from '@codebuff/common/types/skill' - -const TEST_SKILL: SkillDefinition = { - name: 'release-notes', - description: 'Draft release notes from recent commits', - content: - '---\nname: release-notes\ndescription: Draft release notes\n---\n\nDo the thing.', - filePath: '/tmp/skills/release-notes/SKILL.md', -} - -const createMockParams = (overrides: Partial = {}): RouterParams => - ({ - agentMode: 'DEFAULT', - inputRef: { current: null }, - inputValue: '', - isChainInProgressRef: { current: false }, - isStreaming: false, - logoutMutation: {} as RouterParams['logoutMutation'], - streamMessageIdRef: { current: null }, - addToQueue: mock(() => {}), - clearMessages: mock(() => {}), - saveToHistory: mock(() => {}), - scrollToLatest: mock(() => {}), - sendMessage: mock(async () => {}), - setCanProcessQueue: mock(() => {}), - setInputFocused: mock(() => {}), - setInputValue: mock(() => {}), - setIsAuthenticated: mock(() => {}), - setMessages: mock(() => {}), - setUser: mock(() => {}), - ...overrides, - }) as RouterParams - -const resetChatStore = () => { - useChatStore.getState().setInputMode('default') - useChatStore.getState().setPendingSkillName(null) -} - -beforeEach(() => { - __setSkillsForTests({ [TEST_SKILL.name]: TEST_SKILL }) - resetChatStore() -}) - -afterEach(() => { - __resetSkillRegistryForTests() - resetChatStore() -}) - -describe('/skill: command', () => { - test('bare invocation enters skill input mode instead of sending', async () => { - const command = findCommand('skill:release-notes') - expect(command).toBeDefined() - - const params = createMockParams({ inputValue: '/skill:release-notes' }) - await command!.handler(params, '') - - expect(useChatStore.getState().inputMode).toBe('skill') - expect(useChatStore.getState().pendingSkillName).toBe('release-notes') - expect(params.sendMessage).not.toHaveBeenCalled() - expect(params.addToQueue).not.toHaveBeenCalled() - }) - - test('invocation with trailing text sends immediately', async () => { - const command = findCommand('skill:release-notes') - const params = createMockParams({ - inputValue: '/skill:release-notes for v2.1 only', - }) - await command!.handler(params, 'for v2.1 only') - - expect(useChatStore.getState().inputMode).toBe('default') - expect(params.sendMessage).toHaveBeenCalledTimes(1) - const [{ content }] = (params.sendMessage as ReturnType).mock - .calls[0] as [{ content: string }] - expect(content).toBe(buildSkillPrompt(TEST_SKILL, 'for v2.1 only')) - expect(content).toContain('') - expect(content).toContain('User request: for v2.1 only') - }) -}) - -describe('skill input mode submit', () => { - const enterSkillMode = () => { - useChatStore.getState().setInputMode('skill') - useChatStore.getState().setPendingSkillName(TEST_SKILL.name) - } - - test('submit with text sends the skill plus the user request', async () => { - enterSkillMode() - const params = createMockParams({ inputValue: 'focus on the API changes' }) - await routeUserPrompt(params) - - expect(useChatStore.getState().inputMode).toBe('default') - expect(useChatStore.getState().pendingSkillName).toBeNull() - expect(params.sendMessage).toHaveBeenCalledTimes(1) - const [{ content }] = (params.sendMessage as ReturnType).mock - .calls[0] as [{ content: string }] - expect(content).toBe( - buildSkillPrompt(TEST_SKILL, 'focus on the API changes'), - ) - }) - - test('empty submit runs the skill without a user request', async () => { - enterSkillMode() - const params = createMockParams({ inputValue: '' }) - await routeUserPrompt(params) - - expect(params.sendMessage).toHaveBeenCalledTimes(1) - const [{ content }] = (params.sendMessage as ReturnType).mock - .calls[0] as [{ content: string }] - expect(content).toBe(buildSkillPrompt(TEST_SKILL, '')) - expect(content).not.toContain('User request:') - }) - - test('submit while a turn is running queues instead of sending', async () => { - enterSkillMode() - const params = createMockParams({ - inputValue: 'and be brief', - isStreaming: true, - }) - await routeUserPrompt(params) - - expect(params.sendMessage).not.toHaveBeenCalled() - expect(params.addToQueue).toHaveBeenCalledTimes(1) - const [queued] = (params.addToQueue as ReturnType).mock - .calls[0] as [string] - expect(queued).toBe(buildSkillPrompt(TEST_SKILL, 'and be brief')) - }) - - test('a skill deleted mid-session reports instead of sending nothing', async () => { - enterSkillMode() - __resetSkillRegistryForTests() - const params = createMockParams({ inputValue: 'anything' }) - await routeUserPrompt(params) - - expect(params.sendMessage).not.toHaveBeenCalled() - expect(params.setMessages).toHaveBeenCalled() - expect(useChatStore.getState().inputMode).toBe('default') - }) - - test('leaving skill mode clears the pending skill', () => { - enterSkillMode() - useChatStore.getState().setInputMode('default') - expect(useChatStore.getState().pendingSkillName).toBeNull() - }) -}) diff --git a/cli/src/commands/command-registry.ts b/cli/src/commands/command-registry.ts index b308c7bb91..18d0eea0d3 100644 --- a/cli/src/commands/command-registry.ts +++ b/cli/src/commands/command-registry.ts @@ -9,7 +9,7 @@ import { collectProcessDiagnostics, formatProcessDiagnostics, } from './process-diagnostics' -import { buildInterviewPrompt, buildPlanPrompt, buildReviewPromptFromArgs, buildSkillPrompt } from './prompt-builders' +import { buildInterviewPrompt, buildPlanPrompt, buildReviewPromptFromArgs } from './prompt-builders' import { handleReasoningCommand } from './reasoning' import { runBashCommand } from './router' import { handleUsageCommand } from './usage' @@ -44,9 +44,6 @@ export type RouterParams = { logoutMutation: UseMutationResult streamMessageIdRef: React.MutableRefObject addToQueue: (message: string, attachments?: PendingAttachment[]) => void - /** Whether the message queue currently holds anything. Steering checks it - * so a mid-turn submit can't overtake earlier queued submissions. */ - hasQueuedMessages?: () => boolean clearMessages: () => void saveToHistory: (message: string) => void scrollToLatest: () => void @@ -718,50 +715,36 @@ function createSkillCommand(skillName: string): CommandDefinition { params.saveToHistory(trimmed) params.setInputValue({ text: '', cursorPosition: 0, lastEditDueToNav: false }) - // Bare invocation: like /interview, drop into an input mode so the - // user can add instructions before the skill is sent. Enter with an - // empty composer still runs the skill as-is (the router's skill-mode - // branch), so a no-args run costs one extra keystroke, not a feature. - if (!args.trim()) { - useChatStore.getState().enterSkillMode(skill.name) + // Build the message content with skill context and optional user args + const skillContext = ` +${skill.content} +` + + const userPrompt = `I invoke the following skill:\n\n${skillContext}\n\n` + + (args.trim() + ? `User request: ${args.trim()}` + : '') + + // Check streaming/queue state + if ( + params.isStreaming || + params.streamMessageIdRef.current || + params.isChainInProgressRef.current + ) { + const pendingAttachments = capturePendingAttachments() + params.addToQueue(userPrompt, pendingAttachments) params.setInputFocused(true) params.inputRef.current?.focus() return } - dispatchSkillPrompt(params, skill, args) + params.sendMessage({ + content: userPrompt, + agentMode: params.agentMode, + }) + setTimeout(() => { + params.scrollToLatest() + }, 0) }, }) } - -/** - * Send (or queue, mid-turn) a user-invoked skill prompt. Shared by the - * /skill: args form and the skill input mode's submit (router), so the - * two entry paths for the same feature cannot drift. - */ -export function dispatchSkillPrompt( - params: RouterParams, - skill: { name: string; content: string }, - input: string, -): void { - const userPrompt = buildSkillPrompt(skill, input) - - if ( - params.isStreaming || - params.streamMessageIdRef.current || - params.isChainInProgressRef.current - ) { - params.addToQueue(userPrompt, capturePendingAttachments()) - params.setInputFocused(true) - params.inputRef.current?.focus() - return - } - - params.sendMessage({ - content: userPrompt, - agentMode: params.agentMode, - }) - setTimeout(() => { - params.scrollToLatest() - }, 0) -} diff --git a/cli/src/commands/prompt-builders.ts b/cli/src/commands/prompt-builders.ts index 2435238212..4dc9979778 100644 --- a/cli/src/commands/prompt-builders.ts +++ b/cli/src/commands/prompt-builders.ts @@ -39,26 +39,6 @@ export function buildInterviewPrompt(input: string): string { return `${INTERVIEW_BASE_PROMPT}\n\n${trimmedInput}` } -/** - * Build the prompt for a user-invoked skill. Shared by the /skill: - * command (when it carries trailing text) and the skill input mode's second - * submit, so both entry paths produce byte-identical prompts. - * - * `content` is the whole SKILL.md (frontmatter included) — same as the - * agent-runtime's own skill tool output. - */ -export function buildSkillPrompt( - skill: { name: string; content: string }, - input: string, -): string { - const skillContext = `\n${skill.content}\n` - const trimmedInput = input.trim() - return ( - `I invoke the following skill:\n\n${skillContext}\n\n` + - (trimmedInput ? `User request: ${trimmedInput}` : '') - ) -} - /** * Review scope presets for the review screen. */ diff --git a/cli/src/commands/router.ts b/cli/src/commands/router.ts index 7453034ebe..d9b08aa766 100644 --- a/cli/src/commands/router.ts +++ b/cli/src/commands/router.ts @@ -3,7 +3,6 @@ import { runTerminalCommand } from '@codebuff/sdk' import { - dispatchSkillPrompt, findCommand, type RouterParams, type CommandResult, @@ -26,8 +25,6 @@ import { IS_FREEBUFF } from '../utils/constants' import { getSystemProcessEnv } from '../utils/env' import { terminalCommandBroker } from '../utils/terminal-command-broker' import { getSystemMessage, getUserMessage } from '../utils/message-history' -import { getSkillByName } from '../utils/skill-registry' -import { pushSteeringMessage } from '../utils/steering-buffer' import { capturePendingAttachments, hasProcessingFiles, @@ -261,7 +258,6 @@ export async function routeUserPrompt( isStreaming, streamMessageIdRef, addToQueue, - hasQueuedMessages, saveToHistory, scrollToLatest, sendMessage, @@ -276,10 +272,9 @@ export async function routeUserPrompt( const pendingImages = pendingAttachments.filter((a) => a.kind === 'image') const trimmed = inputValue.trim() - // Allow empty messages if there are pending attachments (images or text). - // Skill mode also accepts an empty submit: it means "run the skill as-is". + // Allow empty messages if there are pending attachments (images or text) const hasAttachments = pendingAttachments.length > 0 - if (!trimmed && !hasAttachments && inputMode !== 'skill') return + if (!trimmed && !hasAttachments) return // DAU signal: one un-sampled event per user-submitted prompt. The CLI's // distinct id resolves to the canonical codebuff user id (anonymous id is @@ -348,38 +343,6 @@ export async function routeUserPrompt( return } - // Handle skill mode input: the user picked a skill (bare /skill:) - // and is now adding instructions. Empty input runs the skill without any. - if (inputMode === 'skill') { - const skillName = useChatStore.getState().pendingSkillName - const skill = skillName ? getSkillByName(skillName) : undefined - - if (!skill) { - // Mode without a resolvable skill (state got out of sync): explain, - // and leave the user's typed text in the composer rather than - // destroying it — only the mode is reset. - setInputMode('default') - setInputFocused(true) - inputRef.current?.focus() - setMessages((prev) => [ - ...prev, - getSystemMessage(`Skill not found: ${skillName ?? '(unknown)'}`), - ]) - return - } - - setInputValue({ text: '', cursorPosition: 0, lastEditDueToNav: false }) - setInputMode('default') - setInputFocused(true) - inputRef.current?.focus() - - if (trimmed) { - saveToHistory(trimmed) - } - dispatchSkillPrompt(params, skill, trimmed) - return - } - // Handle review mode input if (inputMode === 'review') { if (!trimmed) return @@ -467,35 +430,6 @@ export async function routeUserPrompt( streamMessageIdRef.current || isChainInProgressRef.current ) { - // Steer the running turn when possible: plain text is handed to the - // active run and injected at its next step boundary, so the user can - // redirect the agent without waiting the turn out. Falls back to the - // queue for anything the steering hook can't carry faithfully: - // attachments (strings only), a slash command (queued today so it can - // error/execute after the turn), pending `!` bash output (only - // prepareUserMessage folds it into the message that referenced it), a - // non-empty queue (steering would deliver this text ahead of earlier - // submissions), and the window where no run is accepting steering. - const canSteer = - !hasAttachments && - !isSlashCommand(trimmed) && - useChatStore.getState().pendingBashMessages.length === 0 && - !hasQueuedMessages?.() - if (canSteer) { - // Echo the bubble now, so the submit is visible immediately, and hand - // its id to the buffer: if the run ends before draining this entry, - // use-send-message retracts the bubble and requeues the text (which - // mints its own bubble at dequeue) — no invisible message, no dupe. - const steeredMessage = getUserMessage(trimmed) - if ( - pushSteeringMessage({ messageId: steeredMessage.id, text: trimmed }) - ) { - setMessages((prev) => [...prev, steeredMessage]) - setInputFocused(true) - inputRef.current?.focus() - return - } - } const pendingAttachmentsForQueue = capturePendingAttachments() // Pass a copy of pending attachments to the queue addToQueue(trimmed, pendingAttachmentsForQueue) diff --git a/cli/src/components/__tests__/status-bar.test.tsx b/cli/src/components/__tests__/status-bar.test.tsx index 2db6c9afd5..b0af467ef3 100644 --- a/cli/src/components/__tests__/status-bar.test.tsx +++ b/cli/src/components/__tests__/status-bar.test.tsx @@ -1,18 +1,12 @@ import { beforeAll, describe, expect, test } from 'bun:test' -import { FREEBUFF_DEEPSEEK_V4_FLASH_MODEL_ID } from '@codebuff/common/constants/freebuff-model-ids' import { createTestRenderer } from '@opentui/core/testing' import { createRoot, flushSync } from '@opentui/react' import React from 'react' import { StatusBar } from '../status-bar' import { initializeThemeStore } from '../../hooks/use-theme' -import { useChatStore } from '../../state/chat-store' -import { IS_FREEBUFF } from '../../utils/constants' import { getStatusIndicatorState } from '../../utils/status-indicator-state' -import type { FreebuffSessionResponse } from '../../types/freebuff-session' -import type { RunState } from '@codebuff/sdk' - beforeAll(() => { initializeThemeStore() }) @@ -47,60 +41,4 @@ describe('StatusBar', () => { setup.renderer.destroy() } }) - - // The idle session line (and therefore the context readout) only renders in - // freebuff builds — useFreebuffSessionProgress returns null otherwise. - test.skipIf(!IS_FREEBUFF)( - 'renders context usage next to the unlimited label', - async () => { - const now = Date.now() - const session = { - status: 'active', - accessTier: 'full', - instanceId: 'test-instance', - model: FREEBUFF_DEEPSEEK_V4_FLASH_MODEL_ID, - admittedAt: new Date(now - 60_000).toISOString(), - expiresAt: new Date(now + 3_600_000).toISOString(), - remainingMs: 3_600_000, - } as FreebuffSessionResponse - useChatStore.getState().setRunState({ - sessionState: { - mainAgentState: { contextTokenCount: 142_310 }, - }, - } as RunState) - - const statusIndicatorState = getStatusIndicatorState({ - statusMessage: null, - streamStatus: 'idle', - nextCtrlCWillExit: false, - isConnected: true, - }) - // Wide frame: the right-hand flex column takes half the row, and the - // left label truncates rather than wraps. - const setup = await createTestRenderer({ width: 140, height: 3 }) - const root = createRoot(setup.renderer) - flushSync(() => { - root.render( - {}} - statusIndicatorState={statusIndicatorState} - freebuffSession={session} - />, - ) - }) - - try { - await setup.renderOnce() - const frame = setup.captureCharFrame() - // 142,310 of DeepSeek V4 Flash's 1,048,576-token window → 14%. - expect(frame).toContain('unlimited · 142.3K (14%)') - } finally { - flushSync(() => root.unmount()) - setup.renderer.destroy() - useChatStore.getState().setRunState(null) - } - }, - ) }) diff --git a/cli/src/components/chat-input-bar.tsx b/cli/src/components/chat-input-bar.tsx index bd69227021..8f2aadf578 100644 --- a/cli/src/components/chat-input-bar.tsx +++ b/cli/src/components/chat-input-bar.tsx @@ -123,25 +123,8 @@ export const ChatInputBar = ({ }: ChatInputBarProps) => { const inputMode = useChatStore((state) => state.inputMode) const setInputMode = useChatStore((state) => state.setInputMode) - const pendingSkillName = useChatStore((state) => state.pendingSkillName) - - const baseModeConfig = getInputModeConfig(inputMode) - // Skill mode names the pending skill in the banner so the user can see - // what their text will be attached to. Skill names run up to 64 chars; - // keep the banner narrow enough to leave room for typing. - const skillLabel = - inputMode === 'skill' && pendingSkillName - ? pendingSkillName.length > 24 - ? `${pendingSkillName.slice(0, 23)}…` - : pendingSkillName - : null - const modeConfig = skillLabel - ? { - ...baseModeConfig, - label: skillLabel, - widthAdjustment: skillLabel.length + 3, - } - : baseModeConfig + + const modeConfig = getInputModeConfig(inputMode) const askUserState = useChatStore((state) => state.askUserState) const hasAnyPreview = hasSuggestionMenu diff --git a/cli/src/components/status-bar.tsx b/cli/src/components/status-bar.tsx index 3cfeafb1d4..4ad6213455 100644 --- a/cli/src/components/status-bar.tsx +++ b/cli/src/components/status-bar.tsx @@ -1,8 +1,4 @@ -import { - FREEBUFF_DEFAULT_CONTEXT_WINDOW, - FREEBUFF_MODEL_CONTEXT_WINDOWS, - getFreebuffModel, -} from '@codebuff/common/constants/freebuff-models' +import { getFreebuffModel } from '@codebuff/common/constants/freebuff-models' import { TextAttributes } from '@opentui/core' import React, { useEffect, useState } from 'react' @@ -12,9 +8,7 @@ import { ShimmerText } from './shimmer-text' import { useFreebuffSessionProgress } from '../hooks/use-freebuff-session-progress' import { useTheme } from '../hooks/use-theme' -import { useChatStore } from '../state/chat-store' import { formatElapsedTime } from '../utils/format-elapsed-time' -import { formatContextUsage } from '../utils/format-token-count' import { FREEBUFF_COUNTDOWN_VISIBLE_MS, formatFreebuffSessionCountdown, @@ -116,27 +110,6 @@ export const StatusBar = ({ const isUnlimited = freebuffSession?.status === 'active' && !freebuffSession.rateLimit - // Context occupancy of the main agent only: subagent states never land in - // mainAgentState, so their tokens are excluded by construction. The store's - // runState is written at end of turn, which is exactly when the idle branch - // below renders — no mid-turn staleness is visible. - const contextTokenCount = useChatStore( - // Fully optional-chained: runState can be restored from a JSON.parse of - // run-state.json with no shape validation, and a throwing selector would - // crash the whole TUI. - (state) => - state.runState?.sessionState?.mainAgentState?.contextTokenCount, - ) - const contextWindow = - freebuffSession?.status === 'active' - ? (FREEBUFF_MODEL_CONTEXT_WINDOWS[freebuffSession.model] ?? - FREEBUFF_DEFAULT_CONTEXT_WINDOW) - : FREEBUFF_DEFAULT_CONTEXT_WINDOW - const contextUsage = - contextTokenCount !== undefined - ? formatContextUsage(contextTokenCount, contextWindow) - : null - const renderStatusIndicator = () => { switch (statusIndicatorState.kind) { case 'ctrlC': @@ -198,13 +171,6 @@ export const StatusBar = ({ freebuffSession?.status === 'active' ? getFreebuffModel(freebuffSession.model).displayName : null - // One template string on purpose: conditional text-node children - // inside a trip OpenTUI's reconciler (see knowledge.md). - const idleLabel = `${modelName ? `${modelName} · ` : ''}${ - isUnlimited - ? 'unlimited' - : formatFreebuffSessionRemaining(sessionProgress.remainingMs) - }${contextUsage ? ` · ${contextUsage}` : ''}` return ( - {idleLabel} + {modelName ? `${modelName} · ` : ''} + {isUnlimited + ? 'unlimited' + : formatFreebuffSessionRemaining(sessionProgress.remainingMs)} ) } diff --git a/cli/src/contexts/chat-runtime-context.tsx b/cli/src/contexts/chat-runtime-context.tsx index ebe0d61829..978596f215 100644 --- a/cli/src/contexts/chat-runtime-context.tsx +++ b/cli/src/contexts/chat-runtime-context.tsx @@ -188,7 +188,6 @@ export const ChatRuntimeProvider = ({ isProcessingQueueRef: queue.isProcessingQueueRef, resumeQueue: queue.resumeQueue, requeueMessageAtFront: queue.addToQueueFront, - pauseQueue: queue.pauseQueue, continueChat, continueChatId, subscriptionData, diff --git a/cli/src/hooks/use-send-message.ts b/cli/src/hooks/use-send-message.ts index 698859a44e..ea2526a867 100644 --- a/cli/src/hooks/use-send-message.ts +++ b/cli/src/hooks/use-send-message.ts @@ -35,11 +35,6 @@ import { sanitizeRestoredMessages, } from '../utils/send-message-helpers' import { createSendMessageTimerController } from '../utils/send-message-timer' -import { - activateSteering, - deactivateSteering, - drainSteeringMessages as drainSteeringBuffer, -} from '../utils/steering-buffer' import { handleRunCompletion, handleRunError, @@ -88,10 +83,6 @@ interface UseSendMessageOptions { content: string attachments: PendingAttachment[] }) => void - /** Pause the queue. Used when requeueing an undelivered steering message - * after a user interrupt, so the held text doesn't auto-start a new turn - * the user just stopped. */ - pauseQueue?: () => void continueChat: boolean continueChatId?: string subscriptionData?: SubscriptionResponse | null @@ -145,7 +136,6 @@ export const useSendMessage = ({ isProcessingQueueRef, resumeQueue, requeueMessageAtFront, - pauseQueue, continueChat, continueChatId, subscriptionData, @@ -640,16 +630,6 @@ export const useSendMessage = ({ runChatDir, ) }, - // Mid-turn steering: the agent loop calls this at each step - // boundary; texts pushed by the router since the last boundary are - // injected into the running turn as user prompts. The transcript - // bubble was already echoed at push time (router), so this only - // hands over the texts. Returning [] on abort leaves the entries - // in the buffer for the leftover handling below. - drainSteeringMessages: () => { - if (abortController.signal.aborted) return [] - return drainSteeringBuffer(runOwnerId).map((entry) => entry.text) - }, }) // Log a summary only: the full run config contains the entire @@ -673,9 +653,6 @@ export const useSendMessage = ({ }, '[send-message] Sending message with sdk run config', ) - // Open the steering mailbox for this run only once we're committed to - // calling run(); the router falls back to the queue before this point. - activateSteering(runOwnerId) const runState = await client.run(runConfig) // Only adopt and persist the result while this run's chat is still @@ -751,38 +728,6 @@ export const useSendMessage = ({ logger.debug({ error }, '[send-message] Ignoring error after abort') } } finally { - // Close the steering mailbox. Anything the run never drained was - // submitted after its last step boundary; retract its push-time - // bubble (the requeued send mints its own at dequeue) and requeue it - // at the front so it isn't lost. On Esc the queue is paused first, - // matching the 'pause-if-pending' interrupt policy that ran while - // this text was still in the buffer — without the pause, the - // unblocked queue would immediately auto-start a new turn the user - // just tried to stop. Skipped after a mid-run chat switch (the - // message belongs to the old chat, whose queue was already cleared - // by the stop policy) and after non-user aborts like logout, where - // resurrecting input is wrong. - const steeringLeftovers = deactivateSteering(runOwnerId) - if ( - steeringLeftovers.length > 0 && - runChatIsCurrent() && - (!abortController.signal.aborted || - abortController.signal.reason === 'user-interrupt') - ) { - const leftoverIds = new Set( - steeringLeftovers.map((entry) => entry.messageId), - ) - setMessages((prev) => prev.filter((msg) => !leftoverIds.has(msg.id))) - if (abortController.signal.aborted) { - pauseQueue?.() - } - for (const entry of steeringLeftovers.reverse()) { - requeueMessageAtFront?.({ - content: entry.text, - attachments: [], - }) - } - } // Stop exit-flushing this run's checkpoint; the final state (or last // checkpoint, on error) has been saved above. Owner-guarded so an // aborted run resolving late can't clear a newer run's provider. @@ -826,7 +771,6 @@ export const useSendMessage = ({ removeActiveSubagent, requeueMessageAtFront, resumeQueue, - pauseQueue, scrollToLatest, setCanProcessQueue, setFocusedAgentId, diff --git a/cli/src/state/chat-store.ts b/cli/src/state/chat-store.ts index 37b7da8871..84fc0dc051 100644 --- a/cli/src/state/chat-store.ts +++ b/cli/src/state/chat-store.ts @@ -76,10 +76,6 @@ export type ChatStoreState = { /** The currently active top banner, or null if none */ activeTopBanner: TopBannerType inputMode: InputMode - /** Skill awaiting user text while inputMode === 'skill'. Cleared by - * setInputMode whenever the mode moves off 'skill', so Escape and every - * other mode exit reset it without extra bookkeeping. */ - pendingSkillName: string | null isRetrying: boolean /** True while the current retry wait is a server capacity deferral (free * mode shed under high demand) rather than a stream recovery — the status @@ -153,10 +149,6 @@ type ChatStoreActions = { setActiveTopBanner: (banner: TopBannerType) => void closeTopBanner: () => void setInputMode: (mode: InputMode) => void - setPendingSkillName: (name: string | null) => void - /** Atomic skill-mode entry: mode and pending skill set together, so - * 'skill' mode with a null skill is never representable via this path. */ - enterSkillMode: (skillName: string) => void setIsRetrying: (retrying: boolean) => void /** Mark the current wait as a free-mode capacity deferral (implies * isRetrying). Cleared by setIsRetrying(false). */ @@ -212,7 +204,6 @@ const initialState: ChatStoreState = { runState: null, activeTopBanner: null, inputMode: 'default' as InputMode, - pendingSkillName: null as string | null, isRetrying: false, isCapacityWait: false, askUserState: null, @@ -341,20 +332,6 @@ export const useChatStore = create()( setInputMode: (mode) => set((state) => { state.inputMode = mode - if (mode !== 'skill') { - state.pendingSkillName = null - } - }), - - setPendingSkillName: (name) => - set((state) => { - state.pendingSkillName = name - }), - - enterSkillMode: (skillName) => - set((state) => { - state.inputMode = 'skill' - state.pendingSkillName = skillName }), setIsRetrying: (retrying) => @@ -555,7 +532,6 @@ export const useChatStore = create()( : null state.activeTopBanner = initialState.activeTopBanner state.inputMode = initialState.inputMode - state.pendingSkillName = initialState.pendingSkillName state.isRetrying = initialState.isRetrying state.isCapacityWait = initialState.isCapacityWait state.askUserState = initialState.askUserState diff --git a/cli/src/utils/__tests__/format-token-count.test.ts b/cli/src/utils/__tests__/format-token-count.test.ts deleted file mode 100644 index cdbba18bef..0000000000 --- a/cli/src/utils/__tests__/format-token-count.test.ts +++ /dev/null @@ -1,58 +0,0 @@ -import { describe, expect, test } from 'bun:test' - -import { - formatContextUsage, - formatTokenCount, -} from '../format-token-count' - -describe('formatTokenCount', () => { - test('small counts render as-is', () => { - expect(formatTokenCount(0)).toBe('0') - expect(formatTokenCount(982)).toBe('982') - }) - - test('thousands get a K suffix with one decimal', () => { - expect(formatTokenCount(1000)).toBe('1K') - expect(formatTokenCount(14_231)).toBe('14.2K') - expect(formatTokenCount(132_500)).toBe('132.5K') - expect(formatTokenCount(999_949)).toBe('999.9K') - }) - - test('millions get an M suffix', () => { - expect(formatTokenCount(1_000_000)).toBe('1M') - expect(formatTokenCount(1_250_000)).toBe('1.3M') - }) - - test('values that round to 1000K promote to 1M', () => { - expect(formatTokenCount(999_960)).toBe('1M') - }) - - test('garbage is rendered as zero rather than NaN', () => { - expect(formatTokenCount(Number.NaN)).toBe('0') - expect(formatTokenCount(-5)).toBe('0') - }) -}) - -describe('formatContextUsage', () => { - test('formats tokens with a rounded window percentage', () => { - expect(formatContextUsage(14_231, 203_300)).toBe('14.2K (7%)') - expect(formatContextUsage(131_072, 1_048_576)).toBe('131.1K (13%)') - }) - - test('never shows 0% for a non-empty context', () => { - expect(formatContextUsage(1200, 1_048_576)).toBe('1.2K (1%)') - }) - - test('clamps at 100% when the estimate overshoots the window', () => { - expect(formatContextUsage(150_000, 131_072)).toBe('150K (100%)') - }) - - test('returns null when there is nothing to show', () => { - expect(formatContextUsage(0, 131_072)).toBeNull() - expect(formatContextUsage(Number.NaN, 131_072)).toBeNull() - }) - - test('omits the percentage when the window is unknown', () => { - expect(formatContextUsage(14_231, 0)).toBe('14.2K') - }) -}) diff --git a/cli/src/utils/__tests__/steering-buffer.test.ts b/cli/src/utils/__tests__/steering-buffer.test.ts deleted file mode 100644 index e400607198..0000000000 --- a/cli/src/utils/__tests__/steering-buffer.test.ts +++ /dev/null @@ -1,69 +0,0 @@ -import { afterEach, describe, expect, test } from 'bun:test' - -import { - __resetSteeringForTests, - activateSteering, - deactivateSteering, - drainSteeringMessages, - isSteeringActive, - pushSteeringMessage, -} from '../steering-buffer' - -const entry = (text: string, messageId = `msg-${text}`) => ({ - messageId, - text, -}) - -afterEach(() => { - __resetSteeringForTests() -}) - -describe('steering buffer', () => { - test('push fails while no run is active', () => { - expect(isSteeringActive()).toBe(false) - expect(pushSteeringMessage(entry('hello'))).toBe(false) - }) - - test('push/drain round-trips in order while a run is active', () => { - activateSteering('run-1') - expect(isSteeringActive()).toBe(true) - expect(pushSteeringMessage(entry('first'))).toBe(true) - expect(pushSteeringMessage(entry('second'))).toBe(true) - expect(drainSteeringMessages('run-1')).toEqual([ - entry('first'), - entry('second'), - ]) - // Drained means gone. - expect(drainSteeringMessages('run-1')).toEqual([]) - }) - - test('drain is owner-guarded', () => { - activateSteering('run-1') - pushSteeringMessage(entry('for run 1')) - expect(drainSteeringMessages('run-2')).toEqual([]) - expect(drainSteeringMessages('run-1')).toEqual([entry('for run 1')]) - }) - - test('deactivate returns undelivered leftovers exactly once', () => { - activateSteering('run-1') - pushSteeringMessage(entry('too late')) - expect(deactivateSteering('run-1')).toEqual([entry('too late')]) - expect(deactivateSteering('run-1')).toEqual([]) - expect(pushSteeringMessage(entry('after end'))).toBe(false) - }) - - test('a stale run cannot deactivate a newer run', () => { - activateSteering('run-1') - activateSteering('run-2') - pushSteeringMessage(entry('for run 2')) - expect(deactivateSteering('run-1')).toEqual([]) - expect(drainSteeringMessages('run-2')).toEqual([entry('for run 2')]) - }) - - test('activation clears residue from a run that never deactivated', () => { - activateSteering('run-1') - pushSteeringMessage(entry('stale')) - activateSteering('run-2') - expect(drainSteeringMessages('run-2')).toEqual([]) - }) -}) diff --git a/cli/src/utils/create-run-config.ts b/cli/src/utils/create-run-config.ts index b55408cbb8..29a3d97e6e 100644 --- a/cli/src/utils/create-run-config.ts +++ b/cli/src/utils/create-run-config.ts @@ -30,10 +30,6 @@ export type CreateRunConfigParams = { extraCodebuffMetadata?: Record /** Periodic in-flight RunState checkpoints (see RunOptions.onStateSnapshot). */ onStateSnapshot?: (runState: RunState) => void - /** Mid-turn steering: drained by the agent loop at each step boundary; - * returned texts are appended as user prompts and keep the turn going - * (see RunOptions.drainSteeringMessages). */ - drainSteeringMessages?: () => string[] } const SENSITIVE_EXTENSIONS = new Set([ @@ -101,7 +97,6 @@ export const createRunConfig = (params: CreateRunConfigParams) => { costMode, extraCodebuffMetadata, onStateSnapshot, - drainSteeringMessages, } = params return { @@ -118,7 +113,6 @@ export const createRunConfig = (params: CreateRunConfigParams) => { costMode, extraCodebuffMetadata, onStateSnapshot, - drainSteeringMessages, fileFilter: ((filePath: string) => { if (isSensitiveFile(filePath)) return { status: 'blocked' } return { status: 'allow' } diff --git a/cli/src/utils/format-token-count.ts b/cli/src/utils/format-token-count.ts deleted file mode 100644 index b4f3949602..0000000000 --- a/cli/src/utils/format-token-count.ts +++ /dev/null @@ -1,48 +0,0 @@ -import { clamp } from './math' - -/** - * Compact token count for the status bar: 982 → "982", 14_231 → "14.2K", - * 1_250_000 → "1.3M". One decimal, trailing ".0" dropped, so the readout - * stays narrow in an 80-column terminal. - */ -export function formatTokenCount(tokens: number): string { - if (!Number.isFinite(tokens) || tokens < 0) { - return '0' - } - if (tokens < 1000) { - return String(Math.round(tokens)) - } - const format = (value: number, suffix: string): string => { - const rounded = Math.round(value * 10) / 10 - const text = Number.isInteger(rounded) - ? String(rounded) - : rounded.toFixed(1) - return `${text}${suffix}` - } - // Branch on the ROUNDED value: 999,960 rounds to 1000.0K and must render - // as 1M, not "1000K". - if (Math.round(tokens / 100) / 10 < 1000) { - return format(tokens / 1000, 'K') - } - return format(tokens / 1_000_000, 'M') -} - -/** - * "14.2K (7%)" — context occupancy against the model's context window. The - * percentage is rounded but never shown as 0% while tokens are non-zero, so a - * fresh session reads "1%" rather than implying an empty context is tracked - * at all. Returns null when there is nothing meaningful to show. - */ -export function formatContextUsage( - tokens: number, - contextWindow: number, -): string | null { - if (!Number.isFinite(tokens) || tokens <= 0) { - return null - } - if (!Number.isFinite(contextWindow) || contextWindow <= 0) { - return formatTokenCount(tokens) - } - const percent = clamp(Math.round((tokens / contextWindow) * 100), 1, 100) - return `${formatTokenCount(tokens)} (${percent}%)` -} diff --git a/cli/src/utils/input-modes.ts b/cli/src/utils/input-modes.ts index 68563e3de0..e3e5b8fae3 100644 --- a/cli/src/utils/input-modes.ts +++ b/cli/src/utils/input-modes.ts @@ -12,7 +12,6 @@ export type InputMode = | 'plan' | 'review' | 'interview' - | 'skill' | 'usage' | 'image' | 'help' @@ -91,18 +90,6 @@ export const INPUT_MODE_CONFIGS: Record = { disableSlashSuggestions: true, blockKeyboardExit: false, }, - skill: { - icon: null, - // Label is replaced with the pending skill's name at render time - // (chat-input-bar), so the mode banner names what is about to run. - label: 'Skill', - color: 'info', - placeholder: 'add instructions for this skill, or press Enter to run it as-is...', - widthAdjustment: 8, - showAgentModeToggle: false, - disableSlashSuggestions: true, - blockKeyboardExit: false, - }, plan: { icon: null, label: 'Plan', diff --git a/cli/src/utils/skill-registry.ts b/cli/src/utils/skill-registry.ts index 79942f4e99..4668404569 100644 --- a/cli/src/utils/skill-registry.ts +++ b/cli/src/utils/skill-registry.ts @@ -98,10 +98,3 @@ export function getLoadedSkillsMessage(): string | null { export function __resetSkillRegistryForTests(): void { skillsCache = {} } - -/** - * Seed the cache without touching the filesystem. Intended for test scenarios. - */ -export function __setSkillsForTests(skills: SkillsMap): void { - skillsCache = skills -} diff --git a/cli/src/utils/steering-buffer.ts b/cli/src/utils/steering-buffer.ts deleted file mode 100644 index 8a8be34032..0000000000 --- a/cli/src/utils/steering-buffer.ts +++ /dev/null @@ -1,74 +0,0 @@ -/** - * Mid-turn steering buffer. - * - * The agent loop runs in-process (SDK → agent-runtime), and the runtime - * drains `RunOptions.drainSteeringMessages` at every step boundary: any - * returned texts are appended to the conversation as user prompts and keep - * the turn going. This module is the CLI-side mailbox between the composer - * (router) and the active run (use-send-message), mirroring the claim/accept - * shape freebuff-desktop uses for the same hook. - * - * The router echoes the transcript bubble at push time and records its id - * here, so an entry the run never drains can have its bubble retracted when - * the text is requeued as a fresh turn (which mints its own bubble). - * - * Owner-guarded like active-run.ts: an aborted run resolving late must not - * drain or clear a newer run's buffer. - */ - -export type SteeringEntry = { - /** Transcript id of the bubble echoed when the entry was pushed. */ - messageId: string - text: string -} - -let activeOwnerId: string | null = null -let buffer: SteeringEntry[] = [] - -/** Called by use-send-message right before client.run(). */ -export function activateSteering(ownerId: string): void { - activeOwnerId = ownerId - buffer = [] -} - -/** - * Called by use-send-message when the run settles. Returns any entries the - * run never drained (submitted after its last step boundary) so the caller - * can retract their bubbles and requeue the texts instead of dropping them. - */ -export function deactivateSteering(ownerId: string): SteeringEntry[] { - if (activeOwnerId !== ownerId) return [] - activeOwnerId = null - const leftovers = buffer - buffer = [] - return leftovers -} - -/** - * Called by the router on a mid-turn submit. Returns false when no run is - * accepting steering (caller falls back to the queue). - */ -export function pushSteeringMessage(entry: SteeringEntry): boolean { - if (activeOwnerId === null) return false - buffer.push(entry) - return true -} - -/** True while a run is accepting steering pushes. */ -export function isSteeringActive(): boolean { - return activeOwnerId !== null -} - -/** Called by the run's drainSteeringMessages hook at each step boundary. */ -export function drainSteeringMessages(ownerId: string): SteeringEntry[] { - if (activeOwnerId !== ownerId || buffer.length === 0) return [] - const drained = buffer - buffer = [] - return drained -} - -/** Test seam. */ -export function __resetSteeringForTests(): void { - activeOwnerId = null - buffer = [] -} diff --git a/cli/src/utils/time-format.test.ts b/cli/src/utils/time-format.test.ts deleted file mode 100644 index e7351bc0e9..0000000000 --- a/cli/src/utils/time-format.test.ts +++ /dev/null @@ -1,61 +0,0 @@ -import { afterEach, describe, expect, test } from 'bun:test' -import { setSystemTime } from 'bun:test' - -import { formatResetTimeLong } from './time-format' - -describe('formatResetTimeLong', () => { - afterEach(() => { - setSystemTime() - }) - - test('returns empty string for null', () => { - expect(formatResetTimeLong(null)).toBe('') - }) - - test('formats a multi-day reset with remaining hours', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - const resetDate = new Date('2026-01-05T07:00:00.000Z') - - expect(formatResetTimeLong(resetDate)).toBe('4d 7h') - }) - - test('formats a whole number of days with no remaining hours', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - const resetDate = new Date('2026-01-03T00:00:00.000Z') - - expect(formatResetTimeLong(resetDate)).toBe('2d') - }) - - test('formats hours and minutes under a day away', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - const resetDate = new Date('2026-01-01T02:30:00.000Z') - - expect(formatResetTimeLong(resetDate)).toBe('2h 30m') - }) - - test('formats minutes only under an hour away', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - const resetDate = new Date('2026-01-01T00:15:00.000Z') - - expect(formatResetTimeLong(resetDate)).toBe('15m') - }) - - test('falls back to "now" for a date already in the past', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - const resetDate = new Date('2025-12-31T00:00:00.000Z') - - expect(formatResetTimeLong(resetDate)).toBe('now') - }) - - test('accepts an ISO string in addition to a Date', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - - expect(formatResetTimeLong('2026-01-01T01:00:00.000Z')).toBe('1h') - }) - - test('falls back to "now" for an unparseable date string', () => { - setSystemTime(new Date('2026-01-01T00:00:00.000Z')) - - expect(formatResetTimeLong('not-a-date')).toBe('now') - }) -}) diff --git a/common/src/__tests__/freebuff-models.test.ts b/common/src/__tests__/freebuff-models.test.ts index ed03b46c60..f6ce0581c4 100644 --- a/common/src/__tests__/freebuff-models.test.ts +++ b/common/src/__tests__/freebuff-models.test.ts @@ -437,9 +437,9 @@ describe('freebuff model availability', () => { test('GLM 5.3 Flash is UNMETERED, and the two flags that say so agree', () => { // Unmetered on 2026-08-28, matching DeepSeek V4 Flash and MiMo. It was // premium-pooled while its cost was unknown; measured prod spend settled - // that as the cheapest row we serve, 8.9x under the already-unmetered - // V4 Flash. Capping the cheapest model while the dearer ones run uncapped - // inverts the reason caps exist. + // that at $0.000249/msg — the cheapest row we serve, 8.9x under the + // already-unmetered V4 Flash. Capping the cheapest model while the dearer + // ones run uncapped inverts the reason caps exist. expect( getFreebuffPerModelSessionCap(FREEBUFF_GLM_V53_FLASH_MODEL_ID), ).toBeUndefined() @@ -791,7 +791,7 @@ describe('freebuff model availability', () => { test('Kimi K2.7 Code is fully removed from Freebuff', () => { // Removed 2026-07-31 (client pickers went first, on 2026-07-30). The server // half is gone too, so a stale client selection is no longer admitted — - // that tail was still a material daily spend. Paid/BYOK Kimi is unaffected; + // that tail was still spending ~$2.3k/day. Paid/BYOK Kimi is unaffected; // it never resolves through these helpers. expect(SUPPORTED_FREEBUFF_MODELS.map((model) => model.id)).not.toContain( FREEBUFF_KIMI_MODEL_ID, @@ -1235,8 +1235,8 @@ describe('freebuff model availability', () => { }) test('MiniMax M3 is withdrawn: recognised, refused, served to nobody', () => { - // Withdrawn from free mode entirely on 2026-08-20 after its hourly burn - // became the largest single line on the bill. Out of every picker and pool... + // Withdrawn from free mode entirely on 2026-08-20 after reaching $213/hr. + // Out of every picker and pool... expect(FREEBUFF_MODELS.map((model) => model.id)).not.toContain( MINIMAX_M3_MODEL_ID, ) diff --git a/common/src/__tests__/freebuff-trust.test.ts b/common/src/__tests__/freebuff-trust.test.ts new file mode 100644 index 0000000000..e93d58875d --- /dev/null +++ b/common/src/__tests__/freebuff-trust.test.ts @@ -0,0 +1,534 @@ +import { describe, expect, it } from 'bun:test' + +import { + assessFreebuffTrust, + FREEBUFF_TRUST_FALLBACK_LEVEL, + FREEBUFF_TRUST_LEVELS, + FREEBUFF_TRUST_EARNED, + FREEBUFF_TRUST_LIMITS, + FREEBUFF_TRUST_THRESHOLDS, + freebuffTrustLimits, + isAtLeastTrustLevel, + toFreebuffStandingInfo, + type FreebuffTrustSignals, +} from '../constants/freebuff-trust' +import { FREEBUFF_PREMIUM_SESSION_LIMIT } from '../constants/freebuff-models' + +const NOW = new Date('2026-08-11T00:00:00Z') +const DAY_MS = 24 * 60 * 60 * 1000 + +function daysAgo(days: number): Date { + return new Date(NOW.getTime() - days * DAY_MS) +} + +/** An account we know nothing about: every optional signal unknown. This is + * what most pre-provenance accounts actually look like. */ +const UNKNOWN: FreebuffTrustSignals = { + accountCreatedAt: null, + githubAccountCreatedAt: null, + githubOldestRepoCreatedAt: null, + githubPublicRepos: null, + githubFollowers: null, + githubTwoFactorEnabled: null, + activeDays: 0, + approvedBounties: 0, + qualifiedReferrals: 0, + hasPaid: false, + signupPrivacySignals: null, + signupIpSource: null, + signupPrefixAccountCount: null, + mailboxAccountCount: null, + hasUnreversedBanEvent: false, + privacyFlaggedAt: null, + privacyCorroboratedAt: null, + thirdPartyClientAt: null, + currentRiskScore: null, +} + +function signals(overrides: Partial) { + return { ...UNKNOWN, ...overrides } +} + +describe('level ordering', () => { + it('orders least- to most-established', () => { + expect(FREEBUFF_TRUST_LEVELS).toEqual([ + 'new', + 'verified', + 'established', + 'core', + ]) + }) + + it('compares by position, not alphabetically', () => { + expect(isAtLeastTrustLevel('core', 'new')).toBe(true) + expect(isAtLeastTrustLevel('new', 'verified')).toBe(false) + expect(isAtLeastTrustLevel('established', 'established')).toBe(true) + }) +}) + +describe('limit matrix', () => { + it('matches the flat fallback limits at established/full', () => { + // The flat pool and the enforced fallback must not silently meter the same + // established account differently. + const full = freebuffTrustLimits('full', 'established') + expect(full.messagesPerDay).toBe(5_000) + expect(full.messagesPer5Hours).toBe(3_000) + expect(full.dailySpendUsd).toBe(50) + // Premium is deliberately NOT part of that equality any more. It left this + // matrix when Levels shipped (common/src/constants/freebuff-levels.ts): + // the floor here is one session, and everything above it is earned and + // added on top. Pinning it to the old flat 5 would re-assert exactly the + // thing that change undid. + expect(full.premiumSessionsPerDay).toBe( + FREEBUFF_PREMIUM_SESSION_LIMIT, + ) + + const limited = freebuffTrustLimits('limited', 'established') + expect(limited.messagesPerDay).toBe(3_000) + expect(limited.messagesPer5Hours).toBe(2_000) + }) + + it('is monotonic in level on every axis, in both regions', () => { + for (const tier of ['full', 'limited'] as const) { + for (let i = 1; i < FREEBUFF_TRUST_LEVELS.length; i++) { + const lower = FREEBUFF_TRUST_LIMITS[tier][FREEBUFF_TRUST_LEVELS[i - 1]] + const higher = FREEBUFF_TRUST_LIMITS[tier][FREEBUFF_TRUST_LEVELS[i]] + for (const key of Object.keys(lower) as (keyof typeof lower)[]) { + expect(higher[key]).toBeGreaterThanOrEqual(lower[key]) + } + } + } + }) + + it('does not scale session-shape controls, only cost controls', () => { + // Browser sessions and Desktop tabs were deliberately removed: an open + // session costs nothing until it generates, and the generating is already + // bounded. If either reappears here, something re-added a limit that takes + // visible capability from new users and saves nothing. + expect(Object.keys(FREEBUFF_TRUST_LIMITS.full.new).sort()).toEqual([ + 'dailySpendUsd', + 'messagesPer5Hours', + 'messagesPerDay', + 'premiumSessionsPerDay', + 'userMessagesPerDay', + ]) + }) + + it('never lets the limited tier reach a premium session', () => { + // The model gate already refuses it; a non-zero number here would be a + // promise the rest of the system cannot keep. + for (const level of FREEBUFF_TRUST_LEVELS) { + expect(FREEBUFF_TRUST_LIMITS.limited[level].premiumSessionsPerDay).toBe(0) + } + }) + + it('lets a limited-region core member beat a full-region verified user', () => { + // The promise the region split has to make to a real developer abroad. + const limitedCore = freebuffTrustLimits('limited', 'core') + const fullVerified = freebuffTrustLimits('full', 'verified') + expect(limitedCore.messagesPerDay).toBeGreaterThan( + fullVerified.messagesPerDay, + ) + expect(limitedCore.userMessagesPerDay).toBeGreaterThan( + fullVerified.userMessagesPerDay, + ) + expect(limitedCore.dailySpendUsd).toBeGreaterThan( + fullVerified.dailySpendUsd, + ) + }) + + it('keeps every new-account daily budget above the measured p90', () => { + // Sizing anchor from free-mode-rate-limiter.ts: full-tier per-user-per-day + // p90 is 837. A brand-new account doing genuinely heavy work must fit. + expect( + freebuffTrustLimits('full', 'new').messagesPerDay, + ).toBeGreaterThanOrEqual(837) + }) +}) + +describe('scoring', () => { + it('leaves an unknown account at the floor', () => { + const result = assessFreebuffTrust(UNKNOWN, NOW) + expect(result.level).toBe('new') + expect(result.score).toBe(0) + expect(result.cappedBy).toBeNull() + }) + + it('treats null signals as unknown, never as suspicious', () => { + // A pre-provenance account (null everything) must score the same as one + // explicitly checked and found to share nothing. + const unknownProvenance = assessFreebuffTrust( + signals({ githubAccountCreatedAt: daysAgo(400) }), + NOW, + ) + const cleanProvenance = assessFreebuffTrust( + signals({ + githubAccountCreatedAt: daysAgo(400), + signupPrefixAccountCount: 1, + mailboxAccountCount: 1, + }), + NOW, + ) + expect(unknownProvenance.level).toBe(cleanProvenance.level) + expect(unknownProvenance.cappedBy).toBeNull() + }) + + it('reaches established on GitHub age plus ordinary account history', () => { + const result = assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(120), + githubAccountCreatedAt: daysAgo(3 * 365 + 10), + githubOldestRepoCreatedAt: daysAgo(400), + githubPublicRepos: 12, + activeDays: 40, + }), + NOW, + ) + // 10 linked + 20 age + 10 repo + 5 repos + 15 acct age + 10 active + expect(result.score).toBe(70) + expect(result.level).toBe('established') + }) + + it('lets a brand-new limited-region account climb with earned signals alone', () => { + // The route that does not require owning an aged GitHub account: this is + // what the Earn page has to be able to promise. + const result = assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(10), + approvedBounties: 4, + qualifiedReferrals: 5, + }), + NOW, + ) + // 5 acct age + 20 bounties + 15 referrals + expect(result.score).toBe(40) + expect(result.level).toBe('verified') + }) + + it('caps bounty and referral contributions', () => { + const capped = assessFreebuffTrust( + signals({ approvedBounties: 50, qualifiedReferrals: 50 }), + NOW, + ) + expect(capped.score).toBe( + FREEBUFF_TRUST_EARNED.BOUNTY_POINTS * FREEBUFF_TRUST_EARNED.BOUNTY_CAP + + FREEBUFF_TRUST_EARNED.REFERRAL_POINTS * + FREEBUFF_TRUST_EARNED.REFERRAL_CAP, + ) + }) + + it('lets contribution alone reach core, with no GitHub and no payment', () => { + // THE property the earned caps exist for. Before they were raised this + // route peaked at 70 against a threshold of 75, so `core` was reachable + // only by owning an aged GitHub account or by paying — which is backwards + // for a program meant to give developers in unsupported regions a way to + // raise their own limits. + const earned = assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(120), + activeDays: 40, + approvedBounties: FREEBUFF_TRUST_EARNED.BOUNTY_CAP, + qualifiedReferrals: FREEBUFF_TRUST_EARNED.REFERRAL_CAP, + signupPrivacySignals: [], + signupIpSource: 'cloudflare', + }), + NOW, + ) + expect(earned.factors.map((f) => f.id)).not.toContain('github_linked') + expect(earned.score).toBeGreaterThanOrEqual(FREEBUFF_TRUST_THRESHOLDS.core) + expect(earned.level).toBe('core') + }) + + it('keeps paying past the point someone has proved they are real', () => { + // A flat incentive is not an incentive. The tenth referral and the sixth + // bounty must still be worth something, or the program stops pulling + // exactly where it should pull hardest. + const few = assessFreebuffTrust( + signals({ approvedBounties: 4, qualifiedReferrals: 5 }), + NOW, + ) + const many = assessFreebuffTrust( + signals({ approvedBounties: 6, qualifiedReferrals: 10 }), + NOW, + ) + expect(many.score).toBeGreaterThan(few.score) + }) + + it('never returns a score outside 0..100', () => { + const maxed = assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(1000), + githubAccountCreatedAt: daysAgo(4000), + githubOldestRepoCreatedAt: daysAgo(3000), + githubPublicRepos: 100, + githubFollowers: 500, + githubTwoFactorEnabled: true, + activeDays: 300, + approvedBounties: 20, + qualifiedReferrals: 20, + hasPaid: true, + signupPrivacySignals: [], + signupIpSource: 'cloudflare', + }), + NOW, + ) + expect(maxed.score).toBe(100) + expect(maxed.level).toBe('core') + }) + + it('ignores a future-dated timestamp rather than crediting it', () => { + const skewed = assessFreebuffTrust( + signals({ + githubAccountCreatedAt: new Date(NOW.getTime() + 10 * DAY_MS), + }), + NOW, + ) + // Linked (10) but no age credit. + expect(skewed.score).toBe(10) + }) +}) + +describe('caps', () => { + const HIGH_SCORE: Partial = { + accountCreatedAt: daysAgo(400), + githubAccountCreatedAt: daysAgo(2000), + githubOldestRepoCreatedAt: daysAgo(1000), + githubPublicRepos: 20, + githubFollowers: 50, + githubTwoFactorEnabled: true, + activeDays: 100, + approvedBounties: 4, + } + + it('caps a live anonymous network at verified however high the score', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, currentRiskScore: 90 }), + NOW, + ) + expect(result.uncappedLevel).toBe('core') + expect(result.level).toBe('verified') + expect(result.cappedBy).toBe('anonymous_network') + }) + + it('does not cap on a low current risk score', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, currentRiskScore: 10 }), + NOW, + ) + expect(result.level).toBe('core') + expect(result.cappedBy).toBeNull() + }) + + it('caps a VPN signup at established, not lower', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, signupPrivacySignals: ['vpn'] }), + NOW, + ) + expect(result.level).toBe('established') + expect(result.cappedBy).toBe('signup_privacy_egress') + }) + + it('credits a clean signup rather than capping it', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, signupPrivacySignals: [] }), + NOW, + ) + expect(result.cappedBy).toBeNull() + expect(result.factors.some((f) => f.id === 'clean_signup')).toBe(true) + }) + + it('applies the lowest cap when several bind', () => { + const result = assessFreebuffTrust( + signals({ + ...HIGH_SCORE, + signupPrivacySignals: ['vpn'], + mailboxAccountCount: 5, + }), + NOW, + ) + expect(result.level).toBe('verified') + expect(result.cappedBy).toBe('shared_mailbox') + }) + + it('leads the next steps with the cap, since points cannot clear it', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, currentRiskScore: 90 }), + NOW, + ) + expect(result.nextSteps[0]?.id).toBe('cap_anonymous_network') + expect(result.nextSteps[0]?.label).toMatch(/VPN/i) + }) + + it('caps an account with unreversed enforcement history', () => { + const result = assessFreebuffTrust( + signals({ ...HIGH_SCORE, hasUnreversedBanEvent: true }), + NOW, + ) + expect(result.level).toBe('verified') + expect(result.cappedBy).toBe('past_enforcement') + }) +}) + +describe('next steps', () => { + it('leads with connecting GitHub for an account that has none', () => { + const result = assessFreebuffTrust(UNKNOWN, NOW) + expect(result.nextSteps[0]?.id).toBe('connect_github') + expect(result.nextSteps[0]?.points).toBe(30) + }) + + it('stops offering steps the user has already exhausted', () => { + const result = assessFreebuffTrust( + signals({ + approvedBounties: FREEBUFF_TRUST_EARNED.BOUNTY_CAP, + qualifiedReferrals: FREEBUFF_TRUST_EARNED.REFERRAL_CAP, + }), + NOW, + ) + expect(result.nextSteps.map((s) => s.id)).not.toContain('bounties') + expect(result.nextSteps.map((s) => s.id)).not.toContain('referrals') + }) + + it('offers the remaining value, not the full value, of a partial step', () => { + const result = assessFreebuffTrust(signals({ approvedBounties: 2 }), NOW) + const remaining = + (FREEBUFF_TRUST_EARNED.BOUNTY_CAP - 2) * + FREEBUFF_TRUST_EARNED.BOUNTY_POINTS + expect(result.nextSteps.find((s) => s.id === 'bounties')?.points).toBe( + remaining, + ) + }) +}) + +describe('wire shape', () => { + it('never puts a raw limit on the wire', () => { + // A published limit is a published target: the abuse pattern here is + // sustained pacing just under the caps, so the caps stay server-side. This + // asserts the payload rather than the component, because the payload is + // what a future surface would reach for. + const info = toFreebuffStandingInfo( + assessFreebuffTrust(signals({ approvedBounties: 4 }), NOW), + 'limited', + ) + expect(info).not.toHaveProperty('limits') + + // Scoped to the distinctive values. Small ones (a $3 spend cap, 5 premium + // sessions) collide with legitimate point values in the copy — "worth 3 + // points" is not a leaked limit, and asserting on them would fail for the + // wrong reason. + const serialized = JSON.stringify(info) + const distinctive = Object.values( + freebuffTrustLimits('limited', info.level), + ).filter((value) => value >= 100) + expect(distinctive.length).toBeGreaterThan(0) + for (const value of distinctive) { + expect(serialized).not.toContain(String(value)) + } + expect(info.accessTier).toBe('limited') + }) + + it('describes each axis in words', () => { + const info = toFreebuffStandingInfo( + assessFreebuffTrust(signals({ approvedBounties: 4 }), NOW), + 'limited', + ) + expect(info.highlights.map((h) => h.label)).toEqual([ + 'Prompts a day', + 'Work per prompt', + 'Premium models', + ]) + for (const highlight of info.highlights) { + expect(highlight.value).not.toMatch(/\d/) + } + }) + + it('frames premium as a region fact where no level can unlock it', () => { + // Every limited-row level has 0 premium sessions, so calling it a + // shortfall would send the user chasing points that cannot buy it. + const limited = toFreebuffStandingInfo( + assessFreebuffTrust(signals({ approvedBounties: 4 }), NOW), + 'limited', + ) + expect( + limited.highlights.find((h) => h.label === 'Premium models')?.value, + ).toMatch(/region/i) + + const full = toFreebuffStandingInfo( + assessFreebuffTrust(signals({ approvedBounties: 4 }), NOW), + 'full', + ) + expect( + full.highlights.find((h) => h.label === 'Premium models')?.value, + ).not.toMatch(/region/i) + }) + + it('reports the next threshold, and nothing beyond core', () => { + const verified = toFreebuffStandingInfo( + assessFreebuffTrust( + signals({ approvedBounties: 4, activeDays: 10 }), + NOW, + ), + 'full', + ) + expect(verified.level).toBe('verified') + expect(verified.nextLevel).toBe('established') + expect(verified.nextLevelAt).toBe(FREEBUFF_TRUST_THRESHOLDS.established) + + const core = toFreebuffStandingInfo( + assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(400), + githubAccountCreatedAt: daysAgo(2000), + githubOldestRepoCreatedAt: daysAgo(1000), + githubPublicRepos: 20, + githubFollowers: 50, + githubTwoFactorEnabled: true, + activeDays: 100, + approvedBounties: 4, + }), + NOW, + ), + 'full', + ) + expect(core.level).toBe('core') + expect(core.nextLevel).toBeNull() + expect(core.nextLevelAt).toBeNull() + }) + + it('explains a cap in the copy the client renders', () => { + const info = toFreebuffStandingInfo( + assessFreebuffTrust( + signals({ + accountCreatedAt: daysAgo(400), + githubAccountCreatedAt: daysAgo(2000), + activeDays: 100, + currentRiskScore: 99, + }), + NOW, + ), + 'full', + ) + expect(info.cappedBy).toBe('anonymous_network') + expect(info.cappedReason).toMatch(/VPN/i) + }) + + it('reports no cap when the cap sits at or above the earned level', () => { + // An account scoring 'new' is not "capped at verified" — nothing bound. + const info = toFreebuffStandingInfo( + assessFreebuffTrust(signals({ currentRiskScore: 99 }), NOW), + 'full', + ) + expect(info.level).toBe('new') + expect(info.cappedBy).toBeNull() + }) +}) + +describe('failure behaviour', () => { + it('falls back to the level that matches the flat limits', () => { + // A broken resolver must cost us the enforcement, never the users: if this + // were 'new', one degraded query would throttle the whole product. + expect(FREEBUFF_TRUST_FALLBACK_LEVEL).toBe('established') + expect(freebuffTrustLimits('full', FREEBUFF_TRUST_FALLBACK_LEVEL)).toEqual( + freebuffTrustLimits('full', 'established'), + ) + }) +}) diff --git a/common/src/constants/__tests__/freebuff-limited-subscriber.test.ts b/common/src/constants/__tests__/freebuff-limited-subscriber.test.ts index 3e3329e66b..afaa46f52f 100644 --- a/common/src/constants/__tests__/freebuff-limited-subscriber.test.ts +++ b/common/src/constants/__tests__/freebuff-limited-subscriber.test.ts @@ -6,7 +6,6 @@ import { getFreebuffModelsForAccessTier, isFreebuffSessionModelAllowedForAccessTier, isFreebuffWebModelAllowedForLimitedTier, - isFreebuffWebModelId, resolveFreebuffSessionModelForAccessTier, resolveFreebuffWebModelForLimitedTier, } from '../freebuff-models' @@ -70,17 +69,6 @@ describe('paid plans at limited access', () => { } expect(FREEBUFF_SUBSCRIPTION_MODEL_IDS).toHaveLength(4) }) - - test('every plan model resolves in the Web catalog', () => { - // The plans page renders the plan lineup via getFreebuffWebModel, which - // FALLS BACK to MiMo 2.5 for an id the Web catalog lacks — it would - // advertise the one model its own copy says a plan escapes, and nothing - // would error. The page filters such ids out; this is what makes the - // drift loud instead of silently shrinking that panel. - for (const model of FREEBUFF_SUBSCRIPTION_MODEL_IDS) { - expect(isFreebuffWebModelId(model, { includeGodOnly: true })).toBe(true) - } - }) }) /** @@ -97,9 +85,9 @@ describe('paid plans at limited access', () => { describe('a plan model survives resolution, not just the allowlist', () => { test('unpaid limited access still coerces every plan model to MiMo', () => { for (const model of FREEBUFF_SUBSCRIPTION_MODEL_IDS) { - expect(resolveFreebuffSessionModelForAccessTier(model, 'limited')).toBe( - LIMITED_FREEBUFF_MODEL_ID, - ) + expect( + resolveFreebuffSessionModelForAccessTier(model, 'limited'), + ).toBe(LIMITED_FREEBUFF_MODEL_ID) } }) @@ -138,15 +126,15 @@ describe('a plan model survives resolution, not just the allowlist', () => { // And it gained at least one row it could not pick before. expect(paid.length).toBeGreaterThan(free.length) for (const id of paid) { - expect( - isFreebuffSessionModelAllowedForAccessTier(id, 'limited', true), - ).toBe(true) + expect(isFreebuffSessionModelAllowedForAccessTier(id, 'limited', true)).toBe( + true, + ) } }) test('full access is untouched by the widened catalog', () => { - expect( - getFreebuffModelsForAccessTier('full', true).map((m) => m.id), - ).toEqual(getFreebuffModelsForAccessTier('full').map((m) => m.id)) + expect(getFreebuffModelsForAccessTier('full', true).map((m) => m.id)).toEqual( + getFreebuffModelsForAccessTier('full').map((m) => m.id), + ) }) }) diff --git a/common/src/constants/free-agents.ts b/common/src/constants/free-agents.ts index 651adba81d..5933acd8b0 100644 --- a/common/src/constants/free-agents.ts +++ b/common/src/constants/free-agents.ts @@ -500,7 +500,7 @@ export const FREE_MODE_AGENT_MODELS: Record> = { // been hidden from every client picker in 75fb0ade6 (2026-07-30) while // deliberately staying valid here and in session admission, so released // clients weren't broken mid-session. That tail kept costing real spend - // (a double-digit share of free-mode cost) because CLI builds older than 75fb0ade6 + // (~$2.3k/day, 19% of free-mode cost) because CLI builds older than 75fb0ade6 // never drop a saved Kimi preference, and nothing forces those users to // upgrade. Every remaining free-mode Kimi request now 403s with // 'free_mode_invalid_agent_model'. Paid/BYOK Kimi is unaffected: the diff --git a/common/src/constants/freebuff-models.ts b/common/src/constants/freebuff-models.ts index bee3bc77a2..111609f5eb 100644 --- a/common/src/constants/freebuff-models.ts +++ b/common/src/constants/freebuff-models.ts @@ -388,10 +388,10 @@ export const FREEBUFF_FABLE_5_MODEL_ID = 'anthropic/claude-fable-5' * (see docs/freebuff-muse-spark.md) that the browser can render a wait for. * The CLI has no such queue and would just surface 429s. * - * Contributor pricing (a small fraction of Standard's published per-M rates; - * the negotiated numbers stay out of this exported file) is bought with - * training rights over prompts and completions, which is why this is - * `dataUse: 'training'` and carries the AI-training warning. + * Contributor pricing ($0.10/$0.002/$0.20 per M against Standard's + * $1.25/$0.15/$4.25) is bought with training rights over prompts and + * completions, which is why this is `dataUse: 'training'` and carries the + * AI-training warning. */ export const FREEBUFF_MUSE_SPARK_12_CONTRIBUTOR_MODEL_ID = 'meta/muse-spark-1.2-contributor' @@ -1066,12 +1066,12 @@ const DEEPSEEK_V4_FLASH_MODEL = { // can hold the peak window exists again, and Flash is once more the row whose // whole cost doubles inside it. // - // Flash is a large share of fleet spend and DeepSeek doubles its price for - // ten hours a day. Measured 2026-08-24 09:00Z, inside the window (per-message - // figures in the internal cost notes — measured $ numbers do not belong in - // this file, which is exported to the public repo): Pro at Cheaper Inference - // cost within 2% of peak Flash, so redirecting saved nothing, while Luna ran - // at roughly half. + // Flash is ~46% of fleet spend and DeepSeek doubles it for ten hours a day. + // Measured 2026-08-24 09:00Z, inside the window: + // + // Flash @ DeepSeek peak $0.005621/msg + // Pro @ Cheaper Inf. $0.005731/msg (1.02x — saves nothing) + // Luna @ Cheaper Inf. $0.002659/msg (2.11x CHEAPER) // // Hence the fallback points at LUNA, not Pro. The old pointer named Pro from // when Pro was the flat-priced row; it is now merely the same price as the @@ -1080,18 +1080,24 @@ const DEEPSEEK_V4_FLASH_MODEL = { // REOPENED 2026-08-28. The closure above was correct on its own measurement // and was invalidated by its own effect. // - // That 08-24 reading caught Luna at its WARM price, taken before Flash's - // traffic was displaced onto it. Closing Flash is what moved a flood of - // unfamiliar prefixes onto Luna's lane, and a prefix cache is the whole cost - // of these rows: Luna's cache rate collapsed inside the window and its price - // went with it. Re-measured 2026-08-28, hourly: absorbing Luna became the - // DEAREST of the three per message; peak Flash about half of that; and Flash - // on Luminal — which is not DeepSeek and so has no peak surcharge at all — - // cheaper than both by ~4x (~8x at the hour peak pricing begins, same model, - // same minute). + // That 08-24 reading priced Luna at $0.002659/msg -- its WARM price, taken + // before Flash's traffic was displaced onto it. Closing Flash is what moved + // ~30k msg/hr of unfamiliar prefixes onto Luna's lane, and a prefix cache is + // the whole cost of these rows: Luna fell from ~95% cache to 59-68% inside + // the window and its price went with it. Re-measured 2026-08-28, hourly: + // + // Luna @ Cheaper Inf. (absorbing) $0.00925/msg 59-68% cache + // Flash @ DeepSeek peak $0.0042-0.0057/msg + // Flash @ LUMINAL $0.00103/msg 96% cache, NO peak card + // + // The ordering inverted: the row we closed to save money is now half the + // price of the row we sent its traffic to, and Luminal -- which is not + // DeepSeek and so has no peak surcharge at all -- is cheaper than both by 4x. + // Measured at 01:00Z, the hour peak pricing begins: DeepSeek $0.00416, + // Luminal $0.00053, same model, same minute. // - // The closure therefore cost a meaningful daily sum of excess Luna spend, - // against a saving premised on a price that no longer existed. + // Cost of the closure, measured over 04:00-12:00Z: ~$1,180/day of excess Luna + // spend, against a saving premised on a price that no longer exists. // // A closure justified by a measurement must be rechecked when the thing it // measured is downstream of the closure itself. This one was not, for four @@ -1338,10 +1344,13 @@ const GLM_V53_FLASH_MODEL = { // UNMETERED, like DeepSeek V4 Flash and MiMo — the two other rows in // FREEBUFF_STANDARD_MODEL_IDS. It was premium-pooled while its true cost was // unknown; measured production spend has now settled that, and it is the - // CHEAPEST row we serve (per-message and per-session figures live in the - // internal cost notes, not in this exported file). + // CHEAPEST row we serve: // - // This row is 4.6x cheaper per session than MiMo and 8.9x cheaper than V4 + // glm-5.3-flash (Merge, 91.7% cache) $0.000249/msg $0.0196/session + // deepseek-v4-flash (already unmetered) $0.002223/msg $0.1752/session + // mimo-v2.5 (already unmetered) $0.001151/msg $0.0907/session + // + // So this row is 4.6x cheaper per session than MiMo and 8.9x cheaper than V4 // Flash, both of which already run with no ceiling at all. Keeping a session // cap on the cheapest model while the dearer ones are uncapped inverts the // reason caps exist. @@ -1642,8 +1651,10 @@ export const FREEBUFF_MODELS = [ // a new user's first send cannot fail because a pool ran dry. // // And it is the cheapest row we serve, by a wide margin — measured production - // spend per message puts MiMo at 4.6x this row and V4 Flash at 8.9x (exact - // figures in the internal cost notes, not in this exported file). + // spend, per message: + // glm-5.3-flash $0.000249 (this row) + // mimo-v2.5 $0.001151 4.6x + // deepseek-v4-flash $0.002223 8.9x // // WHAT THIS GIVES UP, stated plainly because the previous ordering note was // written to prevent exactly this move: this row is the DEEP one, and depth @@ -1697,9 +1708,9 @@ export const FREEBUFF_MODELS = [ // nothing else and sit outside every number the picker shows. export const FREEBUFF_PREMIUM_MODEL_IDS = [ FREEBUFF_GPT_5_6_LUNA_MODEL_ID, - // GLM 5.3 Flash left on 2026-08-28: measured production spend made it the - // cheapest row we serve, 8.9x under the already-unmetered V4 Flash. - // See GLM_V53_FLASH_MODEL for what leaving here also + // GLM 5.3 Flash left on 2026-08-28: measured production spend put it at + // $0.000249/msg, the cheapest row we serve and 8.9x under the already- + // unmetered V4 Flash. See GLM_V53_FLASH_MODEL for what leaving here also // drops. This list and that entry's `premium` flag must always agree — // isFreebuffPremiumModelId reads this one while FREEBUFF_STANDARD_MODEL_IDS // is derived from the flag, so a disagreement makes a row premium for the @@ -1758,10 +1769,10 @@ export const FREEBUFF_PER_MODEL_SESSION_CAPS: Readonly< // EMPTY SINCE 2026-08-27, and deliberately kept as a table rather than // deleted. GLM 5.3 Flash was the only entry — capped at 2/day as a // MEASUREMENT WINDOW while its true cost was unknown, exactly as the comment - // above describes. That window has now closed: the lane held a high cache - // rate on its pinned vendor and came in ~6x under the OpenRouter route it - // replaced (measured figures in docs/freebuff-merge-gateway.md, which is - // not exported). The cap has therefore come off, and the + // above describes. That window has now closed: the lane was measured at + // 93.6% cache on its pinned vendor and ~$0.00059 per model call, which is + // 6.2x under the OpenRouter route it replaced (see + // docs/freebuff-merge-gateway.md). The cap has therefore come off, and the // model is metered by the SHARED premium pool alone — it is in // FREEBUFF_WEB_PREMIUM_MODEL_IDS via FREEBUFF_PREMIUM_MODEL_IDS, so a // full-access account may spend any of its FREEBUFF_PREMIUM_SESSION_LIMIT @@ -1814,8 +1825,8 @@ export const FREEBUFF_DEEPSEEK_SESSION_WINDOW_HOURS = * clients that need it are the ones already installed. */ export const FREEBUFF_PAUSED_FREE_MODEL_IDS: readonly string[] = [ - // Withdrawn from free mode entirely on 2026-08-20. Its hourly burn became - // the largest single line on the bill — and is not worth that at any tier. + // Withdrawn from free mode entirely on 2026-08-20. It reached $213/hr — the + // largest single line on the bill — and is not worth that at any tier. // // PAUSED rather than deleted, which is the difference between withdrawing a // model and breaking the clients that still ask for it. Every released CLI and @@ -2108,9 +2119,11 @@ export const FREEBUFF_DESKTOP_PREMIUM_BUCKET_MODEL_IDS = [ // GLM 5.3 Flash LEFT on 2026-08-29, and on this list's own criterion rather // than as a side effect of unmetering it the day before. Membership is "a // bill we would not want to underwrite at three at once", and measured - // production spend puts it at the CHEAPEST row we serve — well under both - // MiMo and V4 Flash per session, and both of those already run 3 tabs - // (figures in the internal cost notes, not in this exported file). + // production spend puts it at $0.000249/msg — the CHEAPEST row we serve: + // + // glm-5.3-flash $0.000249/msg $0.0196/session <- 3 tabs, now + // mimo-v2.5 $0.001151/msg $0.0907/session <- 3 tabs already + // deepseek-v4-flash $0.002223/msg $0.1752/session <- 3 tabs already // // Three concurrent tabs of it is a smaller bill than three of either row this // list has always allowed, so keeping it here failed the test on its face. @@ -2320,7 +2333,7 @@ export type FreebuffWebPremiumModelId = * want of quota. * * It is also `availability: 'always'` and the cheapest row we serve - * (measured per-message, 4.6x under MiMo and 8.9x under V4 Flash). The cost + * ($0.000249/msg measured, 4.6x under MiMo and 8.9x under V4 Flash). The cost * and availability arguments are therefore both strictly better than the Luna * it replaces; the argument it LOSES is latency, since this is the deep row * running `defaultEffort: 'max'`. See FREEBUFF_MODELS for that trade in full. @@ -2358,9 +2371,9 @@ export const DEFAULT_FREEBUFF_MODEL_ID: FreebuffModelId = * availability wins and is the change this comment expects to be made. * * The cost half is not close. Cache reads are ~98% of browser tokens, and this - * row's list cache-read rate is nearly double Luna's — but measured per - * message on the traffic that actually runs it bills an order of magnitude - * LESS than Luna (figures in the internal cost notes, not here). + * row reads cache at $0.015/M against Luna's $0.008/M list — but it bills + * $0.000249/msg measured against Luna's $0.002659, an order of magnitude + * apart on the traffic that actually runs. * * Kept as its own constant from DEFAULT_FREEBUFF_MODEL_ID (CLI/Desktop) so the * browser surfaces can steer independently. They name the same model today and diff --git a/common/src/constants/freebuff-standing.ts b/common/src/constants/freebuff-standing.ts deleted file mode 100644 index 2a3515982e..0000000000 --- a/common/src/constants/freebuff-standing.ts +++ /dev/null @@ -1,123 +0,0 @@ -/** - * Freebuff account standing ("Access Level") — the PRESENTATIONAL half. - * - * This file carries everything a client may see: the level names, their - * user-facing labels and blurbs, and the wire shapes for standing info. The - * actual limit matrix, thresholds and scorer live in - * `./freebuff-trust.ts`, which is deliberately EXCLUDED from the public-repo - * export (see scripts/public-export-manifest.txt) — a published limit is a - * published target, so the numbers must never ship in a public file. Keep - * that split when adding here: names, labels and shapes only, never numbers. - */ - -import type { FreebuffAccessTier } from './freebuff-models' - -/** - * Ordered least- to most-established. The order is load-bearing: - * `FREEBUFF_TRUST_LEVELS.indexOf` is how "at least X" comparisons are done, so - * inserting a level in the middle re-ranks every comparison in one edit rather - * than requiring each call site to be found. - */ -export const FREEBUFF_TRUST_LEVELS = [ - 'new', - 'verified', - 'established', - 'core', -] as const - -export type FreebuffTrustLevel = (typeof FREEBUFF_TRUST_LEVELS)[number] - -/** The level an account holds before anything is known about it. Every failure - * path in the resolver must land somewhere DEFINITE, and this is not it — see - * `FREEBUFF_TRUST_FALLBACK_LEVEL`. */ -export const FREEBUFF_TRUST_MIN_LEVEL: FreebuffTrustLevel = 'new' - -/** - * The level used when signals cannot be loaded (database error, timeout). - * - * `established` and NOT `new`, and this is the single most consequential - * constant in the file. This resolver runs on the free-mode hot path; if a - * Postgres hiccup dropped every caller to `new`, one degraded dependency would - * throttle the entire product to a fifth of its capacity, and it would look - * exactly like an outage nobody could attribute. Failing to the level that - * reproduces roughly today's flat limits means a broken resolver costs us the - * enforcement, never the users. Same reasoning as the signup gate's fail-open. - */ -export const FREEBUFF_TRUST_FALLBACK_LEVEL: FreebuffTrustLevel = 'established' - -export function isAtLeastTrustLevel( - level: FreebuffTrustLevel, - minimum: FreebuffTrustLevel, -): boolean { - return ( - FREEBUFF_TRUST_LEVELS.indexOf(level) >= - FREEBUFF_TRUST_LEVELS.indexOf(minimum) - ) -} - -/** User-facing name. Never says "trust", "risk" or "score" — a user reading - * their own level is reading an explanation of their limits, not a verdict on - * their character. */ -export const FREEBUFF_TRUST_LEVEL_LABELS: Record = { - new: 'Getting started', - verified: 'Verified', - established: 'Established', - core: 'Core member', -} - -/** One line of user-facing copy per level, shown under the label. */ -export const FREEBUFF_TRUST_LEVEL_BLURBS: Record = { - new: 'Welcome! Your account is brand new, so limits start small. They open up quickly — the steps below take a few minutes.', - verified: - 'Your account is verified. You have solid daily limits, and a bit of history unlocks the next level.', - established: - 'You are an established Freebuff user with generous limits on messages, spend and premium sessions.', - core: 'You are a core member. You get the highest free limits we offer, in every region.', -} - -/** A signal that moved the score, in user-facing language. */ -export interface FreebuffTrustFactor { - id: string - label: string - points: number -} - -/** Something the user can do to move up, with what it is worth. */ -export interface FreebuffTrustNextStep { - id: string - label: string - detail: string - points: number - /** Where the UI should send them. Relative to the freebuff web app. */ - href?: string -} - -export interface FreebuffStandingHighlight { - label: string - value: string -} - -/** - * NOTE FOR CALLERS: `highlights` is what the level WOULD grant, which is only - * what the account actually gets once `FREEBUFF_TRUST_LEVELS=enforce`. Both - * producers gate on that (the Earn route and the session `standing` field), so - * a client that receives this can render it as fact. A third producer must do - * the same — see the comment in freebuff/web/src/app/api/web/standing/route.ts - * for what happens otherwise. - */ -export interface FreebuffStandingInfo { - level: FreebuffTrustLevel - label: string - blurb: string - score: number - /** Score at which the next level starts, or null at `core`. */ - nextLevelAt: number | null - nextLevel: FreebuffTrustLevel | null - cappedBy: string | null - cappedReason: string | null - factors: FreebuffTrustFactor[] - nextSteps: FreebuffTrustNextStep[] - accessTier: FreebuffAccessTier - /** Semantic, never numeric — see FreebuffStandingHighlight. */ - highlights: FreebuffStandingHighlight[] -} diff --git a/common/src/constants/freebuff-subscriptions.ts b/common/src/constants/freebuff-subscriptions.ts index 24a53e7803..fe362ac4f6 100644 --- a/common/src/constants/freebuff-subscriptions.ts +++ b/common/src/constants/freebuff-subscriptions.ts @@ -44,9 +44,8 @@ export const FREEBUFF_SUBSCRIPTION_MODEL_IDS: readonly string[] = Object.freeze( /** * The expensive half of the pool, sub-capped within each day. * - * Measured 2026-08-21: Luna and DeepSeek V4 Pro each cost roughly 4-5x Flash - * per hour-session (dollar figures live in the internal cost notes, not in - * this exported file). Without a sub-cap a subscriber + * Measured 2026-08-21: Luna $0.758 and DeepSeek V4 Pro $0.605 per hour-session, + * against Flash at $0.156 — roughly 4-5x. Without a sub-cap a subscriber * spending every daily session on Luna costs 5x one spending them on Flash, at * the same price, so the daily allowance would have to be priced for the worst * case and would be small for everyone. @@ -184,9 +183,8 @@ export const FREEBUFF_SUBSCRIPTION_TIERS: readonly FreebuffSubscriptionTier[] = introPriceUsd: 5, // Resized 2026-08-27, before the public rollout. The pre-rollout // figures were sized against god-only testing and priced most of a - // month of premium use into $8 — at Luna's measured per-session cost, - // 4/day was an order of magnitude more compute than the price - // (figures in the internal cost notes). The 5-DAY window is what actually bounds a + // month of premium use into $8 — at Luna's measured $0.758/session, + // 4/day was ~$91 of compute. The 5-DAY window is what actually bounds a // heavy week (3/day would allow 15 in five days; 10 is the real cap), // which is why the two numbers are not simply proportional. dailySessions: 3, diff --git a/common/src/constants/freebuff-trust.ts b/common/src/constants/freebuff-trust.ts new file mode 100644 index 0000000000..674e4fb22c --- /dev/null +++ b/common/src/constants/freebuff-trust.ts @@ -0,0 +1,1011 @@ +/** + * Freebuff account standing ("Access Level") — the per-account policy layer + * that decides how much free capacity an account gets. + * + * ## Why this exists + * + * Every free-mode control before this was keyed on the ACCOUNT and applied the + * same number to every account: the same 5,000 requests/day, the same $50 + * spend budget, the same 6 premium sessions. That is only a bound if accounts + * are scarce, and `docs/freebuff-signup-gate.md` is the record of them not + * being. A relay pooling 100 minted accounts is entitled, entirely within the + * rules, to 100x every per-user limit. + * + * The signup gate raised the price of minting an account. This raises the + * price of a FRESH one being worth anything: a brand-new account from an + * unverifiable network gets a small fraction of the capacity, and an account + * that has demonstrably existed and done work for months gets considerably + * MORE than the flat limits ever gave it. Same fleet spend, redistributed + * toward the people we actually want to serve. + * + * ## Two axes, deliberately separate + * + * `FreebuffAccessTier` (full / limited) is a REGION property, resolved per + * request from the caller's IP country. `FreebuffTrustLevel` is an ACCOUNT + * property, resolved from durable facts about the account. They multiply: + * limits are a matrix, not a sum. A limited-region user can climb to `core` + * and get more than a full-region `new` account — which is the whole point of + * shipping this alongside the region split rather than instead of it. + * + * ## What the level must never be + * + * A ban input, or a reason to serve a degraded model. Everything here produces + * a NUMBER — a limit — and a limit that is reached produces a retryable 429 + * naming the remedy. `docs/freebuff-abuse-detection.md` records what it cost + * the last time a soft signal was allowed to convict (659 wrongly-banned + * accounts, 2026-08-03); none of the signals below is stronger than the ones + * that did it. + * + * ## Naming + * + * User-facing copy says "Access Level" and never "trust", because a user shown + * a low trust score reads an accusation. The code says `trustLevel` because + * that is what it is and a euphemism in an identifier costs a reader time. + * `FREEBUFF_TRUST_LEVEL_LABELS` is the one bridge between the two. + */ + +import type { FreebuffAccessTier } from './freebuff-models' + +// --------------------------------------------------------------------------- +// Levels +// --------------------------------------------------------------------------- + +/** + * Ordered least- to most-established. The order is load-bearing: + * `FREEBUFF_TRUST_LEVELS.indexOf` is how "at least X" comparisons are done, so + * inserting a level in the middle re-ranks every comparison in one edit rather + * than requiring each call site to be found. + */ +export const FREEBUFF_TRUST_LEVELS = [ + 'new', + 'verified', + 'established', + 'core', +] as const + +export type FreebuffTrustLevel = (typeof FREEBUFF_TRUST_LEVELS)[number] + +/** The level an account holds before anything is known about it. Every failure + * path in the resolver must land somewhere DEFINITE, and this is not it — see + * `FREEBUFF_TRUST_FALLBACK_LEVEL`. */ +export const FREEBUFF_TRUST_MIN_LEVEL: FreebuffTrustLevel = 'new' + +/** + * The level used when signals cannot be loaded (database error, timeout). + * + * `established` and NOT `new`, and this is the single most consequential + * constant in the file. This resolver runs on the free-mode hot path; if a + * Postgres hiccup dropped every caller to `new`, one degraded dependency would + * throttle the entire product to a fifth of its capacity, and it would look + * exactly like an outage nobody could attribute. Failing to the level that + * reproduces roughly today's flat limits means a broken resolver costs us the + * enforcement, never the users. Same reasoning as the signup gate's fail-open. + */ +export const FREEBUFF_TRUST_FALLBACK_LEVEL: FreebuffTrustLevel = 'established' + +export function isAtLeastTrustLevel( + level: FreebuffTrustLevel, + minimum: FreebuffTrustLevel, +): boolean { + return ( + FREEBUFF_TRUST_LEVELS.indexOf(level) >= + FREEBUFF_TRUST_LEVELS.indexOf(minimum) + ) +} + +function lowerOf( + a: FreebuffTrustLevel, + b: FreebuffTrustLevel, +): FreebuffTrustLevel { + return isAtLeastTrustLevel(a, b) ? b : a +} + +/** User-facing name. Never says "trust", "risk" or "score" — a user reading + * their own level is reading an explanation of their limits, not a verdict on + * their character. */ +export const FREEBUFF_TRUST_LEVEL_LABELS: Record = { + new: 'Getting started', + verified: 'Verified', + established: 'Established', + core: 'Core member', +} + +/** One line of user-facing copy per level, shown under the label. */ +export const FREEBUFF_TRUST_LEVEL_BLURBS: Record = { + new: 'Welcome! Your account is brand new, so limits start small. They open up quickly — the steps below take a few minutes.', + verified: + 'Your account is verified. You have solid daily limits, and a bit of history unlocks the next level.', + established: + 'You are an established Freebuff user with generous limits on messages, spend and premium sessions.', + core: 'You are a core member. You get the highest free limits we offer, in every region.', +} + +// --------------------------------------------------------------------------- +// Limits +// --------------------------------------------------------------------------- + +/** + * Everything a level controls, in one place. + * + * Every field is a per-account ceiling over a time window. None is a global + * budget and none is a concurrency cap — read each doc comment rather than + * inferring from the name. + */ +export interface FreebuffTrustLimits { + /** + * User prompts per Pacific day. NOT requests: one prompt is one root agent + * run, and an agentic turn behind it may make dozens of LLM calls. + * + * This is the limit that means what a person thinks "messages" means, and it + * is the one an honest heavy user should be able to feel without hitting. + * Counted only when a call OPENS a root run (`isNewPromptWindow`), so a + * caller that reuses one run id to hide many prompts pays the run-reuse + * guard's ceiling instead. + */ + userMessagesPerDay: number + /** Total free-mode LLM requests in any 5-hour window: prompts plus every + * subagent, tool loop and retry underneath them. */ + messagesPer5Hours: number + /** Total free-mode LLM requests per day. The overall ceiling — a premium + * request consumes this budget as well as its own. */ + messagesPerDay: number + /** + * Settled non-BYOK provider cost, in USD, since midnight Pacific, at or + * above which no FRESH session is admitted. Live sessions and reclaims are + * never interrupted, so the honest reading is "how much we will spend + * starting new work for this account today". + */ + dailySpendUsd: number + /** + * Premium-model sessions per Pacific day (the shared premium pool's base + * entitlement). Referral, streak and operator entitlement is ADDED on top of + * this, so a level never takes away something a user earned. + * + * Zero at every limited-region level: the limited tier cannot reach a + * premium model at all, and the number would be decoration. + */ + premiumSessionsPerDay: number +} + +/** + * Why the `premiumSessionsPerDay` column was compressed. + * + * It used to run 2 / 4 / 5 / 10. It now runs 2 / 3 / 4 / 5, and the change is + * at the TOP rather than the bottom: `established` tracks + * `FREEBUFF_PREMIUM_SESSION_LIMIT` as it always has, and `core` came down from + * 10 to 5. + * + * `core` was doing two jobs. It was the abuse control's verdict — this account + * is demonstrably real — and it was also the only reward in the product big + * enough to notice, reachable only through facts a user cannot act on today + * (an aged GitHub account, months of history). "Why do I only get four" had no + * answer anybody could act on this afternoon. + * + * The reward half moved to `freebuff-levels.ts`, which is denominated in + * something a user can go and do right now, and which tops out at + * `FREEBUFF_LEVEL_SESSION_CEILING` (7) — above every value in this column. So + * a core member is no worse off than before once they engage at all, and the + * route to the ceiling is open to a brand-new account in an unsupported + * region, which is exactly who it was closed to before. + * + * This column only selects anything under `FREEBUFF_TRUST_LEVELS=enforce`; + * under the default `observe` the flat base applies, so these numbers and the + * `FREEBUFF_LEVEL_SESSIONS` revert do not interact. + * + * The other four axes are unchanged and stay here, because they are cost + * controls rather than rewards. A Level must never be able to buy its way into + * a bigger daily SPEND budget, or the incentive and the abuse control end up + * pointing at the same dial. + *//** + * Two axes were deliberately REMOVED from this interface, and the reasoning is + * worth keeping so they are not quietly re-added. + * + * **Concurrent Desktop tabs** (`FREEBUFF_DESKTOP_SESSION_LIMITS.unlimited`) are + * still enforced but not level-scaled. A PAID PLAN does raise them + * (`freebuffDesktopSessionLimits`), which is not a contradiction: that is a + * purchase, not a reward for engagement, and it moves both ceilings together + * rather than metering one axis of a free account. That is a session-SHAPE control rather + * than a cost control: starting a session costs nothing, and a session that + * sits idle costs nothing either. What costs money is the traffic inside it, + * and that is already bounded four different ways by the fields above. + * + * Scaling it by level would therefore have taken visible, immediate capability + * away from exactly the users we most need to keep — a new user discovers "you + * may only open one tab" instantly, and it reads as a product that is broken + * rather than as a budget — in exchange for no measurable saving at all. The + * premium-session pool stays level-scaled because a premium session is the one + * whose mere existence commits us to expensive inference. + * + * **Browser sessions per day** used to be the other example here, capped at 6 + * on Web and Cloud and unlimited everywhere else. That pool was removed on + * 2026-08-18 by this same argument taken one step further: if session count is + * the wrong thing to meter, it is the wrong thing to meter on one surface + * too. + */ + +/** + * The matrix. Region tier picks the row, account level picks the column. + * + * ## How these numbers were chosen + * + * `established` × `full` is the current flat-limit fallback (5,000/day, + * 3,000/5h, $50, 5 premium), and `established` × `limited` reproduces the + * limited row (3,000/day, 2,000/5h). Keeping the matrix aligned with the flat + * fallback is deliberate: observe/off mode, resolver failures and enforced + * `established` accounts must all receive the same baseline. `core` remains + * the raise, while the levels below `established` tighten newer accounts. + * + * Sizing for the two new levels below `established` is anchored on the + * per-user-per-day distributions in `free-mode-rate-limiter.ts` (full tier p50 + * 131, p90 837, p99 2,351): `verified` at 3,000/day sits above the full tier's + * p99, and `new` at 1,200/day sits above its p90. So a genuinely new user + * doing genuinely heavy work still fits, and the accounts that do not fit are + * the ones doing several times what any measured human does on their first + * day. + * + * `core` is roughly 1.6x `established` rather than unbounded. It is a reward, + * not an exemption — an account that reaches `core` and is then compromised or + * sold should still cost a bounded amount, and the fleet-wide spend has to + * survive every core member using their allowance on the same day. + * + * ## The limited row is not merely the full row scaled down + * + * Its floor is deliberately harsher (`new` × limited is a third of `new` × + * full) because that intersection — brand-new account, unsupported region, + * often VPN — is the exact shape of the reselling farms. Its ceiling is + * deliberately generous (`core` × limited beats `verified` × full on every + * axis it can — premium is region-gated, not level-gated) because the entire + * promise this makes to a real developer in an unsupported country is that the + * region is a starting point and not a cage. + */ +export const FREEBUFF_TRUST_LIMITS: Record< + FreebuffAccessTier, + Record +> = { + full: { + new: { + userMessagesPerDay: 120, + messagesPer5Hours: 800, + messagesPerDay: 1_200, + dailySpendUsd: 8, + premiumSessionsPerDay: 2, + }, + verified: { + userMessagesPerDay: 300, + messagesPer5Hours: 1_800, + messagesPerDay: 3_000, + dailySpendUsd: 20, + premiumSessionsPerDay: 3, + }, + established: { + userMessagesPerDay: 600, + messagesPer5Hours: 3_000, + messagesPerDay: 5_000, + dailySpendUsd: 50, + premiumSessionsPerDay: 4, + }, + core: { + userMessagesPerDay: 1_000, + messagesPer5Hours: 5_000, + messagesPerDay: 8_000, + dailySpendUsd: 90, + premiumSessionsPerDay: 5, + }, + }, + limited: { + new: { + userMessagesPerDay: 40, + messagesPer5Hours: 400, + messagesPerDay: 500, + dailySpendUsd: 3, + premiumSessionsPerDay: 0, + }, + verified: { + userMessagesPerDay: 120, + messagesPer5Hours: 1_000, + messagesPerDay: 1_500, + dailySpendUsd: 10, + premiumSessionsPerDay: 0, + }, + established: { + userMessagesPerDay: 350, + messagesPer5Hours: 2_000, + messagesPerDay: 3_000, + dailySpendUsd: 25, + premiumSessionsPerDay: 0, + }, + core: { + userMessagesPerDay: 700, + messagesPer5Hours: 3_500, + messagesPerDay: 5_500, + dailySpendUsd: 55, + premiumSessionsPerDay: 0, + }, + }, +} + +export function freebuffTrustLimits( + accessTier: FreebuffAccessTier, + level: FreebuffTrustLevel, +): FreebuffTrustLimits { + return FREEBUFF_TRUST_LIMITS[accessTier][level] +} + +// --------------------------------------------------------------------------- +// Signals +// --------------------------------------------------------------------------- + +/** + * Everything the score is computed from. + * + * **`null` means "unknown", never "clean" and never "suspicious".** Every + * account created before `docs/freebuff-signup-gate.md` shipped has null + * provenance, and every account that never linked GitHub has null GitHub + * facts. A scorer that read null as bad would demote most of the existing user + * base overnight; one that read it as good would hand every fresh signup a + * clean slate. Unknown earns nothing and costs nothing — which lands those + * accounts at whatever their other, positive signals justify. + */ +export interface FreebuffTrustSignals { + /** `user.created_at`. */ + accountCreatedAt: Date | null + /** `referral_qualification.github_account_created_at` — set by GitHub + * server-side and not backdatable, which is what makes it worth points at + * all. A commit author date, by contrast, forges in one command. */ + githubAccountCreatedAt: Date | null + /** Oldest public repo's creation date. Same non-forgeability. */ + githubOldestRepoCreatedAt: Date | null + githubPublicRepos: number | null + githubFollowers: number | null + githubTwoFactorEnabled: boolean | null + /** Distinct Pacific days the account has used free mode + * (`freebuff_daily_usage`). Cheap history that cannot be bought. */ + activeDays: number + /** Approved bounty submissions. Reviewed proof-of-work — the single + * strongest earned signal here, and the one a limited-region user can act + * on today without owning an aged GitHub account. */ + approvedBounties: number + /** Qualified referrals GIVEN (`referral_v2`, active + qualified). */ + qualifiedReferrals: number + /** Has ever paid us anything. Wired now and worth points now; payments are + * planned, and the day they land this needs no scorer change. */ + hasPaid: boolean + /** Privacy signals recorded at SIGNUP (`user.signup_privacy_signals`), split + * on comma. Empty array = checked and clean; null = never checked. */ + signupPrivacySignals: readonly string[] | null + /** `user.signup_ip_source`. Anything other than `edge_secret`/`cloudflare` + * means the caller had some influence over the address. */ + signupIpSource: string | null + /** Accounts sharing this account's signup /24 or /48. Includes this account, + * so 1 is the clean value. */ + signupPrefixAccountCount: number | null + /** Accounts sharing this account's normalized mailbox. Includes this + * account. */ + mailboxAccountCount: number | null + /** A ban event that was NOT reversed. Live bans never reach here (banned + * accounts are refused before any of this runs), so this is history: an + * account that was actioned and then unbanned on appeal. */ + hasUnreversedBanEvent: boolean + /** `user.privacy_flagged_at` — first request ever seen on an ipinfo-flagged + * anonymizing egress, uncorroborated. Sticky by construction: written once, + * never cleared by code. The WEAK member of the sticky trio, so it carries + * the mildest cap below. */ + privacyFlaggedAt: Date | null + /** `user.privacy_corroborated_at` — first request where a second provider + * agreed the egress was anonymizing. Sticky. */ + privacyCorroboratedAt: Date | null + /** `user.third_party_client_at` — first free-mode request carrying a tool + * schema no Freebuff client ships. Sticky, and behavioural rather than + * network-derived, which is what makes it worth a hard cap. */ + thirdPartyClientAt: Date | null + /** + * The privacy verdict on the CURRENT request, from `getFreeModeRiskScore`. + * The one live signal in an otherwise durable set, and the one a user can + * change in a second by toggling a VPN — which is why it can only CAP a + * level, never contribute points. + */ + currentRiskScore: number | null +} + +/** A signal that moved the score, in user-facing language. */ +export interface FreebuffTrustFactor { + id: string + label: string + points: number +} + +/** Something the user can do to move up, with what it is worth. */ +export interface FreebuffTrustNextStep { + id: string + label: string + detail: string + points: number + /** Where the UI should send them. Relative to the freebuff web app. */ + href?: string +} + +export interface FreebuffTrustAssessment { + level: FreebuffTrustLevel + /** 0-100. Exposed so a user can see movement between levels, and so the + * thresholds below are auditable from the outside. */ + score: number + /** The level the score alone earned, before caps. Equal to `level` unless a + * cap applied — which is how the UI knows to explain the cap rather than + * telling someone with 80 points to keep earning points. */ + uncappedLevel: FreebuffTrustLevel + /** Which cap bound, if any. */ + cappedBy: string | null + factors: FreebuffTrustFactor[] + nextSteps: FreebuffTrustNextStep[] +} + +// --------------------------------------------------------------------------- +// Scoring +// --------------------------------------------------------------------------- + +/** + * The two signals a user can act on TODAY, from any country, with no aged + * GitHub account and no money. + * + * ## Why the caps are where they are + * + * They were originally 4 bounties (20) and 5 referrals (15). Together that is + * 35 points against a `core` threshold of 75, and the whole earned-only route + * — bounties, referrals, 90 days of account age, 30 active days, a clean + * residential signup — topped out at **70**. Five points short. `core` was + * literally unreachable by contribution: it required either an aged GitHub + * account or a payment. + * + * That is backwards for a program whose stated purpose is to give developers + * in unsupported regions a way to raise their own limits. It also made the + * incentive flat exactly where it should be steep — a user who completed ten + * bounties and referred twenty people scored the same as one who did four and + * five. + * + * At 6 and 10 the earned-only route reaches 95, so `core` is attainable by + * work alone, and each additional bounty or referral keeps paying well past + * the point where someone has proved they are real. + * + * ## Why raising them costs nothing against abuse + * + * Neither is cheap to manufacture. A bounty is reviewed proof-of-work, daily + * capped, and carries an anti-fraud agreement the claimant signs + * (docs/freebuff-bounties.md). A qualified referral requires the REFERRED + * account to hold a GitHub account four calendar months old and to actually + * use the product, and referral farming has its own detector and clawback path + * (docs/freebuff-abuse-referral-farming.md). An operator who can produce ten + * qualified referrals has already cleared a higher bar than this scorer sets. + */ +export const FREEBUFF_TRUST_EARNED = { + BOUNTY_POINTS: 5, + BOUNTY_CAP: 6, + REFERRAL_POINTS: 3, + REFERRAL_CAP: 10, +} as const + +const MAX_BOUNTY_POINTS = + FREEBUFF_TRUST_EARNED.BOUNTY_POINTS * FREEBUFF_TRUST_EARNED.BOUNTY_CAP +const MAX_REFERRAL_POINTS = + FREEBUFF_TRUST_EARNED.REFERRAL_POINTS * FREEBUFF_TRUST_EARNED.REFERRAL_CAP + +/** Minimum score for each level. `new` is the floor and needs no entry. */ +export const FREEBUFF_TRUST_THRESHOLDS: Record< + Exclude, + number +> = { + verified: 25, + established: 50, + core: 75, +} + +const DAY_MS = 24 * 60 * 60 * 1000 +const MONTH_MS = 30 * DAY_MS +const YEAR_MS = 365 * DAY_MS + +function ageMs(date: Date | null, now: Date): number | null { + if (!date) return null + const age = now.getTime() - date.getTime() + // A future date is a clock skew or a bad backfill, not an old account. + return age >= 0 ? age : 0 +} + +/** Highest threshold `score` clears. */ +function levelForScore(score: number): FreebuffTrustLevel { + if (score >= FREEBUFF_TRUST_THRESHOLDS.core) return 'core' + if (score >= FREEBUFF_TRUST_THRESHOLDS.established) return 'established' + if (score >= FREEBUFF_TRUST_THRESHOLDS.verified) return 'verified' + return 'new' +} + +const PRIVACY_EGRESS_SIGNALS = new Set(['vpn', 'proxy', 'tor', 'hosting']) + +function hasPrivacyEgressAtSignup( + signals: readonly string[] | null, +): boolean | null { + if (signals === null) return null + return signals.some((signal) => + PRIVACY_EGRESS_SIGNALS.has(signal.trim().toLowerCase()), + ) +} + +/** + * Score an account and resolve its level. + * + * Pure, so the policy is unit-testable and so the same function can run on the + * server (to enforce) and be reasoned about from a test (to check nobody moved + * a threshold by accident). Every I/O concern lives in the caller. + * + * ## Points earn, penalties cap + * + * Positive signals add points. Negative signals mostly do NOT subtract — they + * impose a CEILING on the resulting level. The difference matters: a + * subtracting penalty is defeated by accumulating enough of anything else, + * which is precisely what a farm operator with 200 aged GitHub accounts can + * do. A ceiling is not, and it also degrades honestly — a VPN user who cannot + * exceed `established` still gets `established`, which is today's limits. + */ +export function assessFreebuffTrust( + signals: FreebuffTrustSignals, + now: Date = new Date(), +): FreebuffTrustAssessment { + const factors: FreebuffTrustFactor[] = [] + const nextSteps: FreebuffTrustNextStep[] = [] + + const add = (id: string, label: string, points: number) => { + if (points === 0) return + factors.push({ id, label, points }) + } + const step = (s: FreebuffTrustNextStep) => nextSteps.push(s) + + // --- GitHub ------------------------------------------------------------- + // The heaviest block (up to 45) because it is the only one an abuser has to + // BUY. `docs/referrals.md` sets the economic invariant this inherits: keep + // the reward worth less than the grey-market price of an aged, qualifying + // GitHub account, and farming stops penciling out. + const githubAge = ageMs(signals.githubAccountCreatedAt, now) + if (githubAge === null) { + step({ + id: 'connect_github', + label: 'Connect your GitHub account', + detail: + 'Linking a GitHub account you have had for a while is the fastest way to raise your limits. We read the account and oldest-repo creation dates, which GitHub sets and nobody can backdate.', + points: 30, + href: '/web/settings', + }) + } else { + add('github_linked', 'GitHub account connected', 10) + if (githubAge >= 3 * YEAR_MS) { + add('github_age', 'GitHub account over 3 years old', 20) + } else if (githubAge >= YEAR_MS) { + add('github_age', 'GitHub account over a year old', 15) + } else if (githubAge >= 6 * MONTH_MS) { + add('github_age', 'GitHub account over 6 months old', 10) + } else { + step({ + id: 'github_age', + label: 'Your GitHub account is still new', + detail: + 'Account age is worth up to 20 points and grows on its own — nothing to do here but keep the same account connected.', + points: 10, + }) + } + + const repoAge = ageMs(signals.githubOldestRepoCreatedAt, now) + if (repoAge !== null && repoAge >= 6 * MONTH_MS) { + add('github_repo', 'Public repo over 6 months old', 10) + } + if ((signals.githubPublicRepos ?? 0) >= 3) { + add('github_repos', '3 or more public repos', 5) + } + if ((signals.githubFollowers ?? 0) >= 5) { + add('github_followers', '5 or more GitHub followers', 5) + } + if (signals.githubTwoFactorEnabled) { + add('github_2fa', 'Two-factor auth enabled on GitHub', 5) + } else if (signals.githubTwoFactorEnabled === false) { + step({ + id: 'github_2fa', + label: 'Turn on two-factor auth for GitHub', + detail: + 'Worth 5 points, and it protects the account your Freebuff limits now depend on.', + points: 5, + href: 'https://github.com/settings/security', + }) + } + } + + // --- Account history ---------------------------------------------------- + // Time and use, which cost an operator real calendar days per account and + // are the only signals a user gets for free by simply being real. + const accountAge = ageMs(signals.accountCreatedAt, now) + if (accountAge !== null) { + if (accountAge >= 90 * DAY_MS) { + add('account_age', 'Freebuff account over 90 days old', 15) + } else if (accountAge >= 30 * DAY_MS) { + add('account_age', 'Freebuff account over 30 days old', 10) + } else if (accountAge >= 7 * DAY_MS) { + add('account_age', 'Freebuff account over 7 days old', 5) + } + } + + if (signals.activeDays >= 30) { + add('active_days', 'Used Freebuff on 30+ days', 10) + } else if (signals.activeDays >= 7) { + add('active_days', 'Used Freebuff on 7+ days', 5) + } + + // --- Earned ------------------------------------------------------------- + // The routes that work from anywhere, on any account age. This is the answer + // to "I am in an unsupported region and my account is new": both of these + // are available today and together are worth 35 points, which is a level and + // a half. + const bountyPoints = + Math.min(signals.approvedBounties, FREEBUFF_TRUST_EARNED.BOUNTY_CAP) * + FREEBUFF_TRUST_EARNED.BOUNTY_POINTS + if (bountyPoints > 0) { + add( + 'bounties', + `${signals.approvedBounties} approved ${signals.approvedBounties === 1 ? 'bounty' : 'bounties'}`, + bountyPoints, + ) + } + if (bountyPoints < MAX_BOUNTY_POINTS) { + step({ + id: 'bounties', + label: 'Complete a bounty', + detail: `Approved bounties are worth ${FREEBUFF_TRUST_EARNED.BOUNTY_POINTS} points each, up to ${MAX_BOUNTY_POINTS}. They are reviewed, they work from any country, and they pay session grants on top.`, + points: MAX_BOUNTY_POINTS - bountyPoints, + href: '/web/earn', + }) + } + + const referralPoints = + Math.min(signals.qualifiedReferrals, FREEBUFF_TRUST_EARNED.REFERRAL_CAP) * + FREEBUFF_TRUST_EARNED.REFERRAL_POINTS + if (referralPoints > 0) { + add( + 'referrals', + `${signals.qualifiedReferrals} qualified ${signals.qualifiedReferrals === 1 ? 'referral' : 'referrals'}`, + referralPoints, + ) + } + if (referralPoints < MAX_REFERRAL_POINTS) { + step({ + id: 'referrals', + label: 'Invite other developers', + detail: `Each friend who signs up with a real GitHub account and uses Freebuff is worth ${FREEBUFF_TRUST_EARNED.REFERRAL_POINTS} points, up to ${MAX_REFERRAL_POINTS} — plus the referral rewards themselves.`, + points: MAX_REFERRAL_POINTS - referralPoints, + href: '/web/earn', + }) + } + + if (signals.hasPaid) { + add('paid', 'Supported Freebuff with a purchase', 25) + } + + // --- Provenance --------------------------------------------------------- + // Small positives only. These cannot earn a level on their own; their job is + // to let a clean, ordinary signup reach `verified` without owning anything. + const signupPrivacy = hasPrivacyEgressAtSignup(signals.signupPrivacySignals) + if (signupPrivacy === false) { + add('clean_signup', 'Signed up from a residential connection', 5) + } + if ( + signals.signupIpSource === 'edge_secret' || + signals.signupIpSource === 'cloudflare' + ) { + add('verified_signup_ip', 'Verified network at signup', 5) + } + + const score = Math.max( + 0, + Math.min( + 100, + factors.reduce((sum, factor) => sum + factor.points, 0), + ), + ) + const uncappedLevel = levelForScore(score) + + // --- Caps --------------------------------------------------------------- + // Applied after scoring, lowest wins. Each one names itself so the UI can + // explain a cap instead of telling a user with a high score to earn more. + let level = uncappedLevel + let cappedBy: string | null = null + const cap = (limit: FreebuffTrustLevel, reason: string) => { + const capped = lowerOf(level, limit) + if (capped !== level) { + level = capped + cappedBy = reason + } + } + + // A reversed ban is invisible here by construction — only unreversed events + // reach this field — so this is an account we actioned and did not take + // back. Not a ban (they are already unbanned) and not permanent, but not + // something to hand extra capacity to either. + if (signals.hasUnreversedBanEvent) { + cap('verified', 'past_enforcement') + } + + // Signed up behind a VPN/proxy/Tor/hosting egress. Capped, not zeroed: this + // describes a lot of privacy-conscious developers as well as every farm, and + // `established` is what everyone had before this file existed. + if (signupPrivacy === true) { + cap('established', 'signup_privacy_egress') + } + + // The live request is on an anonymizing network. Deliberately the only cap + // driven by a per-request signal, and deliberately the harshest, because it + // is the one an abuser toggles: without it, a farm signs up cleanly once and + // then runs everything through a proxy pool at core-member limits. + if (signals.currentRiskScore !== null && signals.currentRiskScore >= 75) { + cap('verified', 'anonymous_network') + } + + // The sticky flags: things this account has DONE, remembered past the + // request that revealed them. Without these, every network cap above is + // defeated by toggling the VPN off for a day — the exact wash-trading of + // signals the caps exist to prevent. They cap rather than subtract for the + // standard reason (see "Points earn, penalties cap"), and they grade by + // evidence weight: + // + // corroborated egress -> verified two providers agreed + // foreign tool schema -> verified behavioural, not network luck + // ipinfo-only egress -> established one provider, the weak signal -- + // `established` is what every + // account had before trust levels + // existed, so this cap forfeits + // only the `core` upside + // + // The 2026-08-03 mass-reversal is the reason none of these goes lower: + // network-derived evidence has wrongly actioned real users before, and a + // permanent flag with a harsh cap would make that mistake permanent too. + if (signals.privacyCorroboratedAt !== null) { + cap('verified', 'past_corroborated_egress') + } + if (signals.thirdPartyClientAt !== null) { + cap('verified', 'third_party_client') + } + if (signals.privacyFlaggedAt !== null) { + cap('established', 'past_privacy_egress') + } + + // Signup networks and mailboxes that many accounts share. `?? 1` matters: + // null is unknown (pre-provenance accounts), and unknown must not cap. + if ((signals.signupPrefixAccountCount ?? 1) >= 8) { + cap('established', 'shared_signup_network') + } + if ((signals.mailboxAccountCount ?? 1) >= 3) { + cap('verified', 'shared_mailbox') + } + + // Steps are ordered by what they are worth, EXCEPT that a binding cap goes + // first regardless. A capped account told to "complete a bounty for 20 + // points" when points are not what binds them is being sent on an errand, so + // the cap is prepended after the sort rather than competing in it — it + // carries no points and would otherwise sink to the bottom. + const earnedSteps = nextSteps.sort((a, b) => b.points - a.points) + const actionableSteps = + cappedBy === null + ? earnedSteps + : [ + { + id: `cap_${cappedBy}`, + label: CAP_REMEDIES[cappedBy]?.label ?? 'Your level is limited', + detail: + CAP_REMEDIES[cappedBy]?.detail ?? + 'Something about this account limits how high your level can go.', + points: 0, + }, + ...earnedSteps, + ] + + return { + level, + score, + uncappedLevel, + cappedBy, + factors: factors.sort((a, b) => b.points - a.points), + nextSteps: actionableSteps, + } +} + +/** + * User-facing explanation of each cap. + * + * Two of these describe something the user can fix in under a minute (turn the + * VPN off, use a different network) and are written to say exactly that. The + * other two describe history, and say so honestly rather than implying an + * action that does not exist — a "next step" a user cannot take is worse than + * no step at all. + */ +const CAP_REMEDIES: Record = { + past_corroborated_egress: { + label: 'This account has used an anonymizing network', + detail: + 'Requests from this account were confirmed to come through a VPN, proxy or similar exit. That history caps this account at Verified. Everything else still counts toward your level.', + }, + past_privacy_egress: { + label: 'This account has connected over a flagged network', + detail: + 'A connection from this account looked like an anonymizing network. That caps this account at Established. If this seems wrong — some office and university networks are misread — contact support.', + }, + third_party_client: { + label: 'A non-Freebuff client has used this account', + detail: + 'Requests from this account carried a client we do not ship. That caps this account at Verified. Only official Freebuff apps are supported on free mode.', + }, + anonymous_network: { + label: 'Turn off your VPN or proxy', + detail: + 'We cannot tell where requests from a VPN, proxy or Tor exit node come from, so those connections are capped at Verified no matter how much you have earned. Reconnect from your normal network and your level updates within a few minutes.', + }, + signup_privacy_egress: { + label: 'You signed up over a VPN or proxy', + detail: + 'That caps this account at Established. Everything else still counts, and the cap applies to this account only — it is not a strike against you.', + }, + shared_signup_network: { + label: 'Many accounts signed up from your network', + detail: + 'Shared offices, campuses and carrier NATs all look like this, so it caps rather than blocks. Approved bounties and referrals still raise your limits within the cap.', + }, + shared_mailbox: { + label: 'Several accounts share your email address', + detail: + 'Address variations that reach one inbox (dots, or anything after a +) count as one mailbox. Using a single account raises your level.', + }, + past_enforcement: { + label: 'This account was actioned in the past', + detail: + 'Your access is fully restored, but the level is capped at Verified. Contact support if you think that is wrong.', + }, +} + +// --------------------------------------------------------------------------- +// Wire shape +// --------------------------------------------------------------------------- + +/** + * What a client is told about its own standing. + * + * Carries the resolved LIMITS as well as the level, because a client that had + * to map level → limits itself would hold a second copy of the matrix above + * and drift from it on the first tuning pass. The server owns the numbers; the + * client renders whatever it is sent. + */ +/** + * What a level means, in words. + * + * ## Why the numbers do not leave the server + * + * Three reasons, and the first is the one that matters most: + * + * 1. **A published limit is a published target.** `docs/freebuff-abuse- + * detection.md` records that the abuse pattern here is not bursting, it is + * "sustained pacing just under the daily caps" — so telling an operator + * exactly where the cap sits is telling them exactly how to sit under it. + * Every threshold in this file is a number we would rather they had to + * discover. + * 2. **The numbers are ours, not theirs.** `dailySpendUsd` in particular is + * our provider cost, and a user shown "$25/day" learns something about our + * margins and nothing about what they may do. + * 3. **Exact figures invite exactly the wrong conversation.** The first + * version showed them and produced people comparing screenshots and asking + * whether a smaller number meant they had been punished. A limit is meant + * to answer "can I get my work done", and that question has a qualitative + * answer. + * + * So `FreebuffStandingInfo` carries these phrases and NOT `FreebuffTrustLimits` + * — the raw matrix never crosses the wire, which means no client can render it + * by accident and no future surface has to remember not to. + * + * Where a user genuinely needs a count, they already have an exact one: the + * model picker renders "N of M sessions used" from the live quota snapshot, + * which is authoritative and per-model. Duplicating it here could only + * disagree with it. + */ +export interface FreebuffStandingHighlight { + label: string + value: string +} + +const LIMIT_PHRASES: Record< + FreebuffTrustLevel, + { prompts: string; depth: string; premium: string } +> = { + new: { + prompts: 'Enough to get a project started', + depth: 'Focused, shorter agent runs', + premium: 'Occasional access', + }, + verified: { + prompts: 'Comfortable for everyday work', + depth: 'Full agent runs', + premium: 'Regular access', + }, + established: { + prompts: 'Comfortable on heavy days', + depth: 'Long runs with plenty of subagents', + premium: 'Generous access', + }, + core: { + prompts: 'The most we offer', + depth: 'The most we offer', + premium: 'The most we offer', + }, +} + +export function freebuffStandingHighlights( + accessTier: FreebuffAccessTier, + level: FreebuffTrustLevel, +): FreebuffStandingHighlight[] { + const phrases = LIMIT_PHRASES[level] + return [ + { label: 'Prompts a day', value: phrases.prompts }, + { label: 'Work per prompt', value: phrases.depth }, + { + label: 'Premium models', + // Stated as a region fact rather than as something this account lacks: + // no level in the limited row can reach a premium model, so framing it + // as a level shortfall would send the user chasing points that cannot + // buy it. + value: + freebuffTrustLimits(accessTier, level).premiumSessionsPerDay > 0 + ? phrases.premium + : 'Not available in your region yet', + }, + ] +} + +/** + * NOTE FOR CALLERS: `highlights` is what the level WOULD grant, which is only + * what the account actually gets once `FREEBUFF_TRUST_LEVELS=enforce`. Both + * producers gate on that (the Earn route and the session `standing` field), so + * a client that receives this can render it as fact. A third producer must do + * the same — see the comment in freebuff/web/src/app/api/web/standing/route.ts + * for what happens otherwise. + */ +export interface FreebuffStandingInfo { + level: FreebuffTrustLevel + label: string + blurb: string + score: number + /** Score at which the next level starts, or null at `core`. */ + nextLevelAt: number | null + nextLevel: FreebuffTrustLevel | null + cappedBy: string | null + cappedReason: string | null + factors: FreebuffTrustFactor[] + nextSteps: FreebuffTrustNextStep[] + accessTier: FreebuffAccessTier + /** Semantic, never numeric — see FreebuffStandingHighlight. */ + highlights: FreebuffStandingHighlight[] +} + +export function toFreebuffStandingInfo( + assessment: FreebuffTrustAssessment, + accessTier: FreebuffAccessTier, +): FreebuffStandingInfo { + const index = FREEBUFF_TRUST_LEVELS.indexOf(assessment.level) + const nextLevel = FREEBUFF_TRUST_LEVELS[index + 1] ?? null + return { + level: assessment.level, + label: FREEBUFF_TRUST_LEVEL_LABELS[assessment.level], + blurb: FREEBUFF_TRUST_LEVEL_BLURBS[assessment.level], + score: assessment.score, + nextLevel, + nextLevelAt: + nextLevel && nextLevel !== 'new' + ? FREEBUFF_TRUST_THRESHOLDS[nextLevel] + : null, + cappedBy: assessment.cappedBy, + cappedReason: assessment.cappedBy + ? (CAP_REMEDIES[assessment.cappedBy]?.detail ?? null) + : null, + factors: assessment.factors, + nextSteps: assessment.nextSteps, + accessTier, + highlights: freebuffStandingHighlights(accessTier, assessment.level), + } +} diff --git a/common/src/constants/provider-routes.ts b/common/src/constants/provider-routes.ts new file mode 100644 index 0000000000..4f3a0bde3f --- /dev/null +++ b/common/src/constants/provider-routes.ts @@ -0,0 +1,730 @@ +export const PROVIDER_ROUTE_IDS = [ + 'fireworks/deployment', + 'fireworks/serverless', + 'minimax/official', + 'xiaomi/official', + 'openrouter/novita/fp8', + 'mimo/openrouter', + 'infron/makora', + 'glm/crof', + 'glm/infron', + 'glm-5-3-flash/crof', + 'glm-5-3-flash/fallback', + 'deepseek/openrouter', + 'deepseek/crof', + 'deepseek/cheaper-inference', + 'deepseek/luminal', + 'deepseek/fusioncode', + 'deepseek/runinfra', + 'deepseek/official', + 'luna/fallback', + 'luna/primary', +] as const + +export type ProviderRouteId = (typeof PROVIDER_ROUTE_IDS)[number] + +/** + * A GPT-5.6 Luna session that diverted off its primary lane because that lane + * was OUT OF CAPACITY, and should now enter on the fallback instead. + * + * Named for the ROLE, not the upstream, and deliberately so: this value is + * persisted in `free_session.provider_route` and read back unvalidated, so an + * id naming its provider either forces a migration or becomes a name that lies + * the moment the fallback moves. {@link MIMO_OPENROUTER_PROVIDER_ROUTE} carries + * the same warning for the same reason. {@link LUNA_FALLBACK_UPSTREAM} below + * says who currently serves it. + * + * Why pin at all: Luna's lanes each keep their OWN prompt cache, and an agent + * turn re-sends its whole prefix every step, so ~96.5% of its tokens are cache + * reads. Moving a live session between providers therefore costs a full cold + * prefill on a prefix it has already paid to warm. Pinning moves SESSIONS, not + * requests: the session that overflowed stays on the fallback and warms there + * once, instead of re-paying at every turn. + */ +/** + * A Flash session GRANTED the FusionCode front lane by the admission + * controller — the same shape as the Luminal grant, and granted only when + * Luminal declined, since Luminal is ~45% cheaper. Like Luminal's pin this is + * a grant, not a cohort mark: its own env switch is the drain. + */ +export const DEEPSEEK_FUSIONCODE_PROVIDER_ROUTE = + 'deepseek/fusioncode' satisfies ProviderRouteId + +export const LUNA_FALLBACK_PROVIDER_ROUTE = + 'luna/fallback' satisfies ProviderRouteId + +/** + * A Luna session explicitly returned to the primary lane after the fallback + * failed it. + * + * Distinct from UNPINNED, which also enters on the primary: an unpinned session + * simply never diverted, while this one diverted and came back. Keeping them + * apart is what lets the fallback rate be read as "sessions that overflowed" + * rather than "sessions that overflowed and stayed overflowed". + */ +export const LUNA_PRIMARY_PROVIDER_ROUTE = + 'luna/primary' satisfies ProviderRouteId + +/** + * Who serves {@link LUNA_FALLBACK_PROVIDER_ROUTE} today. Repointing this moves + * every session already pinned there, with no migration — which is the whole + * reason the id above does not name a provider. + */ +export const LUNA_FALLBACK_UPSTREAM = 'cheaper-inference' as const + +export const FIREWORKS_DEPLOYMENT_PROVIDER_ROUTE = + 'fireworks/deployment' satisfies ProviderRouteId +export const FIREWORKS_SERVERLESS_PROVIDER_ROUTE = + 'fireworks/serverless' satisfies ProviderRouteId +export const MINIMAX_OFFICIAL_PROVIDER_ROUTE = + 'minimax/official' satisfies ProviderRouteId +/** + * MiMo 2.5's OpenRouter lane — the ENTRY lane since 2026-08-23, when Xiaomi's + * direct rate limit made it untenable as the primary (see mimo-router.ts for + * the measured 429 rate that settled it). + * + * THE PIN IS NOW INERT, exactly as {@link GLM_CROF_PROVIDER_ROUTE} became when + * its lane was promoted: it names the lane a session would enter anyway. It is + * still recognized on READ, and must stay so, because ~48k sessions a day were + * pinned here under the old order and their pins outlive the deploy. Reading it + * as "start at the primary" is exactly right for them — they are already warm + * on this lane, so they carry on with no cold prefill. + * + * Like {@link DEEPSEEK_INFRON_MAKORA_PROVIDER_ROUTE} it says *which lane*, NOT + * which upstream serves it — that is {@link MIMO_OPENROUTER_UPSTREAM_ORDER} + * below, so repointing the upstream moves every session here with no migration. + * + * Named generically on purpose. Its predecessor + * {@link MIMO_NOVITA_PROVIDER_ROUTE} baked the upstream into a value that gets + * persisted in `free_session.provider_route` and read back unvalidated, so + * changing upstreams meant either a migration or a name that lies. + */ +export const MIMO_OPENROUTER_PROVIDER_ROUTE = + 'mimo/openrouter' satisfies ProviderRouteId +/** + * Marks a MiMo session as diverted OFF the OpenRouter lane onto Xiaomi's direct + * API — the backup since the 2026-08-23 swap, and the direction this pin has + * pointed only since then. + * + * It exists for the reason {@link GLM_CROF_PROVIDER_ROUTE} does: depth bought + * not with money but with a SECOND ACCOUNT AND A SECOND BALANCE. Both MiMo + * lanes ran dry inside 14 hours (Xiaomi 2026-08-22 17:30Z, OpenRouter + * 2026-08-23 07:35Z), which is the whole argument for keeping two funded + * accounts and a 402 that can cross between them. + * + * Reuses the long-declared but never-written `xiaomi/official` id, so no + * `PROVIDER_ROUTE_IDS` entry had to be added. + */ +export const MIMO_XIAOMI_PROVIDER_ROUTE = + 'xiaomi/official' satisfies ProviderRouteId +/** + * The upstreams that serve {@link MIMO_OPENROUTER_PROVIDER_ROUTE}, preferred + * first: Xiaomi's OWN endpoint, reached through OpenRouter's account rather + * than our direct API key, with Novita behind it purely as depth. + * + * It lives next to the route id because the two are meant to be read together — + * the id says which lane a session is pinned to, this says who serves it — and + * in `common/` rather than in mimo-router.ts so `scripts/mimo-smoke.ts` can + * assert against the REAL value without importing the billing chain. A smoke + * test that restates this config would keep passing after the upstream moved, + * reporting a lane it no longer covers. + * + * Novita served the lane until 2026-08-01, which was most of what made the + * fallback expensive. Measured over 6h of prod that day, attributing rows by + * whether their cost reproduces our Xiaomi formula: + * + * openrouter/novita 18,268 reqs (42.6%) 19.2% cache $364.11 $0.138/M in + * xiaomi direct 24,631 reqs (57.4%) 90.5% cache $63.76 $0.018/M in + * + * Novita is also 20% dearer per token before caching ($0.168/$0.336 against + * Xiaomi's $0.14/$0.28) and prices cache reads at $0.0034/M against $0.0028/M. + * Xiaomi's OpenRouter endpoint is priced identically to our direct rate, so the + * lane costs the same as the primary and differs only in whose rate limit and + * prompt cache it draws on — which is, since 2026-08-23, the entire reason this + * lane is the ENTRY rather than the backup. + */ +export const MIMO_OPENROUTER_UPSTREAM_ORDER = [ + 'xiaomi/fp8', + 'novita/fp8', +] as const + +/** + * Fresh OpenRouter `provider` block for the MiMo lane. + * + * TWO ENTRIES DELIBERATELY, and it must never go back to one. A pinned session + * has no health check and no un-pin path — `routeWithStickyFallback` short + * -circuits straight to the fallback — so if the lane's only upstream is down, + * one transient Xiaomi blip wedges every remaining request in that session + * rather than degrading a single one. That is not hypothetical: a one-deep pin + * on DeepSeek's Infron lane took out 1,160 requests across 191 users for ~32h + * when `makora` went offline (2026-07-26/27, fixed in #1045). + * + * Verified live on this lane 2026-08-01: within an `allow_fallbacks:false` + * order list, an unroutable first entry is SKIPPED rather than failing the + * request — `order:['nosuchprovider','xiaomi/fp8']` returned 200 from Xiaomi — + * while `order:['xiaomi/fp8','novita/fp8']` still serves from Xiaomi when it is + * healthy. So the depth costs nothing in the normal case and is the difference + * between a degraded session and a wedged one in the bad case. Novita is second + * because it is the best-understood host for this model: it served ~43% of MiMo + * traffic through late July, so its compatibility with our request shape is + * proven, and it only ever serves when Xiaomi cannot. + * + * Returns a NEW object with a NEW array every call. These arrays get aliased + * into an outgoing request body, and this file's peer + * `INFRON_PROVIDER_ORDER` was made copy-on-assignment in #1045 for exactly that + * reason — one downstream mutation would corrupt routing process-wide. + */ +export function mimoOpenRouterProvider(): Record { + return { + order: [...MIMO_OPENROUTER_UPSTREAM_ORDER], + allow_fallbacks: false, + } +} +/** + * LEGACY MiMo fallback pin, written while the OpenRouter lane was hardcoded to + * Novita FP8. Still recognized on READ so sessions pinned before + * {@link MIMO_OPENROUTER_PROVIDER_ROUTE} shipped keep serving from that lane + * instead of silently reverting to a Xiaomi-direct attempt they already failed. + * Never written anymore; expires with its session. Inert since the 2026-08-23 + * swap for the same reason its successor is — see there. + */ +export const MIMO_NOVITA_PROVIDER_ROUTE = + 'openrouter/novita/fp8' satisfies ProviderRouteId +/** + * GLM 5.2's CrofAI lane. Was the backup for one day (the 2026-08-20 Infron + * cutover); the ENTRY lane again since 2026-08-21, when production measurement + * showed Infron costs 2x — see GLM_INFRON_PROVIDER_ROUTE for the numbers. + * + * Still recognized on READ so the handful of sessions pinned here during that + * day keep working. The pin is now inert: it names the lane they would enter + * anyway. + * + * GLM 5.2 entered the day served by CrofAI alone. It now runs a two-lane sticky + * cascade — Infron's Alibaba group first, CrofAI behind it — so this id exists + * for the same reason {@link MIMO_OPENROUTER_PROVIDER_ROUTE} does: to record + * that a session left the entry lane and must not be sent back, because each + * upstream keeps its own prompt cache and flapping between them re-pays a cold + * prefill every turn. + * + * NOTE THE DIRECTION OF THE MONEY, because it is the reverse of every other + * cascade in this file: CrofAI — the BACKUP — is cheaper on all three terms, + * by 1.83x on input and output and 2.75x on cache reads. So this lane is not + * depth bought with money, it is depth bought with a SECOND ACCOUNT AND A + * SECOND BALANCE — the property {@link DEEPSEEK_RUNINFRA_PROVIDER_ROUTE} exists + * for, and the one CrofAI's own 401 "Not Enough Credits" proved GLM needed. The + * order was set by request; the measured table behind those ratios is in + * web/src/llm-api/glm-router.ts. + * + * Like its peers the id names the LANE, not the upstream — that is + * INFRON_PROVIDER_ORDER, keyed by model — so repointing which Alibaba region + * serves the entry lane needs no migration, while renaming this value would: + * it is persisted in `free_session.provider_route` and read back unvalidated. + */ +export const GLM_CROF_PROVIDER_ROUTE = 'glm/crof' satisfies ProviderRouteId +/** + * GLM 5.2's Infron lane — the BACKUP again as of 2026-08-21, after one hour of + * head-to-head production traffic settled the question the cutover opened. + * + * Both lanes served comparable work in the same window (147k vs 143k average + * input tokens, 95 vs 89 distinct users): + * + * msgs cache $/msg $/M input median $/prompt + * Infron 2,014 94.6% 0.024241 0.1640 0.1265 + * CrofAI 1,203 89.0% 0.012063 0.0842 0.0615 + * + * Infron really does cache BETTER — 94.6% against 89.0% — and is still 2.0x + * dearer per message and 2.06x per user prompt, because its cache reads cost + * 2.75x. The measurement matches the rate card to within a percent, so this is + * price, not noise. + * + * IT CANNOT BE FIXED BY WARMING. Solving h*0.1375 + (1-h)*0.55 = 0.0842 for + * Infron's break-even hit rate gives h = 1.129 — above 100%. At a perfect cache + * Infron would still cost $0.1375/M against CrofAI's measured $0.0842. The + * order can only flip if CrofAI falls below ~65% cache while Infron holds + * ~100%, so do not re-litigate this on a single bad CrofAI hour. + * + * What Infron is still here for is what it was always actually buying: a second + * account and a second prepaid balance behind a model whose entitlement is + * earned. CrofAI answers 401 "Not Enough Credits" when its balance runs dry, + * and that used to be a total outage. + */ +export const GLM_INFRON_PROVIDER_ROUTE = 'glm/infron' satisfies ProviderRouteId +/** + * A GLM 5.3 Flash session that diverted OFF the Merge Gateway lane and should + * now enter on the OpenRouter route it was served from before 2026-08-27. + * + * Named for the ROLE, not the upstream, for the reason spelled out on + * {@link LUNA_FALLBACK_UPSTREAM}: this value is persisted in + * `free_session.provider_route` and read back unvalidated, so an id naming its + * provider either forces a migration or becomes a name that lies the moment the + * fallback moves. + * + * WHY THERE IS A FALLBACK AT ALL, when Merge is cheaper on every term by 5-6x + * (table below): Merge bills a PREPAID BALANCE, and it reports the remaining + * one on every response as `x-credit-balance-usd`. A prepaid balance behind a + * single-lane model is the exact outage this repo has already taken twice — + * CrofAI's 401 "Not Enough Credits" made GLM 5.2 a total outage, which is why + * {@link GLM_INFRON_PROVIDER_ROUTE} exists, and the same shape took DeepSeek's + * cascade to three lanes. The balance on this account was $20 at integration, + * which is a trial, not a runway. + * + * THE PRICE TABLE, read from the gateway's own /v1/models on 2026-08-27 and + * confirmed against billed `cost` on live requests to the cent: + * + * input/M cache read/M output/M + * Merge Gateway $0.012 $0.003 $0.04 + * OpenRouter (cheap) $0.075 $0.015 $0.25 + * ratio 6.25x 5.0x 6.25x + * + * THE LINE THAT DECIDES IT: Merge's FRESH INPUT ($0.012/M) is below + * OpenRouter's CACHE READ ($0.015/M). So Merge at a 0% hit rate is still + * cheaper than OpenRouter at a perfect one, and no cache-rate regression can + * make this lane the expensive choice. That is the opposite of every previous + * cutover in this file — the CrofAI DeepSeek lane needed ~90% cache to break + * even and delivered 60-85%, and Infron above cannot break even at any hit rate + * — so the usual "compare lanes at their MEASURED hit rate or not at all" trap + * has nothing to bite on here. Measured anyway, and note WHICH number to quote: + * a tight loop on a byte-identical prefix gives 95.8%, but a growing 14-turn + * conversation — the shape an agent turn actually has — gives 57.0%, and that + * is the one the bill follows. Measured saving on that run: 5.00x. Even so, + * Merge at 0% cache cost less than OpenRouter would have at 100% (1.27x), which + * is the whole argument. + * + * The fallback is therefore bought with money, unlike Infron's: diverting costs + * ~6x. It is worth it only because the alternative is serving nothing. + */ +/** + * A GLM 5.3 Flash session that diverted off Merge Gateway onto CrofAI — the + * MIDDLE rung, added 2026-08-27. + * + * COST PARITY WITH THE LANE BEHIND IT, not an improvement on it, and the + * difference matters. Costed at an EQUAL 82.3% cache against the repo's + * reference agent turn, CrofAI looks 1.24x cheaper than the OpenRouter band. + * Measured at the rates the two lanes ACTUALLY deliver on identical work, + * CrofAI hit 74.2% and OpenRouter 85.6%, which puts them at $0.02548 and + * $0.02364 per M of input — OpenRouter 1.08x ahead. The card advantage + * reverses, so do not justify this rung on price. + * + * It is here for DEPTH: a second account and a second prepaid balance between + * Merge and OpenRouter, the same thing {@link GLM_INFRON_PROVIDER_ROUTE} buys + * for GLM 5.2 — and it is worth more than usual here because Merge's + * availability is poor. On the day this was added Merge's two vendors took + * turns failing (`zai` 14-of-20 errors in the morning, `particle` 0-of-24 by + * evening) and unpinned traffic did NOT route around the sick one, while CrofAI + * was 6/6 at every point one of them was not. CrofAI is also much faster: p50 + * 3.4s against OpenRouter's 14.4s on the same 14 turns. + * + * Merge's OTHER vendor is deliberately not a rung. `zai` is priced at exactly + * OpenRouter's cheap band, so it is dominated by both lanes below it. + * + * The lane that IS a large saving is the one in front: Merge at $0.015/M of + * fresh input beats both of these at their measured cache rates (1.58x under + * OpenRouter) even at a 0% hit rate, and 3-5x under it at any normal one. + */ +/** + * GLM 5.3 Flash's OpenRouter endpoint preference, healthiest first. + * + * ADDED 2026-08-28 IN RESPONSE TO USER REPORTS ("very unstable and stops + * sometimes"), which the telemetry bore out precisely. Over 24h of prod: + * + * stream-interrupt rate 0.78% the WORST of any model in the catalog — + * 4.6x MiMo and V4 Flash, 5.6x Luna + * provider failures 1,285 = 4.29% of all GLM 5.3 traffic + * of which GMICloud 1,170 91.1%, nearly all `Backend request failed + * with status 400 / backend_error` + * of which Z.AI 98 7.6%, Z.AI's OWN account: error 1113, + * "Insufficient balance or no resource + * package" — upstream of us, not our key + * + * The route previously carried a ceiling and nothing else, on the reasoning + * that "there is no endpoint worth PREFERRING — they are the same price". + * Production falsified the premise: the three endpoints under + * FREEBUFF_GLM_V53_FLASH_MAX_PRICE are the same price and are emphatically NOT + * the same reliability. OpenRouter itself had already deranked GMICloud to + * `status: -2` while continuing to send it most of our traffic. + * + * PREFERENCE, NOT EXCLUSION, and that distinction is the whole design. Only + * THREE endpoints sit under the ceiling (Z.AI, Novita, GMICloud); everything + * else on this model is exactly 2x. Dropping GMICloud with `ignore` would leave + * two — one of which is the Z.AI account that is already out of credit — and + * when every endpoint under a ceiling is unavailable OpenRouter returns 404 + * rather than serving above it. That is the Ox Alpha trap, and its documented + * fix is never to raise the number. So GMICloud stays reachable as a last + * resort and simply stops being the default. + * + * All three serve the same `fp8` quantization of the same dated build + * (`z-ai/glm-5.3-flash-20260826`), so this reorders reliability without + * touching output quality. + * + * A SECOND BENEFIT, since each endpoint keeps its own prompt cache: preferring + * one stops the route spraying across three of them. A measured 16-turn run on + * this lane was served by GMICloud x10, Novita x3 and Z.AI x1 and cached 85.4%; + * concentrating the traffic should raise that as well as steady it. + */ +export const GLM_V53_FLASH_OPENROUTER_UPSTREAM_ORDER = [ + // Healthy at status 0, and carries no failures in the 24h window above. + 'novita/fp8', + // Also status 0, but its own balance is dry — worth second place rather than + // first until that clears, and worth keeping ahead of the deranked one. + 'z-ai/fp8', + // Deliberately LAST rather than absent. See the Ox Alpha note above. + 'gmicloud/fp8', +] as const + +export const GLM_V53_FLASH_CROF_PROVIDER_ROUTE = + 'glm-5-3-flash/crof' satisfies ProviderRouteId +export const GLM_V53_FLASH_FALLBACK_PROVIDER_ROUTE = + 'glm-5-3-flash/fallback' satisfies ProviderRouteId +/** + * DeepSeek V4 Flash's CrofAI lane — and, since the 2026-08-15 cutover, the + * COHORT MARK that says a session belongs on it. + * + * This id now carries two meanings that deliberately coincide. Written at + * ADMISSION it means "this session was admitted after the cutover, so it enters + * the cascade on CrofAI". Written by the CASCADE it means "this session + * diverted onto CrofAI". Both want the same thing — enter on CrofAI — which is + * why pins written by the pre-cutover code needed no migration when the + * cutover shipped, and why nothing has to distinguish them on read. + * + * A session with NO pin is, by construction, one admitted before the cutover + * shipped: it runs the pre-cutover order and keeps the prompt cache it has + * already paid to warm. See `deepseekEntryLane` and + * docs/freebuff-deepseek-provider-cutover.md. + * + * Reference prices per M — CrofAI/Infron/OpenRouter from live billing + * 2026-08-04, RunInfra from runinfra.ai/pricing and Infron re-read from its + * catalog on 2026-08-16, DeepSeek from its published card after the 16:00 UTC + * 2026-08-16 repricing: + * + * input cache read output + * CrofAI 0731 0.1200 0.0030 0.2100 + * RunInfra 0731 0.1300 0.0100 0.2700 + * Infron alibaba 0.2120 0.0210 0.6360 + * DeepSeek off-peak 0.2200 0.0070 0.6600 + * DeepSeek peak 0.4400 0.0140 1.3200 + * (retired) OpenRouter 0.0881 0.0176 0.1761 + * + * A coding turn re-sends its whole prefix every step, so cache reads are most + * of the tokens and that is the term that decides the bill. Infron's own + * 2026-08-16 repricing (from 0.0690/0.0144/0.1375) turned it from the cheapest + * lane into the dearest, which is what put {@link + * DEEPSEEK_RUNINFRA_PROVIDER_ROUTE} ahead of it. + * + * The DeepSeek repricing also made CrofAI the cheapest lane outright rather + * than a near-tie: it is now cheaper than DeepSeek direct on every term, by + * 2.3x on cache reads off-peak and 4.7x at peak. The lane ORDER has not been + * revisited to match — see docs/freebuff-deepseek-provider-cutover.md. + * + * It also serves `deepseek-v4-flash-0731` — the GA build, the same one + * DeepSeek's own API serves — where Infron's undated slug is a frozen preview + * snapshot. So this lane matches the DeepSeek-direct lane behind it in + * behaviour as well as price. + */ +export const DEEPSEEK_CROF_PROVIDER_ROUTE = + 'deepseek/crof' satisfies ProviderRouteId +/** + * DeepSeek's own API — the lane a cutover session diverts to, and the lane + * every PRE-cutover session still enters on. + * + * It was the unpinned default until 2026-08-15, which is why it had no route id + * before: a lane nothing ever moves to needs no name. Now that a cutover + * session can divert here, it has to be able to say so, and to stay — its + * prompt cache is warm on this upstream, and sending the next turn back to a + * CrofAI that just failed would pay a cold prefill to reach a lane we already + * know is unhealthy. The pin skips CrofAI as an ENTRY point only: it stays in + * the order behind this lane, because by the next turn the blip has usually + * cleared and CrofAI is still far cheaper than the lanes below. + * + * Behind CrofAI for cutover sessions because DeepSeek is the side that + * repriced — as of 16:00 UTC 2026-08-16 its cache reads are $0.0070/M off-peak + * and $0.0140/M at peak against CrofAI's $0.0030, so what used to be a + * marginal loss on that term is now a 2.3-4.7x one — and because of the + * failure record that made this a cascade at all: on 2026-08-03/04 it shed peak + * load with 3,934 x 503 "Service is too busy" and diverted 4,997 sessions at + * once, and on 2026-08-11 it accepted 650 requests in six hours and then sent + * nothing, tripping the four-minute first-token watchdog. Note that the + * watchdog is DeepSeek-direct-only (see `handleDeepSeekStream`); no equivalent + * guards the CrofAI lane now serving in front of it. + */ +/** + * DeepSeek V4 Flash's Luminal lane — a small FREE grant, and the only pin in + * this file that is RATIONED rather than reactive. + * + * Every other route id here records where a session ENDED UP after something + * failed. This one records that a session WON a slot: Luminal donated a slice + * of Flash capacity far below our volume, so the pin is minted by an admission + * controller (web/src/server/free-session/luminal-admission.ts) that hands out + * a bounded number of them and stops when Luminal starts refusing. + * + * It is FIRST in the cascade for the sessions that carry it, which no other + * cheap-lane experiment has earned, because it is free and the lanes behind it + * are not: CrofAI's cache reads are $0.0030/M, DeepSeek direct's are + * $0.0070-0.0140/M, and this is $0. The OpenRouter retirement on {@link + * DEEPSEEK_CROF_PROVIDER_ROUTE} is the cautionary tale for adding a lane for + * depth; this is the opposite — a lane added for price, capped so it cannot + * become depth. + * + * Measured against the endpoint on 2026-08-20, which is what made this + * routable at all: + * + * - It SHEDS rather than queues: HTTP 429 in 87-98ms with `retry-after: 1` + * and a structured `rate_limit_error` body. A refused session costs one + * fast round trip and diverts. + * - REQUEST-bound, not token-bound. 32k prompts shed at ~175-224k tok/s + * while 128k prompts sustained 498k tok/s untouched, so the served budget + * (~5-7 req/s) does not shrink as prompts grow. + * - Prefix caching holds at 99.4-99.9% across concurrency, against + * production Flash's 98.8% — which is why admission is per SESSION. A + * request-level share would make every request a cold prefill and consume + * the grant on prefill alone. + * + * A 429 here is TERMINAL for the session, unlike a divert off any other lane: + * the cascade re-pins it onward and it never comes back. That is deliberate. + * The lanes behind this one are sized to take our whole volume, so there is + * nothing to gain by retrying a rationed lane and one wasted round trip per + * turn to lose. + */ +export const DEEPSEEK_LUMINAL_PROVIDER_ROUTE = + 'deepseek/luminal' satisfies ProviderRouteId +export const DEEPSEEK_OFFICIAL_PROVIDER_ROUTE = + 'deepseek/official' satisfies ProviderRouteId +/** + * DeepSeek V4 Pro's entry lane as of 2026-08-21: the Cheaper Inference gateway. + * + * It replaces CrofAI at the front of Pro's cascade AND removes it from the + * cascade entirely — CrofAI's Pro lane was the dated `deepseek-v4-pro-0813` + * slug, stood down the same day as too dear and too inconsistent. So Pro's two + * lanes are now this and DeepSeek direct. + * + * Cheaper on every term than the direct lane behind it ($0.3045/$0.002538/$0.609 + * per M against $0.66/$0.022/$1.98 off-peak) and, unlike direct, FLAT — no peak + * card. That last point is most of the value: direct doubles for ten hours a + * day and those windows carry 26% of tokens against 46% of spend. + * + * The cache question this file's other entries keep raising was measured here + * rather than assumed, and the first answer was wrong. A 13-sample probe showed + * ~70% and read as disqualifying; at scale, warm, it is 100% over 60 sequential + * and 95% over 40 concurrent, holding on both upstreams this gateway routes + * between. Cold-start misses, not routing instability. Compare lanes at their + * MEASURED WARM hit rate — the rule that has now been arrived at three times in + * this file. + */ +export const DEEPSEEK_CHEAPER_INFERENCE_PROVIDER_ROUTE = + 'deepseek/cheaper-inference' satisfies ProviderRouteId +/** + * DeepSeek V4 Flash's LAST resort: the Infron lane, now tier 4 of four. + * + * DEMOTED FROM TIER 3 ON 2026-08-16. It was the cheapest route we had — + * measured live 2026-08-04 on a 45,008-token prompt at $0.069/M input and + * ~$0.0144/M cache read — and Infron then repriced its Alibaba Cloud Int. + * group to $0.212/M input, $0.021/M cache read and $0.636/M output. That is + * 3.1x, 1.4x and 4.6x, and it turns the cheapest lane into the dearest one. + * + * {@link DEEPSEEK_RUNINFRA_PROVIDER_ROUTE} is cheaper on input, cache reads and + * output alike, and a measured agent turn confirms the list prices rather than + * contradicting them: costed on one real 29-call buffbench turn, RunInfra came + * to $0.0444 against this lane's $0.0820 — 1.85x. So Infron ahead of RunInfra + * is wrong on both paper and practice. + * + * So it sits behind {@link DEEPSEEK_CROF_PROVIDER_ROUTE}, {@link + * DEEPSEEK_OFFICIAL_PROVIDER_ROUTE} and {@link + * DEEPSEEK_RUNINFRA_PROVIDER_ROUTE}, and it must never be the lane a session + * settles on. Its own failure mode is why nothing cheap may sit below it: a + * single aggregator account behind one Alibaba provider group, which when + * 4,997 sessions diverted onto it in one window returned 13,286 saturation + * 429s and then ran out of credits entirely, leaking the raw billing error to + * 1,086 users. This only works because the cascade RE-PINS on each hop — a + * session that finds Infron saturated does not pay a doomed Infron attempt on + * every later turn. + * + * The `makora` in the name is historical (that upstream went offline in + * 2026-07). The id says only *that* a session is on this lane, never which + * upstream serves it — that is INFRON_PROVIDER_ORDER, keyed by model — so + * repointing the upstream also moves sessions already pinned here. Renaming the + * value itself would need a migration; it is persisted in + * `free_session.provider_route` and read back unvalidated. + */ +export const DEEPSEEK_INFRON_MAKORA_PROVIDER_ROUTE = + 'infron/makora' satisfies ProviderRouteId +/** + * DeepSeek V4 Flash's SECOND backup: the RunInfra lane, tier 3 of four. + * + * Added because the three lanes ahead of it have each failed in the one way a + * cascade cannot absorb — by running out of money. DeepSeek shed peak load, + * Infron's aggregator account went dry and leaked its billing error to 1,086 + * users, and CrofAI returned 401 "Not Enough Credits" off its own prepaid + * balance. Three lanes whose failures are that correlated with a divert storm + * are, on the worst day, one lane. RunInfra is a fourth independent account + * and balance behind them; that independence, not its price, is the reason it + * exists. + * + * List prices are on {@link DEEPSEEK_CROF_PROVIDER_ROUTE}. It is dearer than + * CrofAI on all three terms, so it can never sit above it; it sits above Infron + * because Infron repriced on 2026-08-16 into the dearest lane we have. + * + * VALIDATED ON A REAL AGENT TURN rather than a price table, which is unusual + * for this file and worth the words. One buffbench task was run end to end + * through this lane against a local server: 29 model calls, 1,202,712 input + * tokens (82.3% cached), 25,287 output tokens. Costing that exact token + * profile against each lane's card: + * + * CrofAI $0.0338 + * RunInfra $0.0444 + * DeepSeek off-peak $0.0705 + * Infron $0.0820 + * DeepSeek peak $0.1409 + * + * So RunInfra ahead of Infron is right by 1.85x on measured traffic, which is + * what this lane's placement rests on. Note the DeepSeek rows sit BELOW + * RunInfra on this workload since the 2026-08-16 repricing — that is a question + * about the entry lane, not about this one, and it belongs to + * docs/freebuff-deepseek-provider-cutover.md rather than here. + * + * The cache has a COLD START. Over that turn it served 82.3% of input tokens + * from cache, but the aggregate hides the shape: the first few calls missed + * outright despite sharing a large prefix, then it held at 98-99% for the + * remaining ~25 calls. A synthetic probe showed the same thing (three misses, + * then 99.1% hits), with ~600ms latency on a hit against ~1,500ms on a miss. + * `prompt_cache_key` did not pin routing. The practical consequence is that a + * long session gets the list rate and a very short one pays closer to fresh + * input — acceptable for a lane only sustained failure reaches. + * + * Two further constraints, both measured rather than published: its hard output + * ceiling is 32,768 tokens (see RUNINFRA_DEEPSEEK_V4_FLASH_MAX_TOKENS — below + * the 48,000 budget the product considers safe, so expect more empty + * length-capped answers here), and it returns no `cost`, so its price table + * bills every request rather than catching a rare gap. + * + * What makes it a better tier-3 than the OpenRouter lane it replaced in the + * cascade: that lane priced cache reads at $0.0176/M and was reached 1,205 + * times in 24h, which is how DeepSeek's daily bill went from $9k to $39.7k. + * RunInfra is 1.76x under it on exactly that term. Depth still costs something + * — 3.3x CrofAI's cache read — which is why it leaves NO RESUMABLE PIN (see + * `asDeepSeekLane`): a session that touches it starts its next turn from the + * primary again rather than settling here for the rest of its hour. + * + * Like its peers the id names the LANE, not the upstream, and it is persisted + * in `free_session.provider_route` and read back unvalidated — so renaming the + * value would need a migration, while repointing what serves it would not. + */ +export const DEEPSEEK_RUNINFRA_PROVIDER_ROUTE = + 'deepseek/runinfra' satisfies ProviderRouteId +/** + * RETIRED as a DeepSeek lane on 2026-08-11. Kept as a recognized id because it + * is persisted in `free_session.provider_route` and read back unvalidated — + * sessions still carrying the pin must not crash; they simply start from the + * primary again, which is the right answer for a lane that no longer exists. + * + * Why it went: it prices cache reads at $0.0176/M against CrofAI's $0.0030, + * and an agent turn re-sends its whole prefix every step, so ~98% of the + * tokens land on exactly that term. Being "last resort" did not bound the + * damage — the pin only moved forward, so one transient 429 on the lane above + * parked a session here for the rest of its hour. It was reached 1,205 times + * in 24h, more often than the lane ahead of it, and DeepSeek's daily bill went + * from $9k to $39.7k over four days while volume rose only 41%. Depth that + * costs 5.9x on the dominant token class is not depth. + * + * The replacement for that depth is retries: the two cheap lanes are attempted + * three times each before anything diverts. + * + * (Historical, for the MiMo lane which still uses OpenRouter:) + * DeepSeek V4 Flash's LAST resort: the OpenRouter lane, tier 4 of four. + * + * Reached when DeepSeek direct and then {@link + * DEEPSEEK_INFRON_MAKORA_PROVIDER_ROUTE} have both failed retryably. Dearer per + * token than Infron but backed by many independent upstreams and a balance that + * is not one account's, which is exactly what a divert storm needs — see the + * Infron route's doc for why the cheap lane cannot be the last one. + * + * Like its MiMo peer it names the LANE, not the upstream — that is {@link + * DEEPSEEK_OPENROUTER_UPSTREAM_ORDER} below, so repointing the order also moves + * every session already pinned here, with no migration. + */ +export const DEEPSEEK_OPENROUTER_PROVIDER_ROUTE = + 'deepseek/openrouter' satisfies ProviderRouteId +/** + * The upstreams that serve {@link DEEPSEEK_OPENROUTER_PROVIDER_ROUTE}, + * preferred first. + * + * Chosen as *cheapest that still preserves the prompt cache*, which for an + * agent workload are not the same axis. Cache-read price is what actually + * drives this bill — a coding turn re-sends a long prefix every step, so most + * input tokens are cache reads — and the OpenRouter catalog splits cleanly on + * it (checked live 2026-08-04, per M): + * + * streamlake/fp8 $0.0881 in $0.0176 cache $0.1761 out fp8 384k max out + * baidu/fp8 $0.0882 in $0.0176 cache $0.1764 out fp8 131k max out + * gmicloud/fp8 $0.0938 in $0.0188 cache $0.1876 out fp8 no stated cap + * ── everything below is >=1.5x the cache-read price ── + * most of the tail $0.14 in $0.0280 cache $0.2800 out + * parasail/coreweave/phala $0.0700 cache (4x) + * morph, mancer/fp4 NO cache read at all + * + * So the three cheapest on input are also the three cheapest on cache read; + * there is no tradeoff to make here, which is why the list is short. + * + * DELIBERATELY NOT `deepseek` (OpenRouter's DeepSeek-first-party endpoint). + * It had by far the best cache read of any entry here, and was tempting for + * that alone, but it is the same upstream whose failure triggers this fallback, + * reached through a middleman — pointing the lane there would divert an outage + * onto itself. The Infron lane can use the 0731 model without this problem + * because it pins independent Alibaba upstreams. (The price argument has since + * evaporated anyway: DeepSeek's 2026-08-16 repricing put first-party cache + * reads at $0.0070/M off-peak and $0.0140/M at peak, at or above the tail + * quoted above. The routing reason is the one that still stands.) + * + * `deepinfra/fp4` is skipped despite sitting third on price: fp4 quantization, + * and a 65,536-token output cap that would truncate long agent turns. fp8 is + * the floor for this lane. + * + * THREE ENTRIES, and it must never go to one — see the identical warning on + * {@link mimoOpenRouterProvider}. A pinned session has no health check and no + * un-pin path, so a one-deep lane turns a single upstream blip into a wedged + * session. That is not hypothetical here: this model's previous fallback was + * pinned one-deep to `makora` and took out 1,160 requests across 191 users for + * ~32h when it went offline (2026-07-26/27, #1045). + * + * Caveat on the third entry: `gmicloud/fp8` does not list `stop` in its + * supported parameters. Without `require_parameters` OpenRouter drops the + * unsupported field rather than refusing to route, so a turn served there does + * not honor the global stop sequence. Accepted for depth — it only serves when + * both fp8 upstreams above it are unavailable — but do not promote it. + */ +export const DEEPSEEK_OPENROUTER_UPSTREAM_ORDER = [ + 'streamlake/fp8', + 'baidu/fp8', + 'gmicloud/fp8', +] as const + +/** + * The OpenRouter output cap this lane requests. + * + * Matches `streamlake/fp8`'s 384,000-token ceiling — the first upstream in the + * order — so a caller's explicit budget can never make the preferred endpoint + * ineligible. Slightly under DeepSeek direct's 393,216, same as the Infron lane + * this replaces: keep fallback requests inside the contract of the route that + * will actually serve them. + */ +export const DEEPSEEK_OPENROUTER_MAX_TOKENS = 384_000 + +/** + * Fresh OpenRouter `provider` block for the DeepSeek V4 Flash lane. + * + * Returns a NEW object with a NEW array every call — these arrays get aliased + * into an outgoing request body, and one downstream mutation would corrupt + * routing process-wide (the bug `INFRON_PROVIDER_ORDER` was made + * copy-on-assignment for in #1045). + * + * `allow_fallbacks: false` is what preserves the prompt cache: it holds the + * session to this order instead of letting OpenRouter spread turns across + * twenty endpoints that each keep their own cache. Cache fragmentation is the + * expensive failure mode, not a slightly dearer per-token rate — measured on + * the MiMo lane, a scattered fallback averaged 27.6k cache_read against ~150k + * prompts, paying a full cold prefill per divert. + */ +export function deepseekOpenRouterProvider(): Record { + return { + order: [...DEEPSEEK_OPENROUTER_UPSTREAM_ORDER], + allow_fallbacks: false, + } +} diff --git a/common/src/testing/mocks/filesystem.ts b/common/src/testing/mocks/filesystem.ts index 6c9703622e..bf2af29bde 100644 --- a/common/src/testing/mocks/filesystem.ts +++ b/common/src/testing/mocks/filesystem.ts @@ -2,7 +2,7 @@ import { mock } from 'bun:test' import type { CodebuffFileSystem } from '../../types/filesystem' import type { Mock } from 'bun:test' -import type { PathLike , Stats } from 'node:fs' +import type { PathLike, Stats } from 'node:fs' export interface CreateMockFsOptions { files?: Record @@ -14,6 +14,7 @@ export interface CreateMockFsOptions { path: string, options?: { recursive?: boolean }, ) => Promise + realpathImpl?: (path: string) => Promise statImpl?: (path: string) => Promise } @@ -31,6 +32,7 @@ export interface MockFsWithMocks { options?: { recursive?: boolean }, ) => Promise > + realpath: Mock<(path: PathLike) => Promise> stat: Mock<(path: PathLike) => Promise> } @@ -43,6 +45,7 @@ export function createMockFs(options: CreateMockFsOptions = {}): MockFs { readdirImpl, writeFileImpl, mkdirImpl, + realpathImpl, statImpl, } = options @@ -79,6 +82,20 @@ export function createMockFs(options: CreateMockFsOptions = {}): MockFs { return undefined } + const defaultRealpath = async (path: PathLike): Promise => { + const pathStr = String(path) + const isKnownPath = + pathStr in writtenFiles || + pathStr in directories || + createdDirs.has(pathStr) + + if (!isKnownPath) { + throw new Error(`Path not found: ${pathStr}`) + } + + return pathStr + } + const defaultStat = async (path: PathLike): Promise => { const pathStr = String(path) const isFile = pathStr in writtenFiles @@ -134,6 +151,10 @@ export function createMockFs(options: CreateMockFsOptions = {}): MockFs { mkdirImpl(String(path), opts) : defaultMkdir + const realpathFn = realpathImpl + ? async (path: PathLike) => realpathImpl(String(path)) + : defaultRealpath + const statFn = statImpl ? async (path: PathLike) => statImpl(String(path)) : defaultStat @@ -143,6 +164,7 @@ export function createMockFs(options: CreateMockFsOptions = {}): MockFs { readdir: mock(readdirFn), writeFile: mock(writeFileFn), mkdir: mock(mkdirFn), + realpath: mock(realpathFn), stat: mock(statFn), } as unknown as MockFs } @@ -153,6 +175,7 @@ export function restoreMockFs(mockFs: MockFs): void { mocks.readdir.mockRestore() mocks.writeFile.mockRestore() mocks.mkdir.mockRestore() + mocks.realpath.mockRestore() mocks.stat.mockRestore() } @@ -162,5 +185,6 @@ export function clearMockFs(mockFs: MockFs): void { mocks.readdir.mockClear() mocks.writeFile.mockClear() mocks.mkdir.mockClear() + mocks.realpath.mockClear() mocks.stat.mockClear() } diff --git a/common/src/types/filesystem.ts b/common/src/types/filesystem.ts index 6fa64e1168..4506b5c87e 100644 --- a/common/src/types/filesystem.ts +++ b/common/src/types/filesystem.ts @@ -6,5 +6,11 @@ import type fs from 'fs' */ export type CodebuffFileSystem = Pick< typeof fs.promises, - 'mkdir' | 'readdir' | 'readFile' | 'stat' | 'unlink' | 'writeFile' + | 'mkdir' + | 'readdir' + | 'readFile' + | 'realpath' + | 'stat' + | 'unlink' + | 'writeFile' > diff --git a/common/src/types/freebuff-session.ts b/common/src/types/freebuff-session.ts index a9c751279d..aaefdbb5e0 100644 --- a/common/src/types/freebuff-session.ts +++ b/common/src/types/freebuff-session.ts @@ -1,5 +1,5 @@ import type { FreebuffAccessTier } from '../constants/freebuff-models' -import type { FreebuffStandingInfo } from '../constants/freebuff-standing' +import type { FreebuffStandingInfo } from '../constants/freebuff-trust' /** * Wire-level shapes returned by `/api/v1/freebuff/session`. Source of truth @@ -139,41 +139,6 @@ export interface FreebuffSubscriptionUsage { * Sent only to callers in the rollout audience, so its absence means "this * account has no subscriptions surface" rather than "no data". */ -/** One window of a Freebucks allowance, as the client should render it. */ -export interface FreebuffFreebucksWindow { - /** Total allowance for the window: free grant + whatever the plan adds. */ - limit: number - /** Freebucks already spent inside the window. */ - spent: number - /** `limit - spent`, floored at zero so a lowered allowance reads as 0. */ - remaining: number - /** ISO instant this window reopens. */ - resetAt: string -} - -/** - * The caller's Freebucks position, present on every authenticated session - * response. Distinct from Trust: Freebucks are a granted, expiring budget that - * buys premium sessions, Trust is earned standing that buys Levels. - */ -export interface FreebuffFreebucksInfo { - /** - * Spendable right now — the MINIMUM remaining across the three windows, which - * is the only number that answers "can I start a session". Rendering the - * daily figure alone would promise Freebucks the weekly cap will refuse. - */ - balance: number - daily: FreebuffFreebucksWindow - weekly: FreebuffFreebucksWindow - monthly: FreebuffFreebucksWindow - /** Which window is currently binding — the one `balance` came from. */ - bindingWindow: 'daily' | 'weekly' | 'monthly' - /** Freebucks the caller's plan adds on top of the free grant, if subscribed. */ - planDaily?: number - /** Session price per model id. Only models Freebucks can buy appear here. */ - prices: Record -} - export interface FreebuffSubscriptionInfo { /** The caller's tier id, or null when they have no live subscription. */ tierId: string | null @@ -380,14 +345,6 @@ export const getRateLimitsByModel = ( * (none, active, ended). Loose parameter type for the same reason as * `getRateLimitsByModel`. Undefined from a server that predates * subscriptions, so callers render nothing rather than an empty upsell. */ -/** The caller's Freebucks block, wherever it rides the response. */ -export const getFreebucksInfo = ( - session: { status: string } | null | undefined, -): FreebuffFreebucksInfo | undefined => - session && 'freebucks' in session - ? (session as { freebucks?: FreebuffFreebucksInfo }).freebucks - : undefined - export const getSubscriptionInfo = ( session: { status: string } | null | undefined, ): FreebuffSubscriptionInfo | undefined => @@ -560,10 +517,6 @@ export type FreebuffSessionServerResponse = ( * catalog is empty or the server predates subscriptions. */ subscription?: FreebuffSubscriptionInfo - /** Spendable Freebucks and the per-model session prices. Rides - * every state for the same reason `subscription` does: the - * balance is shown in the picker, mid-session and after it. */ - freebucks?: FreebuffFreebucksInfo } & FreebuffLimitedModeReason) | ({ status: 'active' @@ -582,10 +535,6 @@ export type FreebuffSessionServerResponse = ( /** Subscription offers and state, so an in-session picker can still * render "subscribed" badges and an upgrade CTA. */ subscription?: FreebuffSubscriptionInfo - /** Spendable Freebucks and the per-model session prices. Rides - * every state for the same reason `subscription` does: the - * balance is shown in the picker, mid-session and after it. */ - freebucks?: FreebuffFreebucksInfo } & FreebuffLimitedModeReason) | ({ /** Session is over. While `instanceId` is present we're inside the @@ -613,10 +562,6 @@ export type FreebuffSessionServerResponse = ( /** Carried like `rateLimitsByModel`: the post-session banner and picker * keep the plan rings without a round-trip. */ subscription?: FreebuffSubscriptionInfo - /** Spendable Freebucks and the per-model session prices. Rides - * every state for the same reason `subscription` does: the - * balance is shown in the picker, mid-session and after it. */ - freebucks?: FreebuffFreebucksInfo } & FreebuffLimitedModeReason) | { /** Another CLI on the same account rotated our instance id. Polling diff --git a/common/src/util/account-deletion-proof.test.ts b/common/src/util/account-deletion-proof.test.ts deleted file mode 100644 index 7072353a98..0000000000 --- a/common/src/util/account-deletion-proof.test.ts +++ /dev/null @@ -1,48 +0,0 @@ -import { describe, expect, test } from 'bun:test' - -import { - createAccountDeletionProof, - verifyAccountDeletionProof, -} from './account-deletion-proof' - -const SECRET = 'account-deletion-proof-secret-at-least-32-chars' - -describe('account deletion proof', () => { - test('round-trips only for the bound subject and code', async () => { - const proof = await createAccountDeletionProof(SECRET, 'user-1', '123456') - - expect( - await verifyAccountDeletionProof(proof, SECRET, 'user-1', '123456'), - ).toBe(true) - expect( - await verifyAccountDeletionProof(proof, SECRET, 'user-2', '123456'), - ).toBe(false) - expect( - await verifyAccountDeletionProof(proof, SECRET, 'user-1', '654321'), - ).toBe(false) - }) - - test('length-prefixes fields so concatenation cannot collide', async () => { - const first = await createAccountDeletionProof(SECRET, 'ab', 'c') - const second = await createAccountDeletionProof(SECRET, 'a', 'bc') - expect(first).not.toBe(second) - }) - - test('refuses a weak shared secret', async () => { - await expect( - createAccountDeletionProof('short', 'user-1', '123456'), - ).rejects.toThrow('at least 32 characters') - }) - - test('rejects malformed and truncated proofs', async () => { - const proof = await createAccountDeletionProof(SECRET, 'user-1', '123456') - expect( - await verifyAccountDeletionProof( - proof.slice(0, -1), - SECRET, - 'user-1', - '123456', - ), - ).toBe(false) - }) -}) diff --git a/common/src/util/account-deletion-proof.ts b/common/src/util/account-deletion-proof.ts deleted file mode 100644 index 6eef9c2e65..0000000000 --- a/common/src/util/account-deletion-proof.ts +++ /dev/null @@ -1,81 +0,0 @@ -const PROOF_VERSION = 'v1' - -const encoder = new TextEncoder() - -function proofPayload(identitySubject: string, code: string): ArrayBuffer { - const subject = identitySubject.trim() - const normalizedCode = code.trim() - const encoded = encoder.encode( - `${PROOF_VERSION}:${subject.length}:${subject}:${normalizedCode.length}:${normalizedCode}`, - ) - // `Uint8Array#buffer` is `ArrayBufferLike` in this repo's older TS lib and - // therefore also admits SharedArrayBuffer, which Web Crypto rejects here. - const payload = new ArrayBuffer(encoded.byteLength) - new Uint8Array(payload).set(encoded) - return payload -} - -function hex(bytes: ArrayBuffer): string { - return Array.from(new Uint8Array(bytes)) - .map((byte) => byte.toString(16).padStart(2, '0')) - .join('') -} - -async function signature( - secret: string, - identitySubject: string, - code: string, -): Promise { - if (secret.length < 32) { - throw new Error( - 'account deletion proof secret must be at least 32 characters', - ) - } - const key = await crypto.subtle.importKey( - 'raw', - encoder.encode(secret), - { name: 'HMAC', hash: 'SHA-256' }, - false, - ['sign'], - ) - return hex( - await crypto.subtle.sign('HMAC', key, proofPayload(identitySubject, code)), - ) -} - -/** - * Server proof that Adbuff deletion completed before a Convex account purge. - * - * The proof is deliberately bound to both the authenticated subject and the - * mailed deletion code. It is never returned to the browser: the Next route - * mints it only after Adbuff reports an active suppression fence and presents - * it on each bounded Convex purge pass. - */ -export async function createAccountDeletionProof( - secret: string, - identitySubject: string, - code: string, -): Promise { - return `${PROOF_VERSION}.${await signature(secret, identitySubject, code)}` -} - -/** Constant-time comparison over fixed-size HMAC hex strings. */ -export async function verifyAccountDeletionProof( - proof: string, - secret: string, - identitySubject: string, - code: string, -): Promise { - const expected = await createAccountDeletionProof( - secret, - identitySubject, - code, - ) - if (proof.length !== expected.length) return false - - let difference = 0 - for (let index = 0; index < expected.length; index += 1) { - difference |= proof.charCodeAt(index) ^ expected.charCodeAt(index) - } - return difference === 0 -} diff --git a/common/src/util/freebuff-model-availability.ts b/common/src/util/freebuff-model-availability.ts index 5c14f2d92c..faedc56827 100644 --- a/common/src/util/freebuff-model-availability.ts +++ b/common/src/util/freebuff-model-availability.ts @@ -54,8 +54,8 @@ export const FREEBUFF_PAUSED_MODEL_NOTICE = * * 2026-08-28: GLM 5.3 Flash is now UNMETERED, joining MiMo and DeepSeek V4 * Flash. The 2-a-day cap came off on 08-27 when its measurement window closed; - * a day of production spend then settled the cost question outright — it is - * the cheapest row we serve per message, 4.6x under MiMo and 8.9x under V4 + * a day of production spend then settled the cost question outright — it bills + * $0.000249/msg, the cheapest row we serve, 4.6x under MiMo and 8.9x under V4 * Flash, both of which already ran uncapped. Keeping a ceiling on the cheapest * model while the dearer ones had none inverted the reason ceilings exist. * diff --git a/freebuff/cli/release/package.json b/freebuff/cli/release/package.json index 008a1cdb07..8592c848fc 100644 --- a/freebuff/cli/release/package.json +++ b/freebuff/cli/release/package.json @@ -1,6 +1,6 @@ { "name": "freebuff", - "version": "0.0.163", + "version": "0.0.162", "description": "The world's strongest free coding agent", "license": "MIT", "bin": { diff --git a/sdk/src/__tests__/list-directory.test.ts b/sdk/src/__tests__/list-directory.test.ts new file mode 100644 index 0000000000..5637a75b4c --- /dev/null +++ b/sdk/src/__tests__/list-directory.test.ts @@ -0,0 +1,224 @@ +import { describe, expect, it, mock } from 'bun:test' + +import path from 'path' + +import { listDirectory } from '../tools/list-directory' + +import type { CodebuffFileSystem } from '@codebuff/common/types/filesystem' +import type { Dirent, PathLike } from 'node:fs' + +const PROJECT_ROOT = path.resolve('workspace', 'project') + +function createFs(realpaths: Record) { + const readdir = mock(async (_path: PathLike) => { + return [ + { + name: 'index.ts', + isDirectory: () => false, + isFile: () => true, + }, + ] as Dirent[] + }) + + const fs = { + realpath: mock(async (path: PathLike) => { + const pathString = String(path) + return realpaths[pathString] ?? pathString + }), + readdir, + } as unknown as CodebuffFileSystem + + return { fs, readdir } +} + +describe('listDirectory', () => { + it('allows listing the project root itself', async () => { + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + }) + + const result = await listDirectory({ + directoryPath: '.', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result[0]).toEqual({ + type: 'json', + value: { + files: ['index.ts'], + directories: [], + path: '.', + }, + }) + expect(readdir).toHaveBeenCalledWith(PROJECT_ROOT, { + withFileTypes: true, + }) + }) + + it('lists a directory inside the project and preserves the requested path', async () => { + const childPath = path.join(PROJECT_ROOT, 'src') + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + [childPath]: childPath, + }) + + const result = await listDirectory({ + directoryPath: 'src', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + files: ['index.ts'], + directories: [], + path: 'src', + }, + }, + ]) + expect(readdir).toHaveBeenCalledWith(childPath, { + withFileTypes: true, + }) + }) + + it('returns the normal list error when the requested directory is missing', async () => { + const missingPath = path.join(PROJECT_ROOT, 'missing') + const readdir = mock(async (_path: PathLike) => [] as Dirent[]) + const fs = { + realpath: mock(async (requestedPath: PathLike) => { + const requestedPathString = String(requestedPath) + if (requestedPathString === missingPath) { + throw new Error( + `ENOENT: no such file or directory, realpath '${missingPath}'`, + ) + } + return requestedPathString + }), + readdir, + } as unknown as CodebuffFileSystem + + const result = await listDirectory({ + directoryPath: 'missing', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + errorMessage: `Failed to list directory: ENOENT: no such file or directory, realpath '${missingPath}'`, + }, + }, + ]) + expect(readdir).not.toHaveBeenCalled() + }) + + it('rejects sibling paths that only share the project prefix', async () => { + const siblingPath = path.resolve(PROJECT_ROOT, '..', 'project-evil') + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + [siblingPath]: siblingPath, + }) + + const result = await listDirectory({ + directoryPath: '../project-evil', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + errorMessage: + "Invalid path: Path '../project-evil' is outside the project directory.", + }, + }, + ]) + expect(readdir).not.toHaveBeenCalled() + }) + + it('rejects the project parent directory', async () => { + const parentPath = path.dirname(PROJECT_ROOT) + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + [parentPath]: parentPath, + }) + + const result = await listDirectory({ + directoryPath: '..', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + errorMessage: + "Invalid path: Path '..' is outside the project directory.", + }, + }, + ]) + expect(readdir).not.toHaveBeenCalled() + }) + + it('rejects directories that escape through a symlink', async () => { + const symlinkPath = path.join(PROJECT_ROOT, 'link') + const outsidePath = path.resolve(PROJECT_ROOT, '..', 'outside') + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + [symlinkPath]: outsidePath, + }) + + const result = await listDirectory({ + directoryPath: 'link', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + errorMessage: + "Invalid path: Path 'link' is outside the project directory.", + }, + }, + ]) + expect(readdir).not.toHaveBeenCalled() + }) + + it('allows a symlink that resolves inside the project', async () => { + const symlinkPath = path.join(PROJECT_ROOT, 'link') + const realTarget = path.join(PROJECT_ROOT, 'src') + const { fs, readdir } = createFs({ + [PROJECT_ROOT]: PROJECT_ROOT, + [symlinkPath]: realTarget, + }) + + const result = await listDirectory({ + directoryPath: 'link', + projectPath: PROJECT_ROOT, + fs, + }) + + expect(result).toEqual([ + { + type: 'json', + value: { + files: ['index.ts'], + directories: [], + path: 'link', + }, + }, + ]) + expect(readdir).toHaveBeenCalledWith(realTarget, { + withFileTypes: true, + }) + }) +}) diff --git a/sdk/src/tools/list-directory.ts b/sdk/src/tools/list-directory.ts index 3bf66fa968..53f12b660e 100644 --- a/sdk/src/tools/list-directory.ts +++ b/sdk/src/tools/list-directory.ts @@ -2,6 +2,7 @@ import * as path from 'path' import type { CodebuffToolOutput } from '@codebuff/common/tools/list' import type { CodebuffFileSystem } from '@codebuff/common/types/filesystem' +import { isPathInside } from '@codebuff/common/util/path' export async function listDirectory(params: { directoryPath: string @@ -11,9 +12,23 @@ export async function listDirectory(params: { const { directoryPath, projectPath, fs } = params try { - const resolvedPath = path.resolve(projectPath, directoryPath) + const projectRoot = path.resolve(projectPath) + const resolvedPath = path.resolve(projectRoot, directoryPath) + const realProjectRoot = await fs.realpath(projectRoot) + const realResolvedPath = await fs.realpath(resolvedPath) - const entries = await fs.readdir(resolvedPath, { + if (!isPathInside(realProjectRoot, realResolvedPath)) { + return [ + { + type: 'json', + value: { + errorMessage: `Invalid path: Path '${directoryPath}' is outside the project directory.`, + }, + }, + ] + } + + const entries = await fs.readdir(realResolvedPath, { withFileTypes: true, })