diff --git a/docs/public/robots.txt b/docs/public/robots.txt index 340a0f4..597a269 100644 --- a/docs/public/robots.txt +++ b/docs/public/robots.txt @@ -1,3 +1,41 @@ +# neuron-js docs — agentic access policy (docs/public/robots.txt) +# Served at https://sebasoft.github.io/neuron-js/robots.txt +# +# Decision (SebaSOFT, 2026-09): training crawlers are ALLOWED. All AI bots +# listed below are explicitly welcomed so policy scanners read a clear +# allow instead of inferring from a wildcard. + +# --- Citation-eligible search & answer bots --- +User-agent: OAI-SearchBot +Allow: / + +User-agent: PerplexityBot +Allow: / + +User-agent: Claude-SearchBot +Allow: / + +# --- User-triggered fetchers (agent browsing on a user's behalf) --- +User-agent: ChatGPT-User +Allow: / + +User-agent: Perplexity-User +Allow: / + +User-agent: Claude-User +Allow: / + +# --- Training crawlers (decision: ALLOWED) --- +User-agent: GPTBot +Allow: / + +User-agent: ClaudeBot +Allow: / + +User-agent: Google-Extended +Allow: / + +# --- Everything else (browsers, classic SEO crawlers, unlisted bots) --- User-agent: * Allow: / diff --git a/tests/contracts/robots-granular.test.ts b/tests/contracts/robots-granular.test.ts new file mode 100644 index 0000000..6e8a948 --- /dev/null +++ b/tests/contracts/robots-granular.test.ts @@ -0,0 +1,40 @@ +// NJS-SEO-6 follow-up: contract test for the granular robots.txt agentic +// access policy. Pins the SebaSOFT decision (2026-09): training crawlers +// are ALLOWED on the neuron-js documentation site. +import { readFileSync } from "node:fs"; +import { describe, expect, it } from "vitest"; + +const robots = readFileSync("docs/public/robots.txt", "utf8"); + +describe("granular robots.txt agentic policy", () => { + it("explicitly allows citation-eligible AI bots", () => { + for (const bot of ["OAI-SearchBot", "PerplexityBot", "Claude-SearchBot"]) { + expect(robots).toContain(`User-agent: ${bot}`); + } + }); + + it("explicitly allows user-triggered fetchers", () => { + for (const bot of ["ChatGPT-User", "Perplexity-User", "Claude-User"]) { + expect(robots).toContain(`User-agent: ${bot}`); + } + }); + + it("pins the training-crawler decision: ALLOWED", () => { + for (const bot of ["GPTBot", "ClaudeBot", "Google-Extended"]) { + expect(robots).toContain(`User-agent: ${bot}`); + // every explicit rule in this file is an Allow — fail if someone flips policy + const section = robots.split(`User-agent: ${bot}`)[1]?.split("User-agent:")[0] ?? ""; + expect(section).toContain("Allow: /"); + expect(section).not.toContain("Disallow"); + } + expect(robots).toContain("training crawlers are ALLOWED"); + }); + + it("keeps the wildcard open and the sitemap for all agents", () => { + expect(robots).toContain("User-agent: *"); + expect(robots).not.toContain("Disallow"); + expect(robots).toContain( + "Sitemap: https://sebasoft.github.io/neuron-js/sitemap.xml", + ); + }); +});