diff --git a/.github/workflows/site.yml b/.github/workflows/site.yml index d6de84e..d679816 100644 --- a/.github/workflows/site.yml +++ b/.github/workflows/site.yml @@ -63,6 +63,12 @@ jobs: node --version node scripts/check_floor_parity.mjs _site/example-run.json + # search.js against cases that say what a reader should find first, and + # every search-index.json entry against the page and id it names. Also + # the files the page scripts fetch, which the link check cannot see. + - name: Search ranks as it should, and every index entry resolves + run: node scripts/check_search.mjs _site + # The exact bytes the deploy job publishes. On a pull request this is # also a downloadable preview of the site. - name: Keep the assembled site diff --git a/CHANGELOG.md b/CHANGELOG.md index 35ea855..bac0c8c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -43,6 +43,12 @@ different event from one that moved because it was wrong. time, so it works with scripts off; `site/site.js` adds only the conveniences and a light, dark, or automatic theme that is remembered between visits. Form borders and the histogram's bars now meet 3:1 contrast in both themes. +- Search across the explainer and every doc, from the header or with `/` or + Ctrl+K. The index (`search-index.json`) is written at build time from the + rendered sections and fetched from the site only when search opens; the + ranking (`site/search.js`) runs in the browser with no library. + `scripts/check_search.mjs` holds the ranking to its cases and every index + entry to a page and id that exist. ### Changed diff --git a/scripts/build_site.py b/scripts/build_site.py index 694fdc1..2c3903a 100644 --- a/scripts/build_site.py +++ b/scripts/build_site.py @@ -20,6 +20,11 @@ came from. Nothing is written by hand and nothing is fetched at runtime, so the pages cannot drift from ``main``: a deploy re-renders them. +``search-index.json`` + Every section of the explainer and the docs as plain text, for the site's + search (``site/search.js``). A reader's browser fetches it only when they + open search, from the site itself. + The example is regenerated from the **mock** adapter and nothing else. The command is read from the report, and anything other than ``--adapter mock`` is refused, so a vendor run can never be published through this path. The run @@ -596,6 +601,7 @@ def social_meta(title: str, description: str, url: str) -> str: Source on GitHub, Apache-2.0 licensed. No analytics, no trackers, no external requests.

+ @@ -703,6 +709,70 @@ def explainer_page(source: str) -> str: return source.replace(HEADER_SLOT, site_header("", "calculator")) +#: The explainer's name in search results. +EXPLAINER_LABEL = "The ECE floor" +#: Longest body text kept per search entry. Enough to match on and quote from; +#: the index is fetched only when a reader opens search. +SEARCH_TEXT_LIMIT = 4000 + +_SECTION = re.compile(r'
]*>(?P.*?)
', re.S) +_H2 = re.compile(r"]*>(?P.*?)", re.S) +_LEDE = re.compile(r'

(?P.*?)

', re.S) +_H1 = re.compile(r"]*>(?P.*?)", re.S) + + +def _plain(fragment: str) -> str: + """HTML to the words a reader sees: tags dropped, entities decoded.""" + text = re.sub(r"\s+", " ", html.unescape(re.sub(r"<[^>]+>", " ", fragment))) + # A tag becomes a space, so text that ran up to one ("text.") gains + # a space before its punctuation; take it back out. + return re.sub(r" ([.,;:!?)])", r"\1", text).strip() + + +def _entry(label: str, heading: str, url: str, text: str) -> dict[str, str]: + return {"t": label, "h": heading, "u": url, "x": text[:SEARCH_TEXT_LIMIT]} + + +def search_index(rendered: dict[str, Rendered], explainer: str) -> list[dict[str, str]]: + """One entry per section, for ``site/search.js``. + + ``t`` is the page's name, ``h`` the section heading, ``u`` its address + relative to the site root, and ``x`` its plain text. A doc's h1, and any + text before it, is the entry for the page itself, with no fragment. + ``explainer`` is ``site/index.html`` as written. + """ + entries: list[dict[str, str]] = [] + + title = _H1.search(explainer) + lede = _LEDE.search(explainer) + if title is None or lede is None: + raise BuildError("site/index.html has no

or lede paragraph to index") + entries.append(_entry(EXPLAINER_LABEL, _plain(title["text"]), "./", _plain(lede["text"]))) + for match in _SECTION.finditer(explainer): + heading = _H2.search(match["body"]) + if heading is None: + raise BuildError(f"site/index.html section #{match['id']} has no

") + body = _plain(match["body"][heading.end() :]) + entries.append(_entry(EXPLAINER_LABEL, _plain(heading["text"]), f"./#{match['id']}", body)) + + for doc in DOCS: + page = f"docs/{doc.slug}.html" + start = len(entries) + top: list[str] = [] + top_heading = doc.label + for section in rendered[doc.slug]["sections"]: + if section["id"] is None or section["level"] == 1: + if section["level"] == 1: + top_heading = section["heading"] + top.append(section["text"]) + continue + url = f"{page}#{section['id']}" + entries.append(_entry(doc.label, section["heading"], url, section["text"])) + # The page itself leads its sections. + entries.insert(start, _entry(doc.label, top_heading, page, " ".join(top))) + return entries + + def main() -> int: parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) parser.add_argument("--out", type=Path, default=ROOT / "_site", help="output directory") @@ -718,7 +788,9 @@ def main() -> int: example = build_example() prov = provenance() rendered = render_docs(prov) - explainer = explainer_page((SITE / "index.html").read_text(encoding="utf-8")) + source = (SITE / "index.html").read_text(encoding="utf-8") + explainer = explainer_page(source) + index = search_index(rendered, source) except BuildError as problem: print(f"site build refused: {problem}", file=sys.stderr) return 1 @@ -730,6 +802,9 @@ def main() -> int: shutil.copytree(SITE, out, ignore=shutil.ignore_patterns("vendor")) (out / "example-run.json").write_text(json.dumps(example, indent=1) + "\n", encoding="utf-8") (out / "index.html").write_text(explainer, encoding="utf-8") + (out / "search-index.json").write_text( + json.dumps(index, ensure_ascii=False, separators=(",", ":")) + "\n", encoding="utf-8" + ) write_docs(out, prov, rendered) write_404(out) print(f"site assembled in {out}") diff --git a/scripts/check_search.mjs b/scripts/check_search.mjs new file mode 100644 index 0000000..0cf0504 --- /dev/null +++ b/scripts/check_search.mjs @@ -0,0 +1,91 @@ +// Check the site's search. Runs on the runner's own Node with the standard +// library only, like check_floor_parity.mjs. +// +// node scripts/check_search.mjs ranking cases only +// node scripts/check_search.mjs _site also the built index and the files the scripts load +// +// Ranking: site/search.js is loaded as a module and held to a handful of cases +// that say what a reader should get first. Given the assembled site, every +// entry in search-index.json must point at a page that exists and an id on it, +// and the other files the page scripts fetch or load (the worker, the example +// run) must be there, since check_site_links.mjs only follows tags in HTML. + +import { readFileSync, existsSync } from "node:fs"; +import { createRequire } from "node:module"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const here = path.dirname(fileURLToPath(import.meta.url)); +const Search = createRequire(import.meta.url)(path.join(here, "..", "site", "search.js")); + +const problems = []; +const expect = (condition, what) => { + if (!condition) problems.push(what); +}; + +// ---- ranking ------------------------------------------------------------ + +const fixture = [ + { t: "Methodology", h: "Methodology", u: "docs/methodology.html", x: "How plumbline measures what it measures." }, + { t: "Methodology", h: "Binning scheme and the ECE floor", u: "docs/methodology.html#binning", x: "Equal width bins, ten by default." }, + { t: "README", h: "Install", u: "docs/readme.html#install", x: "The ECE floor is computed for you. Bins bins bins bins bins bins bins." }, + { t: "Plan", h: "Calibración", u: "docs/plan.html#c", x: "Accents fold away." }, +]; +const first = (query) => Search.rank(fixture, query, 5)[0]?.entry.u; +const urls = (query) => Search.rank(fixture, query, 5).map((r) => r.entry.u); + +expect(first("ece floor") === "docs/methodology.html#binning", "a heading match outranks a body match"); +expect(first("bins") === "docs/methodology.html#binning" || first("bins") === "docs/readme.html#install", "bins finds a section"); +expect(Search.rank(fixture, "bins", 5)[0].score <= 3 + 5, "body matches are capped per term"); +expect(urls("ece install").join() === "docs/readme.html#install", "every term must match"); +expect(urls("measur").includes("docs/methodology.html"), "a term matches the start of a word"); +expect(!urls("easures").includes("docs/methodology.html"), "a term does not match inside a word"); +expect(first("CALIBRACION") === "docs/plan.html#c", "case and accents are ignored"); +expect(Search.rank(fixture, " ", 5).length === 0, "an empty query finds nothing"); +expect(Search.rank(fixture, "methodology", 5)[0].entry.u === "docs/methodology.html", "a page's own entry leads its sections on a tie"); + +const piece = Search.snippet("one two three floor four five", "floor", 160); +const marked = piece.marks.map(([s, e]) => piece.text.slice(s, e)); +expect(marked.join() === "floor", `snippet marks the term (got ${JSON.stringify(marked)})`); +const long = `${"word ".repeat(80)}needle ${"word ".repeat(80)}`; +const window_ = Search.snippet(long, "needle", 100); +expect(window_.text.startsWith("… ") && window_.text.endsWith(" …"), "a snippet from the middle says so at both ends"); +expect(window_.marks.length === 1 && window_.text.slice(...window_.marks[0]) === "needle", "the mark survives the ellipsis offset"); + +// ---- the built site ----------------------------------------------------- + +const site = process.argv[2]; +let entries = 0; +if (site) { + const read = (p) => readFileSync(path.join(site, ...p.split("/")), "utf8"); + const index = JSON.parse(read("search-index.json")); + entries = index.length; + expect(Array.isArray(index) && index.length > 20, "the index has entries"); + const pages = new Map(); + for (const entry of index) { + for (const key of ["t", "h", "u", "x"]) { + if (typeof entry[key] !== "string") problems.push(`${JSON.stringify(entry).slice(0, 80)}: ${key} is not a string`); + } + const url = new URL(entry.u, "https://site.invalid/"); + let page = url.pathname.slice(1); + if (page === "" || page.endsWith("/")) page += "index.html"; + if (!pages.has(page)) pages.set(page, existsSync(path.join(site, ...page.split("/"))) ? read(page) : null); + const html = pages.get(page); + if (html === null) { + problems.push(`${entry.u}: no page ${page}`); + continue; + } + const id = decodeURIComponent(url.hash.slice(1)); + if (id && !html.includes(`id="${id}"`)) problems.push(`${entry.u}: no id "${id}" on ${page}`); + } + // Loaded by script rather than by a tag, so check_site_links.mjs cannot see them. + for (const file of ["search-index.json", "search.js", "floor-worker.js", "example-run.json", "floor.js"]) { + expect(existsSync(path.join(site, file)), `${file} is missing from the site`); + } +} + +if (problems.length) { + console.error(`check_search: ${problems.length} problem(s):\n ${problems.join("\n ")}`); + process.exit(1); +} +console.log(`check_search: ranking cases pass${site ? `; ${entries} index entries point at pages and ids that exist` : ""}.`); diff --git a/scripts/render_docs.mjs b/scripts/render_docs.mjs index 5440f5f..9beab0f 100644 --- a/scripts/render_docs.mjs +++ b/scripts/render_docs.mjs @@ -92,6 +92,17 @@ function inlineText(token) { .join(""); } +// As inlineText, but a line break inside a paragraph is a space, as it is to a +// reader. Headings never wrap, so their slugs keep using inlineText. +function readableText(token) { + return (token.children ?? []) + .map((child) => { + if (child.type === "text" || child.type === "code_inline") return child.content; + return child.type === "softbreak" || child.type === "hardbreak" ? " " : ""; + }) + .join(""); +} + // Raw HTML is emitted verbatim, so this validates rather than sanitizes: every // "<" in it must open one of the exact allowed tags, or the build fails. function checkHtml(content, where, problems) { @@ -125,7 +136,7 @@ function sectionsOf(tokens) { current = { id: token.attrGet("id"), heading: inlineText(tokens[i + 1]), level: Number(token.tag.slice(1)), parts: [] }; i += 1; // the heading's own inline token is its title, not its body } else if (token.type === "inline") { - current.parts.push(inlineText(token)); + current.parts.push(readableText(token)); } else if (token.type === "fence" || token.type === "code_block") { current.parts.push(token.content); } @@ -294,6 +305,8 @@ function selfTest() { expect(!html("# T\n").includes('class="anchor"'), "no anchor on h1"); const table = html("| a | b |\n|:-:|--:|\n| 1 | 2 |\n"); expect(!table.includes("style=") && table.includes('class="align-right"'), "table alignment as a class"); + const wrapped = render(job("# T\n\nfirst\nsecond\n")).a.sections[0].text; + expect(wrapped === "first second", `a wrapped line keeps its space (got ${JSON.stringify(wrapped)})`); const out = render(job("# Top\n\nlead `x`\n\n## One\n\nbody\n\n```\ncode\n```\n\n#### Deep\n\nmore\n")).a; expect( JSON.stringify(out.headings.map((h) => [h.level, h.id])) === '[[1,"top"],[2,"one"],[4,"deep"]]', @@ -301,7 +314,7 @@ function selfTest() { ); const sections = out.sections.map((s) => [s.id, s.text]); expect(JSON.stringify(sections) === '[["top","lead x"],["one","body code Deep more"]]', `sections ${JSON.stringify(sections)}`); - console.log("render_docs self-test: 20 cases pass"); + console.log("render_docs self-test: 21 cases pass"); } if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { diff --git a/site/base.css b/site/base.css index 026932e..4c030bf 100644 --- a/site/base.css +++ b/site/base.css @@ -119,3 +119,23 @@ kbd { font: 0.75rem var(--mono); border: 1px solid var(--control); border-radius body { background: #fff; color: #000; } a { color: inherit; } } + +/* ---- search: a dialog built by site.js --------------------------------- */ + +dialog.search { width: min(40rem, calc(100vw - 2rem)); max-height: min(36rem, calc(100vh - 4rem)); margin: 4rem auto auto; padding: 0; color: var(--fg); background: var(--bg); border: 1px solid var(--control); border-radius: 8px; box-shadow: 0 12px 40px rgba(0, 0, 0, 0.25); overflow: hidden; } +dialog.search[open] { display: flex; flex-direction: column; } +dialog.search::backdrop { background: rgba(0, 0, 0, 0.4); } +.search-bar { display: flex; gap: 0.5rem; padding: 0.75rem; border-bottom: 1px solid var(--line); } +.search-bar input { flex: 1; min-width: 0; font: 1rem var(--sans); padding: 0.5rem 0.6rem; color: var(--fg); background: var(--bg); border: 1px solid var(--control); border-radius: 4px; } +.search-close { font: 0.85rem var(--sans); color: var(--fg); background: var(--panel); border: 1px solid var(--control); border-radius: 4px; padding: 0 0.7rem; cursor: pointer; } +.search-note { margin: 0; padding: 0.5rem 1rem 0; font-size: 0.85rem; color: var(--muted); } +.search-note:empty { display: none; } +.search-results { list-style: none; margin: 0; padding: 0.5rem; overflow-y: auto; flex: 1; } +.search-results a { display: block; padding: 0.55rem 0.7rem; border-radius: 6px; text-decoration: none; color: var(--fg); } +.search-results a:hover, .search-results a:focus-visible { background: var(--panel); outline-offset: -2px; } +.search-results .where { display: block; font-weight: 600; color: var(--accent); font-size: 0.95rem; } +.search-results .quote { display: block; font-size: 0.85rem; color: var(--muted); line-height: 1.45; margin-top: 0.15rem; } +.search-results mark { background: var(--warn-bg); color: var(--warn-fg); border-radius: 2px; padding: 0 0.1rem; } +.search-hint { margin: 0; padding: 0.45rem 1rem; border-top: 1px solid var(--line); font-size: 0.78rem; color: var(--muted); } +@media (hover: none) { .search-hint { display: none; } } +@media print { dialog.search { display: none !important; } } diff --git a/site/index.html b/site/index.html index 4a92b1d..1b5606a 100644 --- a/site/index.html +++ b/site/index.html @@ -306,6 +306,7 @@

What computes these numbers

No analytics, no trackers, no external requests.

+