diff --git a/.github/workflows/site.yml b/.github/workflows/site.yml index 1836e98..d6de84e 100644 --- a/.github/workflows/site.yml +++ b/.github/workflows/site.yml @@ -3,9 +3,10 @@ name: Site # The site at tmhsdigital.github.io/plumbline carries a JavaScript # reimplementation of the calibration floor. If it disagrees with the Python, # the site misreports the tool, which is worse than having no site. This -# workflow holds the two together. The site is assembled inside this job, and -# the deploy job (not yet added) will need it, so a wrong calculator or a -# broken doc never ships. A stale site is the better failure. +# workflow holds the two together, checks the assembled site as a visitor would +# meet it, and only then deploys. Pull requests run both checks and never +# deploy; a push to main or a manual run deploys only if both pass. A stale +# site is the better failure. # # It is deliberately separate from ci.yml and release.yml: it shares no jobs, # no concurrency group, and no permissions with either. @@ -18,9 +19,11 @@ on: permissions: contents: read +# A newer push to a pull request makes its older run pointless. On main, a run +# is never cancelled part way, so a deploy is never cut off. concurrency: group: site-${{ github.ref }} - cancel-in-progress: true + cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: parity: @@ -44,15 +47,11 @@ jobs: - name: The committed fixture is what the Python produces today run: uv run python scripts/floor_golden.py --check - # The renderer's refusals (raw HTML off the allowlist, broken links and - # anchors, remote images, a modified vendored file) each have a case. - - name: The doc renderer refuses what it should - run: node scripts/render_docs.mjs --self-test - # Regenerates the worked example from the mock adapter, by the command - # docs/example-report.md records, and refuses if the report is stale. - # Renders the repository's docs from this commit, and refuses on a broken - # link or anchor, or raw HTML outside the allowlist. + # docs/example-report.md records, and refuses if any line of the report + # differs from what that command prints today. Renders the repository's + # docs from this commit, and refuses on a broken link or anchor, or raw + # HTML outside the allowlist. - name: Assemble the site run: uv run python scripts/build_site.py --out _site @@ -63,3 +62,58 @@ jobs: run: | node --version node scripts/check_floor_parity.mjs _site/example-run.json + + # The exact bytes the deploy job publishes. On a pull request this is + # also a downloadable preview of the site. + - name: Keep the assembled site + uses: actions/upload-pages-artifact@v5 + with: + path: _site + + docs: + name: the assembled site's links, anchors, and meta tags resolve + needs: parity + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v5 + + # The renderer's refusals (raw HTML off the allowlist, broken links and + # anchors, remote images, a modified vendored file) each have a case. + - name: The doc renderer refuses what it should + run: node scripts/render_docs.mjs --self-test + + - name: Fetch the assembled site + uses: actions/download-artifact@v8 + with: + name: github-pages + path: artifact + + # Checked as published, not as built: the same archive the deploy job + # serves, unpacked. + - name: Every link, anchor, and meta tag resolves, and nothing loads from another origin + run: | + mkdir site-as-published + tar -xf artifact/artifact.tar -C site-as-published + node scripts/check_site_links.mjs site-as-published + + deploy: + name: deploy to GitHub Pages + needs: [parity, docs] + if: github.ref == 'refs/heads/main' && (github.event_name == 'push' || github.event_name == 'workflow_dispatch') + runs-on: ubuntu-latest + permissions: + pages: write + id-token: write + environment: + name: github-pages + url: ${{ steps.deployment.outputs.page_url }} + # One deploy at a time; a queued one waits rather than cancelling. + concurrency: + group: pages + cancel-in-progress: false + + steps: + - name: Publish the checked site + id: deployment + uses: actions/deploy-pages@v5 diff --git a/CHANGELOG.md b/CHANGELOG.md index 07109e0..3092521 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,7 +17,6 @@ different event from one that moved because it was wrong. `synthetic_floor` that reproduces numpy's seeded random stream draw for draw. `scripts/floor_golden.py` exports golden values from the Python and `scripts/check_floor_parity.mjs` holds the port to them within 1e-9 in CI. - The page is not deployed yet. - The site page: the argument, a worked example that derives the example report's ECE line from its 105 rows in the browser, and a sample-size planner. `scripts/build_site.py` regenerates the example from the mock adapter at build @@ -29,6 +28,13 @@ different event from one that moved because it was wrong. the runner's Node. Each page names the commit and build time it came from. The build fails on a broken link or anchor, on raw HTML outside a short allowlist, and on a modified vendored renderer. +- The site is live at , deployed by + `site.yml` from `main` only after the parity check passes and + `scripts/check_site_links.mjs` finds every link, anchor, and meta tag in the + assembled site resolving and nothing loading from another origin. Every page + carries a canonical URL and Open Graph and Twitter card tags; the card image + (`site/og.png`) is rendered from `scripts/og_image.html`. Missing paths get a + 404 page that links back. ### Changed diff --git a/README.md b/README.md index 58cc66d..0cbdc3a 100644 --- a/README.md +++ b/README.md @@ -5,6 +5,9 @@ Measure whether a decision model's probabilities are trustworthy on your own labeled data, and decide what to do about it. +Check where your own ECE sits against its floor, in the browser, with nothing +installed: **[tmhsdigital.github.io/plumbline](https://tmhsdigital.github.io/plumbline/)**. + > **v0.1.0, one maintainer.** The measurement behaviour is settled; the Python > API and the CLI flags are not, and will change in v0.2. Pin a version if you > build on it. diff --git a/scripts/build_site.py b/scripts/build_site.py index a1123e5..ba2564f 100644 --- a/scripts/build_site.py +++ b/scripts/build_site.py @@ -351,11 +351,12 @@ def render_docs(prov: Provenance) -> dict[str, dict[str, str]]: return rendered -def _nav(current: str | None, up: str) -> str: +def _nav(current: str | None, up: str, docs: str = "") -> str: + """Links to the explainer (``up``) and every doc (``docs`` + slug).""" items = [f'
  • The ECE floor
  • '] for doc in DOCS: mark = ' aria-current="page"' if doc.slug == current else "" - items.append(f'
  • {html.escape(doc.label)}
  • ') + items.append(f'
  • {html.escape(doc.label)}
  • ') joined = "\n ".join(items) return f'' @@ -367,6 +368,47 @@ def _nav(current: str | None, up: str) -> str: "%3Cpath d='M5 10h6l-3 5z' fill='%23555'/%3E%3C/svg%3E" ) +#: Where Pages serves the site. Canonical and Open Graph URLs are absolute, and +#: the 404 page links by absolute path, because it is served at any depth. +SITE_URL = "https://tmhsdigital.github.io/plumbline/" +SITE_PATH = "/plumbline/" +OG_IMAGE_ALT = ( + "A calibration claim has a floor: ECE 0.074 on 105 rows sits below the floor's " + "95th percentile of 0.111, so the result is inconclusive." +) + + +def social_meta(title: str, description: str, url: str) -> str: + """Canonical, Open Graph, and Twitter card tags for one page. + + ``site/index.html`` carries the same tags written out by hand, and + ``scripts/check_site_links.mjs`` checks every page has them and that its + canonical and og:url name the page itself. + """ + title, description = html.escape(title), html.escape(description) + image = f"{SITE_URL}og.png" + alt = html.escape(OG_IMAGE_ALT) + return "\n".join( + ( + f'', + '', + '', + f'', + f'', + f'', + f'', + '', + '', + f'', + '', + f'', + f'', + f'', + f'', + ) + ) + + PAGE = """ @@ -374,15 +416,16 @@ def _nav(current: str | None, up: str) -> str: {title} | plumbline +{meta} - - + +
    -

    plumbline

    +

    plumbline

    {nav}
    @@ -398,6 +441,19 @@ def _nav(current: str | None, up: str) -> str: """ +def _page(title: str, description: str, meta: str, root: str, nav: str, body: str) -> str: + return PAGE.format( + title=html.escape(title), + description=html.escape(description, quote=True), + meta=meta, + icon=ICON, + root=root, + nav=nav, + body=body, + repo=REPO, + ) + + def write_docs(out: Path, prov: Provenance, rendered: dict[str, dict[str, str]]) -> None: docs = out / "docs" docs.mkdir() @@ -407,14 +463,8 @@ def write_docs(out: Path, prov: Provenance, rendered: dict[str, dict[str, str]]) f'

    {prov.line(doc)}

    \n' f'
    \n{rendered[doc.slug]["html"]}
    ' ) - page = PAGE.format( - title=html.escape(doc.label), - description=html.escape(doc.blurb, quote=True), - nav=_nav(doc.slug, "../"), - body=body, - repo=REPO, - icon=ICON, - ) + meta = social_meta(f"{doc.label} | plumbline", doc.blurb, f"{SITE_URL}docs/{doc.slug}.html") + page = _page(doc.label, doc.blurb, meta, "../", _nav(doc.slug, "../"), body) (docs / f"{doc.slug}.html").write_text(page, encoding="utf-8") listing = "\n".join( @@ -431,17 +481,24 @@ def write_docs(out: Path, prov: Provenance, rendered: dict[str, dict[str, str]]) f"at {at}.

    \n" f'\n' ) - (docs / "index.html").write_text( - PAGE.format( - title="Documentation", - description="plumbline's documentation, rendered from the repository.", - nav=_nav(None, "../"), - body=index, - repo=REPO, - icon=ICON, - ), - encoding="utf-8", + description = "plumbline's documentation, rendered from the repository." + meta = social_meta("Documentation | plumbline", description, f"{SITE_URL}docs/") + page = _page("Documentation", description, meta, "../", _nav(None, "../"), index) + (docs / "index.html").write_text(page, encoding="utf-8") + + +def write_404(out: Path) -> None: + """Pages serves this for any missing path, at any depth, so it links absolutely.""" + body = ( + '' ) + nav = _nav(None, SITE_PATH, f"{SITE_PATH}docs/") + meta = '' + page = _page("Not found", "No page at this address.", meta, SITE_PATH, nav, body) + (out / "404.html").write_text(page, encoding="utf-8") def main() -> int: @@ -470,6 +527,7 @@ def main() -> int: shutil.copytree(SITE, out, ignore=shutil.ignore_patterns("vendor")) (out / "example-run.json").write_text(json.dumps(example, indent=1) + "\n", encoding="utf-8") write_docs(out, prov, rendered) + write_404(out) print(f"site assembled in {out}") return 0 diff --git a/scripts/check_site_links.mjs b/scripts/check_site_links.mjs new file mode 100644 index 0000000..b2d7c46 --- /dev/null +++ b/scripts/check_site_links.mjs @@ -0,0 +1,195 @@ +// Check the assembled site as a visitor would meet it. Runs on the runner's own +// Node with the standard library only. +// +// node scripts/check_site_links.mjs _site +// node scripts/check_site_links.mjs https://tmhsdigital.github.io/plumbline/ +// +// Given a directory it checks the build before deploy; given the site's URL it +// checks what is actually being served. Either way, starting from the explainer +// and following every internal link, it fails if: +// +// - an internal link, stylesheet, script, or image does not resolve, +// - a #fragment names no id on the page it points to, +// - a page loads anything (script, stylesheet, image, frame) from another origin, +// - a page lacks its canonical, Open Graph, or Twitter card tags, or its +// canonical and og:url do not name the page itself, +// - og.png is not a 1200x630 PNG, +// - the 404 page is missing, links anywhere that does not resolve, or (live) +// a missing path is not answered with it and a 404 status. + +import { readFile } from "node:fs/promises"; +import path from "node:path"; + +const SITE_URL = "https://tmhsdigital.github.io/plumbline/"; +const SITE_PATH = new URL(SITE_URL).pathname; // "/plumbline/" + +const target = process.argv[2]; +if (!target) { + console.error("usage: check_site_links.mjs "); + process.exit(2); +} +const live = /^https?:\/\//.test(target); +const base = live ? (target.endsWith("/") ? target : `${target}/`) : null; + +// Fetch one site path ("" is the explainer). Returns { status, text, bytes }. +async function get(sitePath) { + if (live) { + const response = await fetch(base + sitePath, { redirect: "follow", cache: "no-store" }); + const bytes = new Uint8Array(await response.arrayBuffer()); + return { status: response.status, bytes, text: new TextDecoder().decode(bytes) }; + } + try { + const bytes = new Uint8Array(await readFile(path.join(target, ...sitePath.split("/")))); + return { status: 200, bytes, text: new TextDecoder().decode(bytes) }; + } catch { + return { status: 404, bytes: new Uint8Array(), text: "" }; + } +} + +// Map a reference found on the page at `from` to a site path, or report it as +// external. `from` is a site path such as "docs/plan.html". +function resolve(reference, from) { + if (/^(data|mailto):/i.test(reference)) return { skip: true }; + let url; + if (/^https?:\/\//i.test(reference)) { + if (!reference.startsWith(SITE_URL)) return { external: reference }; + url = new URL(reference); + } else if (reference.startsWith("//")) { + return { external: reference }; + } else { + url = new URL(reference, SITE_URL + from); + } + if (!url.pathname.startsWith(SITE_PATH)) return { error: `${reference} leaves the site's path` }; + let sitePath = decodeURIComponent(url.pathname.slice(SITE_PATH.length)); + if (sitePath === "" || sitePath.endsWith("/")) sitePath += "index.html"; + return { sitePath, fragment: url.hash ? decodeURIComponent(url.hash.slice(1)) : null }; +} + +// Every tag with a reference, as [tag, attribute, value, rel]. +function references(html) { + const found = []; + for (const [, tag, attributes] of html.matchAll(/<(a|link|script|img|iframe|source|video|audio)\b([^>]*)>/gi)) { + const rel = /\brel="([^"]*)"/i.exec(attributes)?.[1] ?? ""; + for (const [, name, value] of attributes.matchAll(/\b(href|src)="([^"]*)"/gi)) { + found.push([tag.toLowerCase(), name.toLowerCase(), value.replaceAll("&", "&"), rel.toLowerCase()]); + } + } + return found; +} + +const ids = (html) => new Set([...html.matchAll(/\bid="([^"]+)"/g)].map((match) => match[1])); +const literal = (text) => text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +const meta = (html, key, value) => + new RegExp(``).test(html); +const metaValue = (html, key) => + new RegExp(``).exec(html)?.[1] ?? null; + +const problems = []; +const pages = new Map(); // site path -> html, for fragment checks +const checked = new Set(); + +// The canonical URL a page should declare. +const canonicalFor = (sitePath) => + SITE_URL + (sitePath.endsWith("index.html") ? sitePath.slice(0, -"index.html".length) : sitePath); + +async function page(sitePath) { + if (!pages.has(sitePath)) { + const { status, text } = await get(sitePath); + pages.set(sitePath, status === 200 ? text : null); + } + return pages.get(sitePath); +} + +function checkSocial(sitePath, html) { + const want = canonicalFor(sitePath); + const canonical = //.exec(html)?.[1]; + if (canonical !== want) problems.push(`${sitePath}: canonical is ${canonical ?? "missing"}, expected ${want}`); + if (metaValue(html, "og:url") !== want) problems.push(`${sitePath}: og:url does not name the page`); + for (const key of ["og:title", "og:description", "og:image:alt", "twitter:title", "twitter:description"]) { + if (!metaValue(html, key)) problems.push(`${sitePath}: ${key} is missing or empty`); + } + for (const [key, value] of [ + ["og:image", `${SITE_URL}og.png`], + ["og:image:width", "1200"], + ["og:image:height", "630"], + ["twitter:card", "summary_large_image"], + ["twitter:image", `${SITE_URL}og.png`], + ]) { + if (!meta(html, key, value)) problems.push(`${sitePath}: ${key} is not ${value}`); + } +} + +// Check one page's references; queue the internal HTML pages it links to. +async function crawl(sitePath, { social = true, servedAs = sitePath } = {}) { + if (checked.has(sitePath)) return; + checked.add(sitePath); + const html = await page(sitePath); + if (html === null) { + problems.push(`${sitePath}: not found`); + return; + } + if (social) checkSocial(sitePath, html); + const queue = []; + for (const [tag, , value, rel] of references(html)) { + const where = `${servedAs}: <${tag}> ${value}`; + if (value.startsWith("#")) { + if (!ids(html).has(decodeURIComponent(value.slice(1)))) problems.push(`${where}: no such id on this page`); + continue; + } + const loads = tag !== "a" && !(tag === "link" && /\bcanonical\b/.test(rel)); + const resolved = resolve(value, servedAs); + if (resolved.skip) continue; + if (resolved.error) { + problems.push(`${where}: ${resolved.error}`); + continue; + } + if (resolved.external) { + if (loads) problems.push(`${where}: loads from another origin`); + continue; + } + const { sitePath: to, fragment } = resolved; + if (to.endsWith(".html")) { + const linked = await page(to); + if (linked === null) problems.push(`${where}: does not resolve`); + else { + if (fragment && !ids(linked).has(fragment)) problems.push(`${where}: no id "${fragment}" on ${to}`); + queue.push(to); + } + } else { + const { status } = await get(to); + if (status !== 200) problems.push(`${where}: does not resolve (${status})`); + } + } + for (const next of queue) await crawl(next); +} + +await crawl("index.html"); + +// og.png: a real PNG of the size the tags claim. +{ + const { status, bytes } = await get("og.png"); + const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength); + const png = status === 200 && bytes.length > 24 && view.getUint32(0) === 0x89504e47; + if (!png) problems.push("og.png: missing or not a PNG"); + else if (view.getUint32(16) !== 1200 || view.getUint32(20) !== 630) { + problems.push(`og.png: ${view.getUint32(16)}x${view.getUint32(20)}, expected 1200x630`); + } +} + +// The 404 page, checked as if served at a deep missing path. +{ + const missing = `no-such-page-${Date.now()}/deeper/still.html`; + if (live) { + const { status, text } = await get(missing); + if (status !== 404) problems.push(`${missing}: status ${status}, expected 404`); + if (!text.includes("

    Not found

    ")) problems.push(`${missing}: not answered with the 404 page`); + } + await crawl("404.html", { social: false, servedAs: missing }); +} + +const htmlPages = [...checked].length; +if (problems.length) { + console.error(`${problems.length} problem(s) on ${live ? base : target}:\n ${problems.join("\n ")}`); + process.exit(1); +} +console.log(`${htmlPages} pages checked on ${live ? base : target}: every link, anchor, and meta tag resolves; nothing loads from another origin.`); diff --git a/scripts/og_image.html b/scripts/og_image.html new file mode 100644 index 0000000..bc7619d --- /dev/null +++ b/scripts/og_image.html @@ -0,0 +1,57 @@ + + + + + + + + +
    +

    plumbline

    +

    A calibration claim
    has a floor

    +
    + + +
    + 105 rows: inconclusive, not a pass + tmhsdigital.github.io/plumbline +
    + + diff --git a/site/index.html b/site/index.html index 6f4679e..159f64b 100644 --- a/site/index.html +++ b/site/index.html @@ -5,6 +5,21 @@ plumbline: the ECE floor + + + + + + + + + + + + + + + diff --git a/site/og.png b/site/og.png new file mode 100644 index 0000000..273ab27 Binary files /dev/null and b/site/og.png differ