diff --git a/CHANGELOG.md b/CHANGELOG.md index 9e7b01d..07109e0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -58,6 +58,10 @@ different event from one that moved because it was wrong. ### Documentation +- `docs/example-report.md` is now checked line for line against its recorded + command on every site build, not only its ECE line, and the README says that + instead of calling it "unedited". + - README rewritten for a reader arriving from a link: what the tool is now precedes what it is not, and decision model, calibration, cascade, Noul, binning noise and the Jev wire format are each defined where they appear. diff --git a/README.md b/README.md index f10c5f0..58cc66d 100644 --- a/README.md +++ b/README.md @@ -79,7 +79,7 @@ This is not a rounding concern. On a few hundred rows, a calibration claim is frequently not measurable at all. So plumbline computes that floor by simulation and prints every inferential -figure against it. Here is a real line from the example report, unedited: +figure against it. Here is a real line from the example report: > ECE 0.0740 over 105 rows (10 equal width bins), against a calibrated-model > floor of 0.0707 (95th percentile 0.1109): INCONCLUSIVE at this sample size. A @@ -222,9 +222,11 @@ only path on which the recalibration numbers mean anything. ## Example report -[docs/example-report.md](docs/example-report.md) is real, unedited output. The -ECE line is quoted in [The argument](#the-argument) above. Three more, each -showing the tool declining to do something: +[docs/example-report.md](docs/example-report.md) is the output of one seeded mock +run, and CI holds it to that: on every change and before every deploy it reruns +the report's recorded command and refuses if any line differs, other than the +date. The ECE line is quoted in [The argument](#the-argument) above. Three more, +each showing the tool declining to do something: > Not reported. recalibration needs at least 200 held-out evaluation rows and > this split has 53. Fitting a temperature on fewer rows produces a number whose diff --git a/docs/example-report.md b/docs/example-report.md index 2525517..8c521c7 100644 --- a/docs/example-report.md +++ b/docs/example-report.md @@ -1,14 +1,16 @@ # Example report -This is real, unedited output from `plumbline run`. Everything below the rule is -exactly what the tool printed. Nothing in it was tuned to look good. +This is real output from `plumbline run`. Everything below the rule is exactly +what the command in the table prints, and CI checks that on every change and +before every deploy: it reruns the command and refuses if any line differs, +other than the date. Nothing in it was tuned to look good. | | | |---|---| | Dataset | `datasets/public/jevbench-hard.jsonl`, the vendored JevBench public fixture | | Rows | 111 read, 105 scored, 6 held back as ordinal score rows | | Arm | the `mock` adapter, seed 7, target accuracy 0.8 | -| Date | 2026-09-21 | +| Date | 2026-09-22 | | Command | `plumbline run datasets/public/jevbench-hard.jsonl --adapter mock --format jevbench --seed 7 --accuracy 0.8 --report example.md` | **The arm is a seeded mock, not a vendor.** It is a deterministic stand-in that diff --git a/scripts/build_site.py b/scripts/build_site.py index 4e1d411..a1123e5 100644 --- a/scripts/build_site.py +++ b/scripts/build_site.py @@ -8,7 +8,10 @@ produced by re-running the command that report records, together with the figures the report prints. The page recomputes the ECE and its floor from the rows, in JavaScript, and ``scripts/check_floor_parity.mjs`` fails the build if - what it derives differs from the report by a single character. + what it derives differs from the report by a single character. Before any + of that, every line of the committed report is compared with what the + command prints today (only the generation date and path separators are + normalized), and the build refuses on any difference. ``docs/`` The repository's markdown docs as they are at the commit being built: each @@ -89,7 +92,7 @@ class Doc: "docs/example-report.md", "example-report", "Example report", - "Unedited output of one mock run: the report the explainer derives its example from.", + "One seeded mock run, rechecked against its command on every build.", ), Doc("docs/PLAN.md", "plan", "Plan", "What is built, what is deliberately not, and why."), Doc("CHANGELOG.md", "changelog", "Changelog", "Notable changes per release."), @@ -143,6 +146,47 @@ def _single(pattern: re.Pattern[str], lines: list[str], what: str) -> re.Match[s return found[0] +#: The report's generation date, the one line that legitimately differs between +#: the committed report and a rerun of its command on a later day. +GENERATED_LINE = re.compile(r"^Generated \d{4}-\d{2}-\d{2} against ") + + +def _comparable(line: str) -> str: + """A report line with what depends on when and where it ran taken out. + + The date is dropped, and the dataset path's separators are made POSIX: the + committed report may have been written on Windows and CI runs on Linux. + Nothing else is normalized. The mock's latencies are simulated from its + seed, so even the latency line must match exactly. + """ + if GENERATED_LINE.match(line): + return GENERATED_LINE.sub("Generated against ", line) + return line.replace("datasets\\public\\", "datasets/public/") + + +def _require_same_report(committed: list[str], regenerated: list[str]) -> None: + """Every line of the committed report must be what its command prints today.""" + ours = [_comparable(line) for line in committed] + theirs = [_comparable(line) for line in regenerated] + if ours == theirs: + return + differing = [ + (number, mine, fresh) + for number, (mine, fresh) in enumerate(zip(ours, theirs, strict=False), start=1) + if mine != fresh + ] + detail = "".join( + f"\n line {number}\n report says: {mine}\n run gives: {fresh}" + for number, mine, fresh in differing[:3] + ) + if len(ours) != len(theirs): + detail += f"\n the report has {len(ours)} lines and the run printed {len(theirs)}" + raise BuildError( + "docs/example-report.md no longer matches what its own command produces." + f"{detail}\nRegenerate the example report before deploying." + ) + + def _example_arguments(command: str, results: Path, report: Path) -> list[str]: """The report's recorded command, pointed at a scratch directory.""" words = shlex.split(command) @@ -195,14 +239,7 @@ def build_example() -> dict[str, Any]: if run.adapter_name != ALLOWED_ADAPTER: raise BuildError(f"the example artifact came from {run.adapter_name!r}, not the mock") - fresh_line = _single(ECE_LINE, regenerated, "ECE").group(1) - if fresh_line != ece_match.group(1): - raise BuildError( - "docs/example-report.md no longer matches what its own command produces.\n" - f" report says: {ece_match.group(1)}\n" - f" run gives: {fresh_line}\n" - "Regenerate the example report before deploying." - ) + _require_same_report(printed, regenerated) probabilities = list(run.probabilities().values) outcomes = run.outcomes