diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e859bc5..41c0161 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -126,6 +126,23 @@ jobs: echo "$out" | grep -q "requires" || { echo "expected a named refusal, got:"; echo "$out"; exit 1; } echo "$out" | grep -qv "Traceback" || { echo "traceback leaked"; exit 1; } + # The house style has no em dashes, and the report's own strings are prose a + # reader sees. datasets/public/ is excluded because it holds vendored + # third-party rows that are reproduced as published. + prose: + name: no em dashes in the docs or the report strings + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v5 + + - name: Tracked markdown and src/ contain no em dash + run: | + if git grep -n -P '\x{2014}' -- '*.md' 'src/' ':!datasets/public/'; then + echo "::error::em dash found above; use a comma, colon, parentheses, or a new sentence" + exit 1 + fi + # Gate 6 found that the built distribution is a different artifact from the # source tree: the entry point, the optional extra boundary, and py.typed all # only exist once packaged. diff --git a/CHANGELOG.md b/CHANGELOG.md index e69e208..93eb29f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,6 +24,13 @@ different event from one that moved because it was wrong. time and refuses if `docs/example-report.md` no longer matches its own command. +### Changed + +- Report bullets separate the label from the figure with a colon instead of an + em dash (`- **Cost**: not reported.`), and the docs no longer use em dashes. + A CI job now fails on an em dash in tracked markdown or `src/`. **This + changes report text, not any number.** + ### Fixed - The inconclusive verdict read as a pass. "Not distinguishable from a @@ -55,7 +62,7 @@ different event from one that moved because it was wrong. since the closest thing to a disclosure this project has had was a vendor's price, which no scanner recognises. -## v0.1.0 — 2026-09-21 +## v0.1.0 (2026-09-21) First release. diff --git a/datasets/public/README.md b/datasets/public/README.md index 4bae37b..a1386f1 100644 --- a/datasets/public/README.md +++ b/datasets/public/README.md @@ -25,7 +25,7 @@ prompts differ, and the scoring differs: - JevBench asks each row through its own harness. plumbline composes a case text from the row's `question.instructions` above its `state` and sends that through whichever adapter is under test, with that adapter's own prompt shape. -- JevBench's rows come in three question types — `choice`, `noul`, and `score`. +- JevBench's rows come in three question types: `choice`, `noul`, and `score`. plumbline asks each row as the type it declares, where the transport has one: a `noul` row is asked as a Noul by `typesafe_wire`, and as a two-option choice by a transport with no Noul. Those are different questions, so every record @@ -53,6 +53,6 @@ python examples/smoke_public_dataset.py Loads this file, runs the mock adapter over it, computes the metrics, and writes an artifact. It makes no network call and spends nothing. The numbers are -meaningless — a seeded mock is answering — and the point is to prove the loader, +meaningless (a seeded mock is answering), and the point is to prove the loader, the runner, the metrics, and the artifact compose on a real file before a live run turns a mistake into money. diff --git a/docs/PLAN.md b/docs/PLAN.md index 5a58c01..8acabcb 100644 --- a/docs/PLAN.md +++ b/docs/PLAN.md @@ -16,16 +16,16 @@ The remaining work is the v0.2 milestone, and #3 comes first. ## Phases -- **Phase 6 — adapters: `local_logits` and `generative`.** Landed. The restricted +- **Phase 6: adapters, `local_logits` and `generative`.** Landed. The restricted softmax over a pinned checkpoint, and the text-generating control arm. -- **Phase 7 — datasets.** Landed. The JSONL loader that refuses unscoreable +- **Phase 7: datasets.** Landed. The JSONL loader that refuses unscoreable rows, the JevBench public fixture and its translation, and the end-to-end smoke run in `examples/smoke_public_dataset.py`. -- **Phase 8 — report and CLI.** Landed. The markdown report groups arms by +- **Phase 8: report and CLI.** Landed. The markdown report groups arms by `probability_semantics`, states every figure's row count and null, and demotes MCE to diagnostics. `src/plumbline/cli.py` exists, so the `plumbline` console script pyproject declares now works: `run`, `report`, `adapters`, `version`. -- **Phase 9 — recalibration, the cascade, and the methodology.** Landed. The +- **Phase 9: recalibration, the cascade, and the methodology.** Landed. The report fits a temperature on a held-out half and prints the verdict rather than the number when the verdict is a refusal; the cascade section ends in one sentence naming the threshold, the coverage, the expected cost, and the cost @@ -71,7 +71,7 @@ Settled during the build. Reopen one only with a reason, not from scratch. - ECE and MCE are always reported against their calibrated-null floor. - MCE is demoted to a diagnostics block, never beside ECE, because it cannot detect gross overconfidence at 500 rows. -- Recalibration has three verdicts — recommended, partial, refused — and emits no +- Recalibration has three verdicts (recommended, partial, refused) and emits no temperature on refusal. - Latency percentiles use nearest rank, not interpolation. - Adapters are organized by transport; a new vendor is config, not code. @@ -312,7 +312,7 @@ side. The figures are deliberately not reproduced here; read the page. That settles the open question that had blocked cost entirely. The previous shipped entry quoted an SDK field description for the output side and carried no input price at all, so `is_priced` was False and **the cost guard refused any run -with `--max-cost-usd` set** — it cannot bound a run it cannot cost. +with `--max-cost-usd` set**: it cannot bound a run it cannot cost. **What plumbline ships is a separate decision from what the tariff says.** MCA 14.1 makes the vendor's pricing information confidential and explicitly diff --git a/docs/example-report.md b/docs/example-report.md index 7572da9..2525517 100644 --- a/docs/example-report.md +++ b/docs/example-report.md @@ -69,15 +69,15 @@ The vendor asserts these probabilities are calibrated. Whether that survives con ### mock -- **Model** — requested `mock-1`, reported `mock-1`. -- **Asked** — 67 choice asked as choice, 38 noul asked as choice. +- **Model**: requested `mock-1`, reported `mock-1`. +- **Asked**: 67 choice asked as choice, 38 noul asked as choice. - 38 noul rows were asked as choice questions, which is a different question from the one the dataset states. Not comparable with an arm that asked them as noul. - Accuracy 0.7714 over 105 rows, against a chance null of 0.3416 (95th percentile 0.4190): better than chance at this sample size. - ECE 0.0740 over 105 rows (10 equal width bins), against a calibrated-model floor of 0.0707 (95th percentile 0.1109): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. - Brier 0.1711 over 105 rows, against a calibrated-model floor of 0.1489 (95th percentile 0.1839): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. -- **Confidence** — AUROC 0.6034 over 105 rows, against a permutation null of 0.4997 (95th percentile 0.6116): INCONCLUSIVE at this sample size. Permutation would often score this well on this many rows, so this dataset cannot tell the two apart. This is not a result in either direction. Collect more rows to make the question answerable. -- **Cost** — not reported. No cost available. None of the 105 cases could be priced, so cost is not reported rather than being shown as zero. 105 rows: tokens were reported, but the model that answered is not priced. -- **Latency** — p50 37.7ms, p95 85.0ms, p99 104.7ms over 105 live calls +- **Confidence**: AUROC 0.6034 over 105 rows, against a permutation null of 0.4997 (95th percentile 0.6116): INCONCLUSIVE at this sample size. Permutation would often score this well on this many rows, so this dataset cannot tell the two apart. This is not a result in either direction. Collect more rows to make the question answerable. +- **Cost**: not reported. No cost available. None of the 105 cases could be priced, so cost is not reported rather than being shown as zero. 105 rows: tokens were reported, but the model that answered is not priced. +- **Latency**: p50 37.7ms, p95 85.0ms, p99 104.7ms over 105 live calls #### Recalibration diff --git a/src/plumbline/report/markdown.py b/src/plumbline/report/markdown.py index bb44f38..eb438a2 100644 --- a/src/plumbline/report/markdown.py +++ b/src/plumbline/report/markdown.py @@ -182,12 +182,12 @@ def _arm(result: RunResult, options: ReportOptions) -> list[str]: lines = ["", f"### {result.adapter_name}", "", *_provenance(result, options)] if excluded: lines.append( - f"- **Excluded** — {excluded} rows of an unsupported question type were not scored." + f"- **Excluded**: {excluded} rows of an unsupported question type were not scored." ) lines.extend(_asked_as(scoreable)) if not successes: - lines.append("- **No figures** — every case failed or was refused, so there is nothing") + lines.append("- **No figures**: every case failed or was refused, so there is nothing") lines.append(" to measure. The failures are in the artifact.") return lines @@ -203,7 +203,7 @@ def _arm(result: RunResult, options: ReportOptions) -> list[str]: ) if failures: lines.append( - f"- **Failures** — {len(failures)} of {len(scoreable)} cases produced no " + f"- **Failures**: {len(failures)} of {len(scoreable)} cases produced no " "prediction and are excluded from accuracy rather than scored wrong." ) @@ -440,7 +440,7 @@ def _scaled(prediction: Prediction, temperature: float) -> float: def _provenance(result: RunResult, options: ReportOptions) -> list[str]: reported = result.model_reported or "not reported" line = ( - f"- **Model** — requested `{result.model_requested}`, reported `{reported}`" + f"- **Model**: requested `{result.model_requested}`, reported `{reported}`" + (f", revision `{result.revision}`" if result.revision else "") + "." ) @@ -461,7 +461,7 @@ def _asked_as(records: Sequence[CaseRecord]) -> list[str]: f"{count} {question_type} asked as {asked}" for (question_type, asked), count in sorted(pairs.items()) ) - lines = [f"- **Asked** — {described}."] + lines = [f"- **Asked**: {described}."] mismatched = { (question_type, asked): count @@ -492,7 +492,7 @@ def _calibration_lines( try: probabilities.require_reportable() except NotCalibratableError as absent: - return [f"- **Calibration** — not reported. {absent}"] + return [f"- **Calibration**: not reported. {absent}"] figure = calibration.ece_figure( probabilities, @@ -521,14 +521,14 @@ def _confidence_lines( values = tuple(record.prediction.confidence for record in successes if record.prediction) if all(value is None for value in values): return [ - "- **Confidence** — not reported. This arm reports no confidence statistic: a " + "- **Confidence**: not reported. This arm reports no confidence statistic: a " "yes/no answer has no distribution to summarize, so the number does not exist " "rather than being missing." ] if any(value is None for value in values): missing = sum(1 for value in values if value is None) return [ - f"- **Confidence** — not reported. {missing} of {len(values)} rows carry no " + f"- **Confidence**: not reported. {missing} of {len(values)} rows carry no " "confidence, and dropping them silently would change which cases the figure " "covers." ] @@ -537,8 +537,8 @@ def _confidence_lines( try: figure = baseline.auroc_figure(series, outcomes, n_boot=options.n_boot, seed=options.seed) except ValueError as undefined: - return [f"- **Confidence** — not reported. {undefined}"] - return [f"- **Confidence** — {figure.statement()}"] + return [f"- **Confidence**: not reported. {undefined}"] + return [f"- **Confidence**: {figure.statement()}"] def _distribution_caveat(result: RunResult, successes: Sequence[CaseRecord]) -> list[str]: @@ -551,7 +551,7 @@ def _distribution_caveat(result: RunResult, successes: Sequence[CaseRecord]) -> if not without: return [] return [ - f"- **Distribution** — {without} of {len(successes)} rows reported a probability " + f"- **Distribution**: {without} of {len(successes)} rows reported a probability " "with no distribution behind it. Those rows are outside the multiclass Brier " "figure, and the temperature that can be fitted for them is the one-parameter " "approximation, which is weaker than the multiclass form even when the " @@ -568,10 +568,10 @@ def _cost_lines( bases=[record.cost_basis for record in scoreable], ) if summary.total_usd is None: - return [f"- **Cost** — not reported. {summary.note}"] + return [f"- **Cost**: not reported. {summary.note}"] line = ( - f"- **Cost** — ${summary.total_usd:.4f} over {summary.priced_cases} priced rows, " + f"- **Cost**: ${summary.total_usd:.4f} over {summary.priced_cases} priced rows, " f"${summary.per_case_usd:.6f} per case" ) if summary.per_correct_usd is not None: @@ -587,11 +587,11 @@ def _latency_lines(result: RunResult) -> list[str]: live = [record.prediction.latency_ms for record in result.live_calls if record.prediction] if not live: return [ - "- **Latency** — not reported. No call went out, so every latency here would " + "- **Latency**: not reported. No call went out, so every latency here would " "be a measurement of disk." ] summary = latency.summarize(live, excluded_cache_hits=len(result.records) - len(live)) - return [f"- **Latency** — {summary}"] + return [f"- **Latency**: {summary}"] def _diagnostics( @@ -615,7 +615,7 @@ def _diagnostics( try: probabilities.require_reportable() except NotCalibratableError: - lines.append("- **MCE** — not reported. This arm reports no probability.") + lines.append("- **MCE**: not reported. This arm reports no probability.") return lines try: @@ -629,7 +629,7 @@ def _diagnostics( ) lines.append(f"- {figure.statement()}") except ValueError as undefined: - lines.append(f"- **MCE** — not reported. {undefined}") + lines.append(f"- **MCE**: not reported. {undefined}") distributions = [record.prediction.distribution for record in successes if record.prediction] if all(distribution is not None for distribution in distributions) and distributions: @@ -643,7 +643,7 @@ def _diagnostics( lines.append(f"- {multiclass.statement()}") else: lines.append( - "- **Multiclass Brier** — not reported. This arm supplied no distribution, so " + "- **Multiclass Brier**: not reported. This arm supplied no distribution, so " "the multiclass form does not exist for it. It is not zero." ) return lines