From 3a71d115e2c75d843c85128dcfb48aa0c4a31827 Mon Sep 17 00:00:00 2001 From: TMHSDigital <154358121+TMHSDigital@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:45:57 -0400 Subject: [PATCH] docs: document the dataset format, fix the vendor command, and name the PyPI collision #61: nothing documented the JSONL format a user writes their own data in. docs/datasets.md gives the fields (id, text, labels, gold_label, and the optional question_type and label_descriptions), what each question type means, an example row of each kind, what the loader refuses and why, and the commands a user needs next: a hosted run with a limit and a cap, a local checkpoint with its pinned 40-hex revision, the cascade's two costs, and plumbline report over existing artifacts. It is hosted on the site under Using plumbline, and a test loads its example rows through the loader so the documented format cannot drift from the real one. #29: the README's hosted-vendor command passed --max-cases 40 against the fixture's 105 scoreable rows, which refuses rather than truncates, so the first real command a user copied sent nothing. It uses --limit 40, and --max-cases now says in its help that it is a guard, not a truncation. #60: the README said pip install plumbline "will not work". It works, and installs an unrelated project with the same name. The README says so and gives pip commands that install this repository, the local_logits missing-extra error says uv sync --extra local (and the pip form) instead of pointing at the PyPI package, and PLAN notes the name is taken. Fixes #29. Fixes #60. Fixes #61. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 13 +++ README.md | 19 ++++- docs/PLAN.md | 4 +- docs/datasets.md | 114 +++++++++++++++++++++++++ scripts/build_site.py | 7 ++ src/plumbline/adapters/local_logits.py | 5 +- src/plumbline/cli.py | 9 +- tests/test_datasets.py | 19 +++++ 8 files changed, 183 insertions(+), 7 deletions(-) create mode 100644 docs/datasets.md diff --git a/CHANGELOG.md b/CHANGELOG.md index f6928ba..0d787bc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -185,6 +185,19 @@ different event from one that moved because it was wrong. ### Documentation +- A new page, [Your own data](docs/datasets.md), gives the row format, what + the loader refuses and why, and the commands a user needs next: a local + checkpoint with its pinned revision, the cascade's two costs, and + `plumbline report` for runs that already happened (#61). A test holds its + example rows to the loader. +- The README's hosted-vendor command used `--max-cases 40` against 105 rows, + which refuses rather than truncates, so it sent nothing (#29); it uses + `--limit 40`, and `--max-cases` says in its help that it is a guard. +- The README said `pip install plumbline` "will not work"; it installs an + unrelated project of the same name (#60). It now says so and gives pip + commands that install this one, and the missing-extra error no longer points + at the PyPI package. + - `docs/example-report.md` is now checked line for line against its recorded command on every site build, not only its ECE line, and the README says that instead of calling it "unedited". diff --git a/README.md b/README.md index 825c5f1..e5a8e44 100644 --- a/README.md +++ b/README.md @@ -109,8 +109,16 @@ That refusal is the product. An older Python gives a resolver error rather than a clear message, so check with `python --version` first. -**plumbline is not on PyPI.** `pip install plumbline` will not work. Clone the -repository; that is the intended install path for v0.1. +**plumbline is not on PyPI, and `pip install plumbline` installs something +else.** The name on PyPI belongs to an unrelated project, so that command +succeeds and gives you the wrong tool. Clone the repository, which is the +intended install path for v0.1. If you use pip rather than uv, install from the +repository itself: + +``` +pip install "plumbline @ git+https://github.com/TMHSDigital/plumbline" +pip install "plumbline[local] @ git+https://github.com/TMHSDigital/plumbline" # with the local extra +``` Nothing in this first section needs an API key or spends anything. @@ -206,7 +214,7 @@ $env:TYPESAFE_API_KEY = "your-key-here" uv run plumbline run datasets/public/jevbench-hard.jsonl ` --adapter typesafe_wire --model jev-latest --format jevbench ` --results results --report results/report.md ` - --max-cases 40 --pricing my-pricing.json + --limit 40 --pricing my-pricing.json ``` @@ -220,7 +228,7 @@ export TYPESAFE_API_KEY="your-key-here" uv run plumbline run datasets/public/jevbench-hard.jsonl \ --adapter typesafe_wire --model jev-latest --format jevbench \ --results results --report results/report.md \ - --max-cases 40 --pricing my-pricing.json + --limit 40 --pricing my-pricing.json ``` @@ -231,6 +239,9 @@ Without a pricing table the run still works; cost reports as unpriced, and Your own data goes in `datasets/private/`, which is gitignored, and that is the only path on which the recalibration numbers mean anything. +[Your own data](docs/datasets.md) gives the row format, what is refused and why, +and the commands to run next: a local checkpoint, the cascade's two costs, and +`plumbline report` for runs that already happened. ## Example report diff --git a/docs/PLAN.md b/docs/PLAN.md index c57c01f..38b1816 100644 --- a/docs/PLAN.md +++ b/docs/PLAN.md @@ -197,7 +197,9 @@ defensible merely because projects usually have one. - **`CITATION.cff`.** Nobody has cited this. A citation file asserting how to cite work nobody has referenced is a claim about its significance rather than a service to a reader. Revisit if someone references the METHODOLOGY results. -- **PyPI publishing.** Premature. It commits the project to a name and to a +- **PyPI publishing.** Premature, and the name is taken: `plumbline` on PyPI is + an unrelated project, so publishing needs a different distribution name + first. It commits the project to a name and to a release cadence before the API has settled, and the API is explicitly not stable before v0.2. The README says cloning is the install path, which is honest and costs a reader one command. Revisit when the CLI flags stop diff --git a/docs/datasets.md b/docs/datasets.md new file mode 100644 index 0000000..debe0db --- /dev/null +++ b/docs/datasets.md @@ -0,0 +1,114 @@ +# Your own data + +plumbline measures a model on your labeled rows, so this is the file format +you write them in, and the commands you run next. Put your file anywhere; the +repository's [`datasets/private/`](../datasets/private) directory is gitignored +for exactly this, so a clone never commits it. + +## The format + +One JSON object per line (JSONL), UTF-8. Blank lines are skipped. Every other +line is one case. + +| field | required | what it is | +|---|---|---| +| `id` | yes | A non-empty string, unique in the file. It names the row in the artifact. | +| `text` | yes | A non-empty string: what the model is shown. | +| `labels` | yes | A list of at least two distinct strings: the options it chooses between. | +| `gold_label` | yes | The correct option. It must be one of `labels`. | +| `question_type` | no | `choice` (the default), `noul`, or `score`. See below. | +| `label_descriptions` | no | An object mapping an option to a sentence describing it. v0.1 loads these but does not yet send them to any adapter ([#39](https://github.com/TMHSDigital/plumbline/issues/39)). | + +`question_type` says what the row actually asks, and plumbline asks it that way: + +- `choice`: pick one of the options. Most classification rows are this. +- `noul`: a yes/no question. The options must be a yes/no pair (`yes`/`no` or + `true`/`false`), because which option is the yes is a fact about your data, + not something to guess. A transport that has a native yes/no question asks it + as one, and gets back one probability rather than a distribution. +- `score`: an ordinal level, such as 0 to 3. v0.1 loads these rows and excludes + them from every figure, because ordering is lost if levels are scored as + unordered options. The load summary says how many there were. + +## An example + +```jsonl +{"id": "t-1", "text": "My card was charged twice for one order.", "labels": ["billing", "shipping", "returns", "other"], "gold_label": "billing"} +{"id": "t-2", "text": "The parcel says delivered but it is not here.", "labels": ["billing", "shipping", "returns", "other"], "gold_label": "shipping"} +{"id": "t-3", "text": "Is this order eligible for a refund?", "labels": ["yes", "no"], "gold_label": "yes", "question_type": "noul"} +{"id": "t-4", "text": "How urgent is this complaint, 0 to 3?", "labels": ["0", "1", "2", "3"], "gold_label": "2", "question_type": "score"} +{"id": "t-5", "text": "Where is my refund?", "labels": ["billing", "returns"], "gold_label": "returns", "label_descriptions": {"billing": "charges and invoices", "returns": "sending an item back, and its refund"}} +``` + +## What is refused, and why + +A row that cannot be scored honestly is refused with its line number, and the +rest of the file still runs. `--strict` refuses the whole run instead. + +- A missing required field, an empty `id` or `text`, or `labels` that is not a + list of at least two distinct entries. +- A `gold_label` that is not one of the row's own `labels`. Scoring it would + mark every model wrong on that row, which reads as a model failure and is a + data error. +- A `question_type` other than the three above. An unknown type is not assumed + to be a choice. +- An `id` already used earlier in the file. +- A line that is not a JSON object. + +A file with no rows at all, or a path that does not exist, is an error rather +than an empty run. + +## Running it + +The mock needs nothing and spends nothing, so start there to check the file: + +``` +uv run plumbline run my-data.jsonl --adapter mock --results results --report results/report.md +``` + +A hosted vendor needs its key in the environment and, for a cost column, a +pricing table you filled in (see the README). `--limit` runs only the first N +cases; `--max-cases` and `--max-cost-usd` refuse the run instead of truncating +it, so they are guards rather than ways to run less: + +``` +uv run plumbline run my-data.jsonl --adapter typesafe_wire --model jev-latest \ + --limit 40 --max-cost-usd 0.50 --pricing my-pricing.json \ + --results results --report results/report.md +``` + +A local checkpoint needs the `local` extra (`uv sync --extra local`), a model +id, and the commit it is pinned to, as 40 hex characters. A branch name or a tag +can move, and everything measured would then be attributed to other weights: + +``` +uv run plumbline run my-data.jsonl --adapter local_logits \ + --model Qwen/Qwen2.5-0.5B-Instruct --revision 7ae557604adf67be50417f59c2c2f167def9a775 \ + --results results --report results/report.md +``` + +## The cascade needs two numbers only you know + +Where to set a threshold depends on what one escalation to the expensive path +costs you and what one wrong answer costs you. plumbline never defaults either, +because a made-up default would decide the threshold. Pass both, in USD: + +``` +uv run plumbline run my-data.jsonl --adapter mock --escalation-cost 0.02 --error-cost 1.00 \ + --results results --report results/report.md +``` + +The threshold is chosen on half of the scored rows and judged on the other half, +so a threshold needs at least 400 scored rows; below that the report says so +instead of printing one. + +## Reports from runs that already happened + +Every run writes an artifact to `--results`. `plumbline report` renders one +document from one or more of them, so a finished run is never repeated to get a +report, or to try different cascade costs: + +``` +uv run plumbline report results/*.json --out results/report.md \ + --escalation-cost 0.02 --error-cost 1.00 +``` diff --git a/scripts/build_site.py b/scripts/build_site.py index 5e5414b..9579a08 100644 --- a/scripts/build_site.py +++ b/scripts/build_site.py @@ -92,6 +92,13 @@ class Doc: "What plumbline is, how to run it, and what it does not do.", USING, ), + Doc( + "docs/datasets.md", + "your-data", + "Your own data", + "The row format, what is refused and why, and the commands to run next.", + USING, + ), Doc( "METHODOLOGY.md", "methodology", diff --git a/src/plumbline/adapters/local_logits.py b/src/plumbline/adapters/local_logits.py index d36f3a1..9cba931 100644 --- a/src/plumbline/adapters/local_logits.py +++ b/src/plumbline/adapters/local_logits.py @@ -311,7 +311,10 @@ def __init__(self, *, model_id: str, revision: str, device: str = "cpu") -> None except ImportError as missing: # pragma: no cover - exercised by installing extras raise PlumblineError( "the local_logits arm needs torch and transformers, which are an optional " - "dependency. Install them with: pip install 'plumbline[local]'" + "dependency. Install them with: uv sync --extra local (with pip, install " + "from the repository: pip install " + '"plumbline[local] @ git+https://github.com/TMHSDigital/plumbline"; the ' + "plumbline on PyPI is an unrelated project)" ) from missing self._torch = torch diff --git a/src/plumbline/cli.py b/src/plumbline/cli.py index b5fdb87..4fd370b 100644 --- a/src/plumbline/cli.py +++ b/src/plumbline/cli.py @@ -57,7 +57,14 @@ def run( workers: Annotated[int, typer.Option(min=1, help="Concurrent requests.")] = 8, cache_dir: Annotated[Path | None, typer.Option("--cache", help="Cache directory.")] = None, max_cost_usd: Annotated[float | None, typer.Option(help="Abort above this.")] = None, - max_cases: Annotated[int | None, typer.Option(min=1, help="Abort above this many.")] = None, + max_cases: Annotated[ + int | None, + typer.Option( + min=1, + help="Refuse the run if it covers more cases than this. A guard, not a " + "truncation: use --limit to run fewer.", + ), + ] = None, semantics: Annotated[ str | None, typer.Option(help="Override probability_semantics (mock only).") ] = None, diff --git a/tests/test_datasets.py b/tests/test_datasets.py index d4213c7..5616c5e 100644 --- a/tests/test_datasets.py +++ b/tests/test_datasets.py @@ -322,3 +322,22 @@ def test_an_unknown_question_type_is_refused_rather_than_assumed_to_be_a_choice( assert report.cases == () assert "question_type" in report.refusals[0].reason + + +def test_the_example_in_the_dataset_docs_loads_as_documented(tmp_path) -> None: + """docs/datasets.md shows the format; the format it shows must be the one the loader reads.""" + import re + from pathlib import Path + + doc = Path(__file__).resolve().parent.parent / "docs" / "datasets.md" + block = re.search(r"```jsonl\n(.*?)```", doc.read_text(encoding="utf-8"), re.S) + assert block is not None, "docs/datasets.md has no jsonl example" + path = tmp_path / "example.jsonl" + path.write_text(block.group(1), encoding="utf-8") + + report = loader.load_jsonl(path) + + assert not report.refusals + kinds = sorted(case.question_type for case in report.cases) + assert kinds == ["choice", "choice", "choice", "noul", "score"] + assert len(report.scoreable) == 4 # the score row loads and is held back