diff --git a/.env.example b/.env.example index 927cf87..c4cdcd5 100644 --- a/.env.example +++ b/.env.example @@ -1,6 +1,10 @@ -# Copy to .env and fill in. .env is gitignored. plumbline reads keys from the -# environment only, never from a config file, and never writes a key to a -# results artifact or a log line. +# The environment variables plumbline reads. Nothing reads a .env file: export +# these in your shell (below), or load this file with your own tooling. plumbline +# reads keys from the environment only, never from a config file, and never +# writes a key to a results artifact or a log line. +# +# bash or zsh, current shell only: +# export TYPESAFE_API_KEY="ts-..." # # PowerShell, current session only: # $env:TYPESAFE_API_KEY = "ts-..." @@ -17,11 +21,11 @@ TYPESAFE_API_KEY= # same for the generative arm. TYPESAFE_BASE_URL= -# Generative control arm. Only needed if you run an Anthropic adapter. +# Generative control arm, which speaks Anthropic's Messages API. Only needed if +# you run the generative adapter. ANTHROPIC_BASE_URL points it at a proxy or a +# compatible endpoint, and like TYPESAFE_BASE_URL is recorded and keyed on. ANTHROPIC_API_KEY= - -# Generative control arm. Only needed if you run an Ollama adapter. -OLLAMA_BASE_URL=http://localhost:11434 +ANTHROPIC_BASE_URL= # Local checkpoints. Only needed with the "local" extra installed. HF_HOME= diff --git a/.github/ISSUE_TEMPLATE/new_adapter_config.yml b/.github/ISSUE_TEMPLATE/new_adapter_config.yml index dc99928..0baa7ae 100644 --- a/.github/ISSUE_TEMPLATE/new_adapter_config.yml +++ b/.github/ISSUE_TEMPLATE/new_adapter_config.yml @@ -1,6 +1,6 @@ name: New vendor or model description: You want plumbline to run against something it does not run against yet. -labels: ["adapter-config"] +labels: ["adapter"] body: - type: markdown attributes: @@ -16,8 +16,8 @@ body: all run on the unchanged adapter. - Any **open-weights checkpoint** is a HuggingFace model id and a pinned revision for `local_logits`. - - Any **chat model** you want as a control arm is a model string for - `generative`. + - Any **Anthropic model** you want as a control arm is a model string + for `generative`, which speaks Anthropic's Messages API only. If one of those fits, you may not need an issue at all. Try it and tell us if it worked, which is still useful. diff --git a/.github/ISSUE_TEMPLATE/site_bug.yml b/.github/ISSUE_TEMPLATE/site_bug.yml new file mode 100644 index 0000000..6974e39 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/site_bug.yml @@ -0,0 +1,35 @@ +name: Site bug +description: Something on tmhsdigital.github.io/plumbline looks wrong or does not work. +labels: ["bug", "site"] +body: + - type: input + id: page + attributes: + label: Page address + description: The full address, including anything after ? or #, since the calculator keeps its inputs there. + placeholder: https://tmhsdigital.github.io/plumbline/?n=500&bins=10&accuracy=0.8#calculator + validations: + required: true + + - type: textarea + id: what-happened + attributes: + label: What happened + description: What the page showed or did, and what you expected instead. The exact text of any message helps. + validations: + required: true + + - type: input + id: browser + attributes: + label: Browser and device + placeholder: Firefox 131 on Windows 11, or Safari on iPhone + validations: + required: true + + - type: textarea + id: console + attributes: + label: Console errors + description: Optional. Anything red in the browser's developer console (F12), pasted as text. + render: text diff --git a/CHANGELOG.md b/CHANGELOG.md index 0d787bc..e3faf7f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -185,6 +185,21 @@ different event from one that moved because it was wrong. ### Documentation +- The load summary said every JevBench row "is asked as a one-of-n choice", + which is not true of an adapter that asks yes/no rows as yes/no questions; + it now says each row is asked as the question type it states, and the + report's "3 choice asked as failed" reads "3 choice rows failed" (#62). + **This changes report text, not any number.** +- CONTRIBUTING said "all three" above four commands and named only the test + jobs as the gate; it now lists every required check and has a section on + working on the site (#63). PLAN no longer lists finished work as to do (#64). + `.env.example` no longer claims a `.env` file is read or lists an Ollama + adapter that does not exist, and the docs describe `generative` as the + Anthropic Messages API it is (#65). The CLI's help says what the tool is and + lists each option's choices, and `plumbline adapters` marks an adapter whose + optional extra is missing (#66). The adapter template applies a label that + exists, and there is a template for site bugs (#67). + - A new page, [Your own data](docs/datasets.md), gives the row format, what the loader refuses and why, and the commands a user needs next: a local checkpoint with its pinned revision, the cascade's two costs, and diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 5344362..d4eace9 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -15,8 +15,9 @@ one of them. Adding a vendor should not add a file. - **Any open-weights checkpoint is a config entry for `local_logits`**, as a HuggingFace model id and a pinned revision. Pin the revision. A moving checkpoint makes every stored result unreproducible. -- **Any chat model you want as a control arm is a model string for - `generative`.** +- **Any Anthropic model you want as a control arm is a model string for + `generative`.** It speaks Anthropic's Messages API only, so another provider's + chat model is a new transport, not a config entry. A new adapter module is for a genuinely new *transport*, which is rare. If you are writing one, say in the issue what wire shape it speaks and why none of the @@ -49,7 +50,7 @@ and the reason, not a value with a caveat attached. ## Before you open a pull request -All three must pass: +All four must pass: ```powershell uv run pytest @@ -79,14 +80,17 @@ The flow: 4. **Open a pull request.** No approval is required, because there is currently one maintainer and a rule demanding one would only demand it of them. CI is the gate that actually matters. -5. **CI must be green** before merge: ruff, ruff format, mypy --strict and - pytest, on Python 3.12 and 3.13, on Ubuntu and Windows. All four jobs are - required checks. +5. **CI must be green** before merge. The required checks are the four test + jobs (ruff, ruff format, mypy --strict and pytest, on Python 3.12 and 3.13, + on Ubuntu and Windows), the quickstart as the README documents it, the prose + check (no em dashes, no `--` used as a dash), the built wheel, and the site's + two checks (the floor agrees with the Python; every link, anchor, meta tag + and policy resolves, and the pages work in a real browser). 6. **Squash on merge.** The branch is deleted automatically afterwards. **Some checks are advisory and do not gate a merge.** CodeQL and Socket Security both report on pull requests, and neither is a required check. The -four CI jobs are the gate. This is deliberate, not an oversight: a +checks listed above are the gate. This is deliberate, not an oversight: a supply-chain advisory is a judgement call that a human should make, and a scanner that can block a merge on a false positive ends up being routed around rather than read. @@ -138,3 +142,29 @@ real labeled data goes. Run artifacts in `results/` are gitignored too. They contain per-case records from your own data, and on a live run, from a vendor call. Do not paste one into an issue without reading it first. + +## Working on the site + +The site at is built from `site/` +and the repository's markdown by `scripts/build_site.py`, and deployed by +`.github/workflows/site.yml` from `main` only. It needs Node 22 or later on +the path (for its built-in WebSocket) and, for the browser check, Chrome. +Nothing is installed by npm. + +``` +uv run python scripts/build_site.py --out _site +node scripts/check_floor_parity.mjs _site/example-run.json +node scripts/check_site_links.mjs _site +node scripts/check_search.mjs _site +node scripts/smoke_site.mjs _site +``` + +- `site/floor.js` is a JavaScript port of the Python floor, held to it within + 1e-9. After changing the floor in Python, run + `uv run python scripts/floor_golden.py` to regenerate + `site/floor-golden.json`, and commit both; CI refuses a stale fixture. +- `docs/example-report.md` is checked line for line against the command it + records. After changing any report wording, rerun that command and replace + everything from `# plumbline report` down; the build refuses otherwise. +- Every page carries a Content-Security-Policy that allows only the site's own + files, so no inline script, inline style block, or `style=` attribute. diff --git a/datasets/public/README.md b/datasets/public/README.md index a1386f1..24e8d57 100644 --- a/datasets/public/README.md +++ b/datasets/public/README.md @@ -48,7 +48,7 @@ rather than living only here. ### Running it ``` -python examples/smoke_public_dataset.py +uv run python examples/smoke_public_dataset.py ``` Loads this file, runs the mock adapter over it, computes the metrics, and writes diff --git a/docs/PLAN.md b/docs/PLAN.md index 38b1816..8c556a5 100644 --- a/docs/PLAN.md +++ b/docs/PLAN.md @@ -44,8 +44,9 @@ All three items are done. Kept here because the answers matter, not the list. 2. **Live call made**, 2026-09-21. See "Live validation" below. Both open questions are settled, and the call found two bugs that had never been exercised. -3. **Decision on going public** is the human's, and the checklist at the bottom - of this page is what is left to do. +3. **Decision on going public**: made. The repository is public, the v0.1.0 + tag is pushed, and the checklist at the bottom of this page records what was + done. ## v0.2 @@ -88,7 +89,7 @@ Settled during the build. Reopen one only with a reason, not from scratch. - Every calibration figure carries its row count, and the artifact records `dataset_rows` beside `dataset_hash`, because the floor depends on n. - The JevBench public rows are vendored as a fixture under MIT with attribution. - They are asked as one-of-n choices through plumbline's own harness, so results + Each is asked as the question type it states, through plumbline's own harness, so results from them are never comparable with JevBench's published numbers. - A yes/no row is asked as a Noul where the transport has one, and every record carries both what the row asks and how it was asked. A noul figure is never @@ -437,21 +438,18 @@ makes. - Whether the residual in the confidence relationship (max 0.0167) is purely wire rounding or a slightly different production formula. Not worth another spend to settle, and nothing in plumbline depends on the answer. -- `datasets/private/.gitkeep` exists on disk but is not tracked, because the - `datasets/private/` ignore rule matches it. A fresh clone therefore has no - such directory even though the README names it as where your own data goes. - Harmless, and fixing it means `datasets/private/*` plus a negation, which - changes the ignore semantics of a data directory. Left alone deliberately - rather than changed on the way out the door. +- ~~`datasets/private/.gitkeep` is not tracked.~~ Resolved: `.gitignore` now + ignores `datasets/private/*` with a negation for `.gitkeep`, so a fresh clone + has the directory the README names and still never commits what goes in it. ## Repo description and topics For the GitHub About box. Paste as is. -**Description** (101 characters): +**Description** (108 characters, as set): ``` -Measure whether a decision model's probabilities hold up on your own labeled data. Not a leaderboard. +Measure whether a decision model's probabilities are trustworthy on your own labeled data. Not a leaderboard. ``` **Topics:** @@ -496,17 +494,10 @@ for you. - [x] History rewritten to remove the vendor price claim, and re-scanned after. The MIT fixture and licence are byte-identical before and after. -**Yours to do:** - -1. **Read the README yourself, once, as a stranger.** It is the whole public - interface and it was written by someone who already knew the answer. -2. **Push the tag** if you are happy with it: `git push origin v0.1.0`. It is - tagged locally and deliberately not pushed. -3. **Flip the repository public.** Not done here, by instruction. -4. **Set the About box** from the description and topics above. -5. **Email TypeSafe** about 16.4 and 14.1 (see "Open questions"). A written yes - converts the mock example report into a real vendor one and would let the - shipped pricing table carry figures again. Not a blocker for anything. -6. **Check the CI badge renders** once the repository is public. A badge - pointing at a private repository's workflow shows as unknown to logged-out - readers. +**Done since:** the v0.1.0 tag is pushed, the repository is public, the About +box carries the description above, and the CI badge renders for logged-out +readers. + +**Still open:** email TypeSafe about 16.4 and 14.1 (see "Open questions"). A +written yes converts the mock example report into a real vendor one and would +let the shipped pricing table carry figures again. Not a blocker for anything. diff --git a/docs/example-report.md b/docs/example-report.md index af3590c..723c4fd 100644 --- a/docs/example-report.md +++ b/docs/example-report.md @@ -48,7 +48,7 @@ Running the same command against a real vendor produces the same shape. See # plumbline report -Generated 2026-09-22 against dataset `c18e9496`, 105 rows. 1 arm(s). +Generated 2026-09-23 against dataset `c18e9496`, 105 rows. 1 arm(s). ## How to read this @@ -59,7 +59,7 @@ Generated 2026-09-22 against dataset `c18e9496`, 105 rows. 1 arm(s). ## Dataset -- 111 rows read from datasets\public\jevbench-hard.jsonl, 111 loaded, 0 refused. Translated from JevBench: the case text is the row's question above its state, and every row is asked as a one-of-n choice. plumbline's harness, prompts and scoring differ from JevBench's, so these numbers are not comparable with theirs. 6 rows wrote the gold label as a JSON number against string options; each was matched to the option of the same name. 38 rows carried criteria that do not describe the options one for one, so their option descriptions were dropped rather than guessed. 6 rows ask for an ordinal score. plumbline v0.1 has no ordinal support: flattening levels into unordered options discards the ordering, so they are loaded, marked, and excluded from scored results. +- 111 rows read from datasets\public\jevbench-hard.jsonl, 111 loaded, 0 refused. Translated from JevBench: the case text is the row's question above its state, and each row is asked as the question type it states. plumbline's harness, prompts and scoring differ from JevBench's, so these numbers are not comparable with theirs. 6 rows wrote the gold label as a JSON number against string options; each was matched to the option of the same name. 38 rows carried criteria that do not describe the options one for one, so their option descriptions were dropped rather than guessed. 6 rows ask for an ordinal score. plumbline v0.1 has no ordinal support: flattening levels into unordered options discards the ordering, so they are loaded, marked, and excluded from scored results. - 6 score rows are excluded from every figure below: plumbline v0.1 scores choice and yes/no questions only. @@ -72,7 +72,7 @@ The vendor asserts these probabilities are calibrated. Whether that survives con ### mock - **Model**: requested `mock-1`, reported `mock-1`. -- **Asked**: 67 choice asked as choice, 38 noul asked as choice. +- **Asked**: 67 choice rows asked as choice, 38 noul rows asked as choice. - 38 noul rows were asked as choice questions, which is a different question from the one the dataset states. Not comparable with an arm that asked them as noul. - Accuracy 0.7714 over 105 rows, against a chance null of 0.3416 (95th percentile 0.4190): better than chance at this sample size. - ECE 0.0740 over 105 rows (10 equal width bins), against a calibrated-model floor of 0.0707 (95th percentile 0.1109): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. diff --git a/src/plumbline/cli.py b/src/plumbline/cli.py index 4fd370b..e89de55 100644 --- a/src/plumbline/cli.py +++ b/src/plumbline/cli.py @@ -16,6 +16,7 @@ import sys from collections.abc import Callable from functools import partial +from importlib.util import find_spec from pathlib import Path from typing import Annotated, NoReturn @@ -30,10 +31,17 @@ from plumbline.report import markdown from plumbline.runner import execute from plumbline.runner.cache import Cache -from plumbline.types import Case, DatasetError, PlumblineError, ProbabilitySemantics +from plumbline.types import ( + PROBABILITY_SEMANTICS, + Case, + DatasetError, + PlumblineError, + ProbabilitySemantics, +) app = typer.Typer( - help="Choose and configure a decision model on your own labeled data.", + help="Measure whether a decision model's probabilities are trustworthy on your own " + "labeled data.", no_args_is_help=True, add_completion=False, ) @@ -44,7 +52,9 @@ @app.command() def run( dataset: Annotated[Path, typer.Argument(help="JSONL file of labeled cases.")], - adapter: Annotated[str, typer.Option(help="Registered adapter name.")] = "mock", + adapter: Annotated[ + str, typer.Option(help=f"One of: {', '.join(registry.available())}.") + ] = "mock", model: Annotated[str | None, typer.Option(help="Model to request.")] = None, revision: Annotated[str | None, typer.Option(help="Pinned checkpoint commit.")] = None, results: Annotated[Path, typer.Option(help="Directory the artifact is written to.")] = Path( @@ -66,7 +76,11 @@ def run( ), ] = None, semantics: Annotated[ - str | None, typer.Option(help="Override probability_semantics (mock only).") + str | None, + typer.Option( + help="Override probability_semantics, mock only: one of " + f"{', '.join(PROBABILITY_SEMANTICS)}." + ), ] = None, seed: Annotated[int, typer.Option(help="Mock seed.")] = 7, accuracy: Annotated[float, typer.Option(help="Mock target accuracy.")] = 0.8, @@ -187,7 +201,12 @@ def report( def adapters() -> None: """List the transports this install can run.""" for name in registry.available(): - typer.echo(name) + missing = [module for module in _EXTRAS.get(name, ()) if find_spec(module) is None] + typer.echo(f"{name} (needs the local extra: uv sync --extra local)" if missing else name) + + +#: Adapters whose dependencies are an optional extra, and the modules it installs. +_EXTRAS = {"local_logits": ("torch", "transformers")} @app.command() @@ -252,8 +271,6 @@ def _build( def _semantics(value: str) -> ProbabilitySemantics: - from plumbline.types import PROBABILITY_SEMANTICS - if value not in PROBABILITY_SEMANTICS: _fail(f"--semantics must be one of {list(PROBABILITY_SEMANTICS)!r}, got {value!r}") return value diff --git a/src/plumbline/datasets/loader.py b/src/plumbline/datasets/loader.py index 3d3ddcc..a61c64c 100644 --- a/src/plumbline/datasets/loader.py +++ b/src/plumbline/datasets/loader.py @@ -156,7 +156,7 @@ def load_jevbench(path: Path | str) -> LoadReport: This is a translation, not a reproduction. JevBench asks its rows through its own harness; plumbline composes a case text from the row's question and - state, asks every row as a one-of-n choice, and scores with its own metrics. + state, asks each row as its question type, and scores with its own metrics. Numbers from here are not comparable with JevBench's published ones, and ``datasets/public/README.md`` says so in the same words. @@ -205,7 +205,8 @@ def load_jevbench(path: Path | str) -> LoadReport: notes = [ "Translated from JevBench: the case text is the row's question above its state, " - "and every row is asked as a one-of-n choice. plumbline's harness, prompts and " + "and each row is asked as the question type it states. plumbline's harness, " + "prompts and " "scoring differ from JevBench's, so these numbers are not comparable with theirs." ] if normalized: diff --git a/src/plumbline/report/markdown.py b/src/plumbline/report/markdown.py index 6ae7c1c..67327e2 100644 --- a/src/plumbline/report/markdown.py +++ b/src/plumbline/report/markdown.py @@ -511,8 +511,14 @@ def _asked_as(records: Sequence[CaseRecord]) -> list[str]: key = (record.question_type, record.asked_as) pairs[key] = pairs.get(key, 0) + 1 + def phrase(question_type: str, asked: str, count: int) -> str: + # A row that failed or was refused was not asked as anything. + if asked in {"failed", "refused"}: + return f"{count} {question_type} rows {asked}" + return f"{count} {question_type} rows asked as {asked}" + described = ", ".join( - f"{count} {question_type} asked as {asked}" + phrase(question_type, asked, count) for (question_type, asked), count in sorted(pairs.items()) ) lines = [f"- **Asked**: {described}."]