From 63098452ec62d17c01b81b50fa79d2bf02b2f73b Mon Sep 17 00:00:00 2001 From: TMHSDigital <154358121+TMHSDigital@users.noreply.github.com> Date: Fri, 25 Sep 2026 19:21:09 -0400 Subject: [PATCH] feat: score ordinal rows by rank, each figure against its own null Score rows were loaded and excluded from every figure, because every metric here is rank-blind. They are now read by three rank-aware figures in a block of their own, never beside a choice figure: - mean absolute error of the expected score, in levels, against a permutation null of the arm's own answers shuffled across rows; - the ranked probability score; - a cumulative calibration error: at each threshold, the predicted probability of being at or below it against how often it was, pooled and binned as ECE is. The last two are read against a calibrated-model floor built by redrawing each row's level from its own distribution. METHODOLOGY is the design. typesafe_wire asks a real Score with the row's rubric, which the JevBench loader now keeps as the levels' descriptions, reads the vendor's expected score as given, and adds its instructions to the cache key only when they change, so existing keys stand. A score row whose options are not integer levels is refused. RunResult's choice accessors read choice rows only, which kept the site's worked example on its 105 predictions. Fixes #6. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 13 + METHODOLOGY.md | 72 ++++- README.md | 12 +- datasets/public/README.md | 5 +- docs/PLAN.md | 11 +- docs/datasets.md | 8 +- docs/example-report.md | 22 +- site/index.html | 2 +- src/plumbline/adapters/typesafe_wire.py | 89 ++++++- src/plumbline/datasets/loader.py | 37 ++- src/plumbline/metrics/ordinal.py | 340 ++++++++++++++++++++++++ src/plumbline/report/markdown.py | 103 ++++++- src/plumbline/runner/execute.py | 24 +- src/plumbline/types.py | 20 +- tests/test_datasets.py | 47 +++- tests/test_ordinal.py | 171 ++++++++++++ tests/test_pipeline_smoke.py | 11 +- tests/test_typesafe_wire.py | 69 ++++- 18 files changed, 975 insertions(+), 81 deletions(-) create mode 100644 src/plumbline/metrics/ordinal.py create mode 100644 tests/test_ordinal.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 8adab8d..a1667b3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,6 +13,19 @@ different event from one that moved because it was wrong. ### Added +- Ordinal score rows are scored (#6). Each is read by three rank-aware + figures, in a score block of its own and never beside a choice figure: mean + absolute error of the expected score in levels, against a permutation null; + the ranked probability score; and a cumulative calibration error, the + predicted probability of being at or below each threshold against how often + it was, pooled and binned as ECE is. The last two are read against a + calibrated-model floor, built by redrawing each row's level from its own + distribution. `typesafe_wire` asks a real Score with the row's rubric, which + the JevBench loader now keeps as the levels' descriptions; the other arms + answer the levels as options and say so. A score row whose options are not + integer levels is refused. On the public fixture the six score rows now run, + so its report covers 111 rows, with the choice figures unchanged over 105. + METHODOLOGY's "Ordinal score questions are scored by rank" is the design. - A temperature per predicted label, tried only when the global fit is refused as the wrong shape or stops short of the floor (#4). It uses the global fit's split and verdict rule, leaves a label with under 100 fit rows as it came and diff --git a/METHODOLOGY.md b/METHODOLOGY.md index 2367c6b..0b4f93e 100644 --- a/METHODOLOGY.md +++ b/METHODOLOGY.md @@ -186,15 +186,73 @@ Confidence on a noul is not reported, and the report says so in those words. A blank cell would suggest the vendor failed to send something; the statistic does not exist for an answer with no distribution. -## Ordinal score questions are not scored in v0.1 +## Ordinal score questions are scored by rank Some datasets ask for a level rather than a label: 0, 1, 2, or 3 daily-rest -violations. The levels are ordered, and every metric here is rank-blind: being -wrong by one level and wrong by three score identically. Flattening the levels -into unordered options would discard exactly the structure that makes the -question a score, so plumbline loads those rows, marks them, and leaves them out -of every figure. The load report and the run report both say how many were held -back. Ordinal support is a v0.2 question, not a formatting one. +violations. The levels are ordered, and every choice metric here is rank-blind: +being wrong by one level and wrong by three score identically, and a +distribution piled on the levels beside the answer is treated no differently +from one spread to both ends. So a score row is never scored as a choice. It is +asked as a score, answered as a distribution over its levels, and read by three +figures of its own, in a block of its own, never beside a choice figure. + +### What an answer is + +A score answer is a probability for each level, from 0 to K minus 1. Its point +answer is the expected score, the probability-weighted mean of the levels, +which can fall between two of them. That is the answer a score vendor returns +and the one a caller would act on, so it is read as given, never replaced by the +most probable level. An arm that returns a single level and no distribution has +that level as its expected score, and only the first figure below applies to it. + +### The three figures + +**Mean absolute error of the expected score, in levels.** The rank-aware +accuracy: an answer one level off costs 1, three levels off costs 3. Lower is +better. It is read against a permutation null: the arm's own expected scores +shuffled across the rows, which keeps how the arm answers and breaks the link to +what each row's level was. An arm clears the null only when its answers carry +information about the gold level. A uniform guess would be the wrong null here, +since an arm that always answered the middle level would beat it knowing +nothing. + +**Ranked probability score.** For each threshold between two adjacent levels, +the squared gap between the predicted probability that the answer is at or below +it and whether it was, summed over the thresholds and divided by their number, +then averaged over rows. It is the ordinal counterpart of the Brier score and a +proper scoring rule: it is best in expectation when the probabilities are the +true ones, and it charges mass by its distance from the answer, so a near miss +costs less than a far one. Zero is perfect. + +**Cumulative calibration error.** What calibration means for an ordered +distribution. A distribution is calibrated when, at every threshold, the +predicted probability that the level is at or below it matches how often it is. +Each row contributes one event per threshold, the predicted cumulative +probability against whether the level was at or below the threshold, and the +events are pooled and read exactly as ECE reads a choice column: ten equal-width +bins and the count-weighted mean gap. The probability of the single most likely +level is not used, because it ignores order: it cannot tell an answer spread +over neighbouring levels from one split between the two ends. + +### The floors + +The ranked probability score and the cumulative calibration error are read +against a calibrated-model floor built the way every floor here is built. The +predicted distributions are held fixed and the gold level of each row is redrawn +from its own distribution, two thousand times, so every resample is calibrated +by construction. The spread of each figure across the resamples is the noise +floor for this exact set of distributions at this row count. A figure above the +floor's 95th percentile is distinguishable from sampling noise; one inside it is +INCONCLUSIVE, with the same wording as every other figure, because nothing was +established either way. + +### What is not done for scores + +Recalibration and the cascade are choice-question tools and are not applied to +score rows. A temperature fitted to a distribution over ordered levels is a +different correction, and a threshold on an expected score is a different +decision; each would need its own design. The score block says so. Score rows +are also left out of every choice figure, as they always were. ## A row that cannot be scored is refused, not scored diff --git a/README.md b/README.md index 14d8635..9d92f5b 100644 --- a/README.md +++ b/README.md @@ -98,7 +98,7 @@ calibrated one both land there on too few rows, and the figure does not say which you have. Taking it as a clean bill of health inverts the conclusion, and it is the easiest mistake to make with this tool. -On that same 105-row run, three of the four headline figures came back +On that same run's 105 choice and yes/no rows, three of the four headline figures came back inconclusive and only accuracy cleared its null. A tool that printed the other three alone would be handing you numbers that look like findings and are not. That refusal is the product. @@ -161,7 +161,7 @@ Expected output shape: ``` 111 rows read from datasets/public/jevbench-hard.jsonl, 111 loaded, 0 refused. ... -artifact: results/20260921T222053+0000-mock-1b96dc91.json +artifact: results/20260925T231349+0000-mock-420956a9.json report: results/report.md ``` @@ -343,10 +343,10 @@ groups side by side. Specific, and none of them are going to surprise you later. -- **Choice and Noul only.** Ordinal Score rows load, are marked, and are excluded - from every figure. Flattening ordered levels into unordered options discards - the ordering that makes them a score, so v0.1 declines rather than - approximating. +- **Score rows get three rank-aware figures and nothing else.** An ordinal level + is read by mean absolute error, the ranked probability score, and a cumulative + calibration error, each against its own null, in a block apart from the + choice figures. Recalibration and the cascade are not applied to them. - **One request per case, no batching.** Cost and latency figures are therefore conservative relative to batched use, where a single call carrying many questions against one shared state is materially cheaper and faster. diff --git a/datasets/public/README.md b/datasets/public/README.md index 24e8d57..fa587d4 100644 --- a/datasets/public/README.md +++ b/datasets/public/README.md @@ -31,8 +31,9 @@ prompts differ, and the scoring differs: by a transport with no Noul. Those are different questions, so every record carries both what the row asks and how it was asked, and the report keeps them apart rather than averaging across the difference. The six `score` rows are - loaded, marked, and excluded from every figure: v0.1 has no ordinal support, - and flattening ordered levels into unordered options discards the ordering. + read by rank, in a block apart from the choice figures, and their rubric is + kept as the levels' descriptions; flattening ordered levels into unordered + options would discard the ordering. - JevBench's score combines intelligence, calibration, speed, and cost into one number. plumbline computes its own metrics, against its own calibrated-null floor, and deliberately publishes no combined score. diff --git a/docs/PLAN.md b/docs/PLAN.md index 60699d9..33a5644 100644 --- a/docs/PLAN.md +++ b/docs/PLAN.md @@ -60,9 +60,11 @@ All three items are done. Kept here because the answers matter, not the list. ## v0.2 -- **Ordinal score questions.** The six score rows in the public fixture are - loaded, marked and excluded; scoring them needs rank-aware metrics, because - every metric here treats wrong-by-one and wrong-by-three identically. +- **Ordinal score questions.** Landed (#6): mean absolute error against a + permutation null, the ranked probability score, and a cumulative calibration + error, each against a calibrated-model floor, in a score block of their own. + `typesafe_wire` asks a real Score with the row's rubric. Recalibration and the + cascade for scores are not designed yet. - **Batching.** One request per case today, so cost and latency are both conservative relative to batched use. The vendor's own documentation describes packing many questions against one shared state in a single call, which is a @@ -105,7 +107,8 @@ Settled during the build. Reopen one only with a reason, not from scratch. - A yes/no row is asked as a Noul where the transport has one, and every record carries both what the row asks and how it was asked. A noul figure is never compared with a two-option-choice figure without that line between them. -- Ordinal score rows are loaded, marked and excluded from every figure in v0.1. +- Ordinal score rows are read by rank in a block of their own and left out of + every choice figure; before v0.2 they were excluded from every figure. - Artifacts never overwrite each other: the timestamp is only accurate to the second, so a repeated name gets a suffix rather than replacing user records. - A refused recalibration prints no number: the verdict, the split sizes, and diff --git a/docs/datasets.md b/docs/datasets.md index f328a96..a287683 100644 --- a/docs/datasets.md +++ b/docs/datasets.md @@ -26,9 +26,11 @@ Blank lines are skipped. Every other line is one case. `true`/`false`), because which option is the yes is a fact about your data, not something to guess. A transport that has a native yes/no question asks it as one, and gets back one probability rather than a distribution. -- `score`: an ordinal level, such as 0 to 3. v0.1 loads these rows and excludes - them from every figure, because ordering is lost if levels are scored as - unordered options. The load summary says how many there were. +- `score`: an ordinal level. Its `labels` are its levels, the integers 0 to + K minus 1 as strings, and a row whose options are not is refused. It is read + by rank, by three figures of its own, and never by the choice figures, which + would score wrong by one level and wrong by three the same. A level's + `label_descriptions` entry is its rubric, which a score transport sends. ## An example diff --git a/docs/example-report.md b/docs/example-report.md index ad66363..bf8a93d 100644 --- a/docs/example-report.md +++ b/docs/example-report.md @@ -48,7 +48,7 @@ Running the same command against a real vendor produces the same shape. See # plumbline report -Generated 2026-09-23 against dataset `1b96dc91`, 105 rows. 1 arm(s). +Generated 2026-09-25 against dataset `420956a9`, 111 rows. 1 arm(s). ## How to read this @@ -59,8 +59,7 @@ Generated 2026-09-23 against dataset `1b96dc91`, 105 rows. 1 arm(s). ## Dataset -- 111 rows read from datasets\public\jevbench-hard.jsonl, 111 loaded, 0 refused. Translated from JevBench: the case text is the row's question above its state, and each row is asked as the question type it states. plumbline's harness, prompts and scoring differ from JevBench's, so these numbers are not comparable with theirs. 6 rows wrote the gold label as a JSON number against string options; each was matched to the option of the same name. 38 rows carried criteria that do not describe the options one for one, so their option descriptions were dropped rather than guessed. 6 rows ask for an ordinal score. plumbline v0.1 has no ordinal support: flattening levels into unordered options discards the ordering, so they are loaded, marked, and excluded from scored results. -- 6 score rows are excluded from every figure below: plumbline v0.1 scores choice and yes/no questions only. +- 111 rows read from datasets\public\jevbench-hard.jsonl, 111 loaded, 0 refused. Translated from JevBench: the case text is the row's question above its state, and each row is asked as the question type it states. plumbline's harness, prompts and scoring differ from JevBench's, so these numbers are not comparable with theirs. 6 rows wrote the gold label as a JSON number against string options; each was matched to the option of the same name. 38 rows carried criteria that do not describe the options one for one, so their option descriptions were dropped rather than guessed. 6 rows ask for an ordinal score. They are scored by rank, in a block of their own, and left out of every choice figure, since a rank-blind figure scores wrong by one and wrong by three the same. --- @@ -72,15 +71,16 @@ The vendor asserts these probabilities are calibrated. Whether that survives con ### mock - **Model**: requested `mock-1`, reported `mock-1`. -- **Option descriptions**: 67 rows carried option descriptions, and this adapter does not send them, so they had no effect on its answers. -- **Asked**: 67 choice rows asked as choice, 38 noul rows asked as choice. +- **Option descriptions**: 73 rows carried option descriptions, and this adapter does not send them, so they had no effect on its answers. +- **Asked**: 67 choice rows asked as choice, 38 noul rows asked as choice, 6 score rows asked as choice. - 38 noul rows were asked as choice questions, which is a different question from the one the dataset states. Not comparable with an arm that asked them as noul. + - 6 score rows were asked as choice questions, which is a different question from the one the dataset states. Not comparable with an arm that asked them as score. - Accuracy 0.7714 over 105 rows, against a chance null of 0.3416 (95th percentile 0.4190): better than chance at this sample size. - ECE 0.0740 over 105 rows (10 equal width bins), against a calibrated-model floor of 0.0707 (95th percentile 0.1109): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. - Brier 0.1711 over 105 rows, against a calibrated-model floor of 0.1489 (95th percentile 0.1839): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. - **Confidence**: AUROC 0.6034 over 105 rows, against a permutation null of 0.4997 (95th percentile 0.6116): INCONCLUSIVE at this sample size. Permutation would often score this well on this many rows, so this dataset cannot tell the two apart. This is not a result in either direction. Collect more rows to make the question answerable. -- **Cost**: not reported. No cost available. None of the 105 cases could be priced, so cost is not reported rather than being shown as zero. 105 rows: tokens were reported, but the model that answered is not priced. -- **Latency**: p50 37.7ms, p95 85.0ms, p99 104.7ms over 105 live calls +- **Cost**: not reported. No cost available. None of the 111 cases could be priced, so cost is not reported rather than being shown as zero. 111 rows: tokens were reported, but the model that answered is not priced. +- **Latency**: p50 36.6ms, p95 85.0ms, p99 104.7ms over 111 live calls #### Recalibration @@ -96,3 +96,11 @@ Read these only after the figures above. MCE is a maximum over bins, decided by - MCE 0.1588 over 105 rows (10 equal width bins), against a calibrated-model floor of 0.1251 (95th percentile 0.2198): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. - Multiclass Brier 0.3526 over 105 rows, against a calibrated-model floor of 0.3168 (95th percentile 0.3975): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. + +#### Score questions + +- 6 rows ask for an ordinal level and are read by rank, by the figures below and by none of the figures above. They are not comparable with a choice figure. +- Mean absolute error 0.6746 levels over 6 rows, against a permutation null of 0.8199 (5th percentile 0.4664): INCONCLUSIVE at this sample size. The same answers shuffled across the rows would often score this well, so this dataset cannot tell whether they carry information about the level. This is not a result in either direction. Collect more rows to make the question answerable. +- Ranked probability score 0.1577 over 6 rows, against a calibrated-model floor of 0.0829 (95th percentile 0.1930): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. +- Cumulative calibration error 0.1796 over 6 rows (19 threshold events, 10 equal width bins), against a calibrated-model floor of 0.1420 (95th percentile 0.2323): INCONCLUSIVE at this sample size. A perfectly calibrated model would often score this badly on this many rows, so this dataset cannot tell the two apart. This is not a clean bill of health: nothing was established either way. Collect more rows to make the question answerable. +- Recalibration and the cascade are not applied to score rows: each would be a different correction or decision on an ordered answer, and neither is designed yet. diff --git a/site/index.html b/site/index.html index a88e8dd..9deb1fa 100644 --- a/site/index.html +++ b/site/index.html @@ -125,7 +125,7 @@

Calibrated-model floor

-

Worked example: the 105-row report

+

Worked example: the report's ECE line

The repository's example report prints this line:

Loading the example...

diff --git a/src/plumbline/adapters/typesafe_wire.py b/src/plumbline/adapters/typesafe_wire.py index ec574c6..e83b7f3 100644 --- a/src/plumbline/adapters/typesafe_wire.py +++ b/src/plumbline/adapters/typesafe_wire.py @@ -38,6 +38,7 @@ Noul, NoulAnswer, RetryPolicy, + Score, SystemOneResponse, TypeSafeClient, ) @@ -64,6 +65,11 @@ "Answer the question stated in the document. Report the probability that the answer is yes." ) +#: What a score row is asked. The row's own rubric arrives as the criteria. +DEFAULT_SCORE_INSTRUCTIONS = ( + "Answer the question stated in the document on the rubric given, one level per criterion." +) + #: The key the case text is filed under in the request state. STATE_KEY = "document" @@ -103,7 +109,7 @@ class TypeSafeWireAdapter(Adapter): without a network or a key. """ - supported_question_types = ("choice", "noul") + supported_question_types = ("choice", "noul", "score") reports_tokens = True """This transport can report token counts, so a blank cost is about the run. @@ -119,6 +125,7 @@ def __init__( model_requested: str = "jev-latest", instructions: str = DEFAULT_INSTRUCTIONS, noul_instructions: str = DEFAULT_NOUL_INSTRUCTIONS, + score_instructions: str = DEFAULT_SCORE_INSTRUCTIONS, api_key: str | None = None, base_url: str | None = None, timeout: float | None = None, @@ -131,6 +138,7 @@ def __init__( self.probability_semantics = check_probability_semantics(probability_semantics) self.instructions = instructions self.noul_instructions = noul_instructions + self.score_instructions = score_instructions # Resolved the way the SDK resolves it, so the endpoint that actually # answers is the one in the cache key and the artifact. None is the # vendor's default, which keeps existing cache entries valid. @@ -158,12 +166,18 @@ def call_params(self) -> Mapping[str, object]: authenticates the caller, it does not change the answer, and a cache key is written to disk. """ - return { + params: dict[str, object] = { "instructions": self.instructions, "noul_instructions": self.noul_instructions, "base_url": self.base_url, "question_name": QUESTION_NAME, } + # Added only when changed from the default, so every cache entry written + # before score questions existed keeps its key; a score row's key + # already differs from any other by its question type. + if self.score_instructions != DEFAULT_SCORE_INSTRUCTIONS: + params["score_instructions"] = self.score_instructions + return params #: Option descriptions become the choice's criteria. uses_label_descriptions = True @@ -186,12 +200,12 @@ def classify( if question_type == "noul": return self._classify_noul(text, labels) + if question_type == "score": + return self._classify_score(text, labels, descriptions or {}) if question_type != "choice": raise CaseRefusedError( - f"typesafe_wire does not ask {question_type!r} questions. A score question " - "asks for an ordinal level, and asking it as a choice between unordered " - "options throws the ordering away, so the case is refused rather than " - "answered as something else." + f"typesafe_wire does not ask {question_type!r} questions, so the case is " + "refused rather than answered as something else." ) described = descriptions or {} @@ -267,7 +281,68 @@ def _classify_noul(self, text: str, labels: list[str]) -> Prediction: }, ) - def _ask(self, text: str, question: Choice | Noul) -> tuple[SystemOneResponse, float]: + def _classify_score( + self, text: str, labels: list[str], descriptions: Mapping[str, str] + ) -> Prediction: + """Ask an ordinal row as a Score, with its rubric, and keep the order. + + The answer is an expected score and a probability per level. The + expected score is the vendor's answer and is kept as given; ``label`` is + the level nearest it, so the row has a level to show, and is never used + in place of it. + """ + levels = sorted(labels, key=int) + # A rubric criterion per level, in level order; a level the dataset gave + # no words for is described by its own number. + question = Score( + instructions=self.score_instructions, + criteria=[descriptions.get(level) or level for level in levels], + ) + response, latency_ms = self._ask(text, question) + try: + answer = response.scores[QUESTION_NAME] + except KeyError: + raise WireContractError( + f"no score answer named {QUESTION_NAME!r} in the response. A score was " + f"asked and something else came back. Answers present: " + f"{sorted(response.answers)!r}" + ) from None + + by_level = {str(level): float(value) for level, value in answer.probabilities.items()} + distribution = self._checked_distribution(by_level, labels) + expected = float(answer.score) + if not int(levels[0]) <= expected <= int(levels[-1]): + raise WireContractError( + f"the API reported an expected score of {expected!r}, outside the levels " + f"{levels[0]} to {levels[-1]}. Nothing is clamped." + ) + nearest = min(levels, key=lambda level: (abs(int(level) - expected), int(level))) + + usage = response.usage + return Prediction( + label=nearest, + prob_selected=distribution[nearest], + distribution=distribution, + confidence=answer.confidence, + latency_ms=latency_ms, + cost_usd=None, + input_tokens=usage.input_tokens, + output_tokens=usage.output_tokens, + model_reported=response.model, + raw={ + "model": response.model, + "question_name": QUESTION_NAME, + "asked_as": "score", + "expected_score": expected, + "probabilities": distribution, + "usage": { + "input_tokens": usage.input_tokens, + "output_tokens": usage.output_tokens, + }, + }, + ) + + def _ask(self, text: str, question: Choice | Noul | Score) -> tuple[SystemOneResponse, float]: """One request, one question, with the latency it took.""" started = time.perf_counter() response = self._client.system_one( diff --git a/src/plumbline/datasets/loader.py b/src/plumbline/datasets/loader.py index 5f32c84..1e65220 100644 --- a/src/plumbline/datasets/loader.py +++ b/src/plumbline/datasets/loader.py @@ -221,6 +221,8 @@ def load_jevbench(path: Path | str) -> LoadReport: exactly the options. On the yes/no rows the criteria describe the statement rather than the options, and mapping them across would be a guess about someone else's file, so they are dropped and counted. + * A score row's criteria are its rubric, a list with one entry per level in + level order, and become the levels' descriptions when the counts match. """ path = Path(path) rows = _read_rows(path) @@ -274,12 +276,12 @@ def load_jevbench(path: Path | str) -> LoadReport: "one for one, so their option descriptions were dropped rather than guessed." ) - unsupported = sum(1 for case in cases if not case.is_scoreable) - if unsupported: + scores = sum(1 for case in cases if case.question_type == "score") + if scores: notes.append( - f"{unsupported} rows ask for an ordinal score. plumbline v0.1 has no ordinal " - "support: flattening levels into unordered options discards the ordering, " - "so they are loaded, marked, and excluded from scored results." + f"{scores} rows ask for an ordinal score. They are scored by rank, in a block " + "of their own, and left out of every choice figure, since a rank-blind " + "figure scores wrong by one and wrong by three the same." ) return LoadReport( @@ -413,6 +415,12 @@ def _case_from_record(record: Mapping[str, Any]) -> Case: "unrecognized type is refused rather than assumed to be a choice: asking a " "question the wrong way round is not something to guess at." ) + if question_type == "score" and not _integer_levels(list(labels)): + raise DatasetError( + f"a score row's options are its levels, the integers 0 to K minus 1, and this " + f"row's are {list(labels)!r}. It is refused rather than scored by rank against " + "an order that is not there." + ) return Case( id=case_id, @@ -444,6 +452,16 @@ def _jevbench_record( and {str(key) for key in criteria} == set(labels) ): descriptions = {str(key): str(value) for key, value in criteria.items()} + elif ( + question.get("type") == "score" + and isinstance(criteria, list) + and isinstance(labels, list) + and len(criteria) == len(labels) + and _integer_levels(labels) + ): + # A rubric lists the levels in order: its first entry describes level 0. + ordered = sorted(labels, key=int) + descriptions = {level: str(entry) for level, entry in zip(ordered, criteria, strict=True)} instructions = _string_or_none(question.get("instructions")) or "" text = _TEXT_JOIN.join(part for part in (instructions, _render_state(raw.get("state"))) if part) @@ -458,10 +476,17 @@ def _jevbench_record( "question_type": question.get("type", "choice"), }, normalized, - isinstance(criteria, Mapping) and bool(criteria) and descriptions is None, + bool(criteria) and isinstance(criteria, Mapping | list) and descriptions is None, ) +def _integer_levels(labels: list[str]) -> bool: + try: + return sorted(int(label) for label in labels) == list(range(len(labels))) + except ValueError: + return False + + def _render_state(state: Any) -> str: """A string state as written; anything else as stable JSON. diff --git a/src/plumbline/metrics/ordinal.py b/src/plumbline/metrics/ordinal.py new file mode 100644 index 0000000..96970d8 --- /dev/null +++ b/src/plumbline/metrics/ordinal.py @@ -0,0 +1,340 @@ +"""Rank-aware figures for ordinal score rows. + +A score row asks for a level, 0 to K minus 1, and the levels are ordered. Every +choice metric is rank-blind: wrong by one level and wrong by three cost the +same. The three figures here are not: + +- **Mean absolute error** of the expected score, in levels, read against a + permutation null: the arm's own answers shuffled across the rows. +- **Ranked probability score**, the ordinal counterpart of the Brier score. +- **Cumulative calibration error**: at every threshold between two levels, the + predicted probability that the level is at or below it against how often it + was, pooled and binned as ECE bins a choice column. + +The second and third are read against a calibrated-model floor built the way +every floor here is: the predicted distributions held fixed and each row's gold +level redrawn from its own distribution. METHODOLOGY, "Ordinal score questions +are scored by rank", is the design. +""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from dataclasses import dataclass + +import numpy as np +from numpy.typing import NDArray + +from plumbline.metrics.calibration import ( + DEFAULT_BINNING, + DEFAULT_N_BINS, + DEFAULT_N_BOOT, + Binning, + FloorBand, + _judgment, + bin_assignments, + is_distinguishable, +) + + +@dataclass(frozen=True) +class ScoreAnswer: + """One score row, ready to be measured.""" + + levels: tuple[int, ...] + """The row's levels, ascending.""" + probabilities: tuple[float, ...] | None + """One probability per level, in the order of ``levels``, or None when the arm + returned a level and no distribution.""" + expected: float + """The expected score: the vendor's own, or the probability-weighted mean.""" + gold: int + + +def levels_of(labels: Sequence[str]) -> tuple[int, ...]: + """A score row's options as ordered integer levels.""" + try: + levels = sorted(int(label) for label in labels) + except ValueError: + raise ValueError( + f"score levels must be integers, got {list(labels)!r}. A score row's options " + "are its levels, 0 to K minus 1." + ) from None + if len(set(levels)) != len(levels): + raise ValueError(f"score levels repeat: {list(labels)!r}") + return tuple(levels) + + +def answer_from( + labels: Sequence[str], + distribution: Mapping[str, float] | None, + label: str, + gold_label: str, + *, + expected: float | None = None, +) -> ScoreAnswer: + """A score answer from what a prediction carries. + + ``expected`` is the vendor's expected score when it returned one; it is read + as given. Otherwise it is the probability-weighted mean of the levels, or, + for an arm that returned no distribution, the level it answered. + """ + levels = levels_of(labels) + by_level = {int(option): option for option in labels} + probabilities = ( + tuple(float(distribution.get(by_level[level], 0.0)) for level in levels) + if distribution + else None + ) + if expected is None: + expected = ( + float(np.dot(probabilities, levels)) if probabilities is not None else float(int(label)) + ) + return ScoreAnswer( + levels=levels, probabilities=probabilities, expected=expected, gold=int(gold_label) + ) + + +def _with_distributions(answers: Sequence[ScoreAnswer]) -> list[ScoreAnswer]: + return [answer for answer in answers if answer.probabilities is not None] + + +def cumulative_events( + answers: Sequence[ScoreAnswer], +) -> tuple[NDArray[np.float64], NDArray[np.float64]]: + """One event per row per threshold: P(level <= threshold), and whether it was.""" + predicted: list[float] = [] + observed: list[float] = [] + for answer in _with_distributions(answers): + assert answer.probabilities is not None + cumulative = np.clip(np.cumsum(answer.probabilities)[:-1], 0.0, 1.0) + predicted.extend(float(value) for value in cumulative) + observed.extend(1.0 if answer.gold <= level else 0.0 for level in answer.levels[:-1]) + return np.asarray(predicted, dtype=np.float64), np.asarray(observed, dtype=np.float64) + + +def _event_weights(answers: Sequence[ScoreAnswer]) -> NDArray[np.float64]: + """Each event's weight in the ranked probability score: 1 / (thresholds * rows).""" + rows = _with_distributions(answers) + return np.concatenate( + [ + np.full(len(answer.levels) - 1, 1.0 / ((len(answer.levels) - 1) * len(rows))) + for answer in rows + ] + ) + + +def ranked_probability_score(answers: Sequence[ScoreAnswer]) -> float: + """Mean over rows of the thresholds' squared cumulative gaps, over their count.""" + predicted, observed = cumulative_events(answers) + if not predicted.size: + raise ValueError("no score answer carries a distribution") + return float(np.dot((predicted - observed) ** 2, _event_weights(answers))) + + +def cumulative_calibration_error( + answers: Sequence[ScoreAnswer], + n_bins: int = DEFAULT_N_BINS, + binning: Binning = DEFAULT_BINNING, +) -> float: + """ECE over the pooled threshold events: the count-weighted mean per-bin gap.""" + predicted, observed = cumulative_events(answers) + if not predicted.size: + raise ValueError("no score answer carries a distribution") + return float(_binned_gap(predicted, observed[None, :], n_bins, binning)[0]) + + +def _binned_gap( + predicted: NDArray[np.float64], + observed: NDArray[np.float64], + n_bins: int, + binning: Binning, +) -> NDArray[np.float64]: + """ECE of fixed predictions against each row of ``observed``, one per draw.""" + indices, _ = bin_assignments(predicted, n_bins, binning) + one_hot = np.zeros((predicted.size, n_bins)) + one_hot[np.arange(predicted.size), indices] = 1.0 + predicted_sums = predicted @ one_hot + gaps: NDArray[np.float64] = np.abs(observed @ one_hot - predicted_sums).sum(axis=1) + return gaps / predicted.size + + +def score_floor( + answers: Sequence[ScoreAnswer], + n_bins: int = DEFAULT_N_BINS, + binning: Binning = DEFAULT_BINNING, + n_boot: int = DEFAULT_N_BOOT, + seed: int = 0, +) -> dict[str, FloorBand]: + """What a calibrated arm scores with these distributions at this row count. + + Each row's gold level is redrawn from its own predicted distribution, so every + resample is calibrated by construction, and both figures are recomputed. + """ + rows = _with_distributions(answers) + if not rows: + raise ValueError("no score answer carries a distribution") + if n_boot < 1: + raise ValueError(f"n_boot must be at least 1, got {n_boot}") + predicted, _ = cumulative_events(rows) + rng = np.random.default_rng(seed) + uniforms = rng.random((n_boot, len(rows))) + observed = np.empty((n_boot, predicted.size)) + position = 0 + for column, answer in enumerate(rows): + assert answer.probabilities is not None + thresholds = len(answer.levels) - 1 + cumulative = np.cumsum(answer.probabilities) + drawn = np.minimum( + np.searchsorted(cumulative, uniforms[:, column], side="right"), thresholds + ) + observed[:, position : position + thresholds] = ( + drawn[:, None] <= np.arange(thresholds)[None, :] + ) + position += thresholds + rps = ((predicted[None, :] - observed) ** 2) @ _event_weights(rows) + calibration = _binned_gap(predicted, observed, n_bins, binning) + + def band(metric: str, values: NDArray[np.float64]) -> FloorBand: + return FloorBand( + metric=metric, + mean=float(values.mean()), + p95=float(np.percentile(values, 95)), + n=len(rows), + n_bins=n_bins, + n_boot=n_boot, + ) + + return {"rps": band("rps", rps), "cumulative ece": band("cumulative ece", calibration)} + + +@dataclass(frozen=True) +class ErrorFigure: + """Mean absolute error in levels, read against the permutation null.""" + + value: float + n: int + null_mean: float + null_p05: float + null_p95: float + + @property + def is_better_than_null(self) -> bool: + return self.value < self.null_p05 + + @property + def is_worse_than_null(self) -> bool: + return self.value > self.null_p95 + + def statement(self) -> str: + head = f"Mean absolute error {self.value:.4f} levels over {self.n} rows, against a " + if self.is_better_than_null: + return ( + head + f"permutation null of {self.null_mean:.4f} (5th percentile " + f"{self.null_p05:.4f}): better than the permutation null at this sample size." + ) + if self.is_worse_than_null: + return ( + head + f"permutation null of {self.null_mean:.4f} (95th percentile " + f"{self.null_p95:.4f}): worse than the permutation null at this sample size. " + "The same answers shuffled across the rows would rarely score this badly, so " + "they run against the levels, which usually means the scale is reversed " + "between the dataset and the arm." + ) + return ( + head + f"permutation null of {self.null_mean:.4f} (5th percentile " + f"{self.null_p05:.4f}): INCONCLUSIVE at this sample size. The same answers " + "shuffled across the rows would often score this well, so this dataset cannot " + "tell whether they carry information about the level. This is not a result in " + "either direction. Collect more rows to make the question answerable." + ) + + +def mean_absolute_error_figure( + answers: Sequence[ScoreAnswer], n_boot: int = DEFAULT_N_BOOT, seed: int = 0 +) -> ErrorFigure: + """How far the expected scores land from the gold levels, against shuffled answers.""" + if not answers: + raise ValueError("no score answers to measure") + expected = np.asarray([answer.expected for answer in answers], dtype=np.float64) + gold = np.asarray([answer.gold for answer in answers], dtype=np.float64) + rng = np.random.default_rng(seed) + shuffled = np.asarray( + [np.abs(rng.permutation(expected) - gold).mean() for _ in range(n_boot)], dtype=np.float64 + ) + return ErrorFigure( + value=float(np.abs(expected - gold).mean()), + n=len(answers), + null_mean=float(shuffled.mean()), + null_p05=float(np.percentile(shuffled, 5)), + null_p95=float(np.percentile(shuffled, 95)), + ) + + +@dataclass(frozen=True) +class OrdinalFigure: + """A ranked probability score or cumulative calibration error with its floor.""" + + metric: str + value: float + n: int + floor: FloorBand + detail: str = "" + + @property + def is_distinguishable(self) -> bool: + return is_distinguishable(self.value, self.floor) + + def statement(self) -> str: + name = {"rps": "Ranked probability score", "cumulative ece": "Cumulative calibration error"} + return ( + f"{name[self.metric]} {self.value:.4f} over {self.n} rows{self.detail}, against a " + f"calibrated-model floor of {self.floor.mean:.4f} (95th percentile " + f"{self.floor.p95:.4f}): {_judgment(self.value, self.floor)}" + ) + + +def score_statements( + answers: Sequence[ScoreAnswer], + n_bins: int = DEFAULT_N_BINS, + binning: Binning = DEFAULT_BINNING, + n_boot: int = DEFAULT_N_BOOT, + seed: int = 0, +) -> list[str]: + """The lines a report prints for an arm's score rows, each with its null.""" + if not answers: + return [] + lines = [mean_absolute_error_figure(answers, n_boot=n_boot, seed=seed).statement()] + rows = _with_distributions(answers) + if not rows: + lines.append( + "Ranked probability score and cumulative calibration error: not reported. This " + "arm returned no distribution over the levels, so there is nothing to score by " + "rank beyond the error above." + ) + return lines + floors = score_floor(rows, n_bins=n_bins, binning=binning, n_boot=n_boot, seed=seed + 1) + counts = {len(answer.levels) - 1 for answer in rows} + per_row = ( + f"{counts.pop()} thresholds a row" + if len(counts) == 1 + else f"{sum(len(answer.levels) - 1 for answer in rows)} threshold events" + ) + lines.append( + OrdinalFigure("rps", ranked_probability_score(rows), len(rows), floors["rps"]).statement() + ) + lines.append( + OrdinalFigure( + "cumulative ece", + cumulative_calibration_error(rows, n_bins, binning), + len(rows), + floors["cumulative ece"], + detail=f" ({per_row}, {n_bins} {binning.replace('_', ' ')} bins)", + ).statement() + ) + if len(rows) < len(answers): + lines.append( + f"{len(answers) - len(rows)} of {len(answers)} score rows returned no " + "distribution, so the two figures above cover the rest and the error covers all." + ) + return lines diff --git a/src/plumbline/report/markdown.py b/src/plumbline/report/markdown.py index 4671598..ef4f676 100644 --- a/src/plumbline/report/markdown.py +++ b/src/plumbline/report/markdown.py @@ -28,11 +28,20 @@ from datetime import UTC, date, datetime from plumbline.datasets.loader import LoadReport, LoadSummary -from plumbline.metrics import baseline, calibration, cascade, cost, latency, recalibration +from plumbline.metrics import ( + baseline, + calibration, + cascade, + cost, + latency, + ordinal, + recalibration, +) from plumbline.metrics.calibration import Binning from plumbline.metrics.cost import DEFAULT_PRICING_MAX_AGE_DAYS from plumbline.runner.execute import CaseRecord, RunResult from plumbline.types import ( + CHOICE_QUESTION_TYPES, SUPPORTED_QUESTION_TYPES, ConfidenceSeries, InsufficientDataError, @@ -231,8 +240,10 @@ def _arm(result: RunResult, options: ReportOptions, heading: str) -> list[str]: record for record in result.records if record.question_type in SUPPORTED_QUESTION_TYPES ] excluded = len(result.records) - len(scoreable) - successes = [record for record in scoreable if record.prediction is not None] - failures = [record for record in scoreable if record.prediction is None] + # Score rows are read by rank in a block of their own, never by the choice + # figures, which would score wrong by one level and wrong by three the same. + choice_rows = [record for record in scoreable if record.question_type in CHOICE_QUESTION_TYPES] + score_rows = [record for record in scoreable if record.question_type == "score"] lines = ["", f"### {heading}", "", *_provenance(result, options)] described = result.config.get("label_descriptions") or {} @@ -246,7 +257,21 @@ def _arm(result: RunResult, options: ReportOptions, heading: str) -> list[str]: f"- **Excluded**: {excluded} rows of an unsupported question type were not scored." ) lines.extend(_asked_as(scoreable)) - lines.extend(_resolution_lines(scoreable, options)) + lines.extend(_resolution_lines(choice_rows, options)) + + if choice_rows: + lines.extend(_choice_block(result, choice_rows, options)) + lines.extend(_score_block(result, score_rows, options, arm_wide=not choice_rows)) + return lines + + +def _choice_block( + result: RunResult, scoreable: Sequence[CaseRecord], options: ReportOptions +) -> list[str]: + """Every figure the choice and yes/no rows get, from accuracy to diagnostics.""" + successes = [record for record in scoreable if record.prediction is not None] + failures = [record for record in scoreable if record.prediction is None] + lines: list[str] = [] if not successes: # One shared reason is almost always an install or setup step (a missing @@ -289,7 +314,9 @@ def _arm(result: RunResult, options: ReportOptions, heading: str) -> list[str]: lines.extend(_calibration_lines(probabilities, outcomes, options)) lines.extend(_confidence_lines(successes, outcomes, options)) lines.extend(_distribution_caveat(result, successes)) - lines.extend(_cost_lines(result, scoreable, options)) + # Cost is the arm's, so it counts the score rows too: they were calls. + every_row = [r for r in result.records if r.question_type in SUPPORTED_QUESTION_TYPES] + lines.extend(_cost_lines(result, every_row, options)) lines.extend(_latency_lines(result)) fit = _fit(probabilities, successes, outcomes, options) @@ -300,6 +327,72 @@ def _arm(result: RunResult, options: ReportOptions, heading: str) -> list[str]: return lines +def _score_block( + result: RunResult, + score_rows: Sequence[CaseRecord], + options: ReportOptions, + *, + arm_wide: bool, +) -> list[str]: + """The ordinal rows' own figures, each with its null. See metrics/ordinal.py.""" + if not score_rows: + return [] + lines = [ + "", + "#### Score questions", + "", + f"- {len(score_rows)} rows ask for an ordinal level and are read by rank, by the " + "figures below and by none of the figures above. They are not comparable with a " + "choice figure.", + ] + successes = [record for record in score_rows if record.prediction is not None] + failures = len(score_rows) - len(successes) + if not successes: + lines.append( + f"- **No figures**: all {failures} score rows failed or were refused; the " + "reasons are in the artifact." + ) + return lines + + answers = [] + for record in successes: + prediction = record.prediction + assert prediction is not None + expected = prediction.raw.get("expected_score") + answers.append( + ordinal.answer_from( + record.labels, + prediction.distribution, + prediction.label, + record.gold_label, + expected=float(expected) if isinstance(expected, int | float) else None, + ) + ) + lines.extend( + f"- {statement}" + for statement in ordinal.score_statements( + answers, + n_bins=options.n_bins, + binning=options.binning, + n_boot=options.n_boot, + seed=options.seed, + ) + ) + if failures: + lines.append( + f"- **Failures**: {failures} of {len(score_rows)} score rows produced no " + "prediction and are excluded rather than scored." + ) + lines.append( + "- Recalibration and the cascade are not applied to score rows: each would be a " + "different correction or decision on an ordered answer, and neither is designed yet." + ) + if arm_wide: + lines.extend(_cost_lines(result, score_rows, options)) + lines.extend(_latency_lines(result)) + return lines + + @dataclass(frozen=True) class _Fit: """What recalibration produced, or why it produced nothing.""" diff --git a/src/plumbline/runner/execute.py b/src/plumbline/runner/execute.py index 8d41d27..3119b7b 100644 --- a/src/plumbline/runner/execute.py +++ b/src/plumbline/runner/execute.py @@ -51,6 +51,7 @@ to_prediction, ) from plumbline.types import ( + CHOICE_QUESTION_TYPES, ArtifactError, Case, CaseRefusedError, @@ -243,13 +244,24 @@ def live_calls(self) -> list[CaseRecord]: """Successful rows that actually went out, so cost and latency mean something.""" return [record for record in self.records if record.ok and not record.from_cache] + @property + def choice_successes(self) -> list[CaseRecord]: + """Successful choice and yes/no rows: what the choice figures read. + + A score row is read by the ordinal figures instead, so it is left out of + every accessor below, which together are the calibratable column. + """ + return [ + record for record in self.successes if record.question_type in CHOICE_QUESTION_TYPES + ] + @property def outcomes(self) -> list[bool]: - return [bool(record.correct) for record in self.successes] + return [bool(record.correct) for record in self.choice_successes] @property def accuracy(self) -> float | None: - successes = self.successes + successes = self.choice_successes return sum(self.outcomes) / len(successes) if successes else None def probabilities(self) -> ProbabilitySeries: @@ -257,7 +269,7 @@ def probabilities(self) -> ProbabilitySeries: return ProbabilitySeries( values=tuple( record.prediction.prob_selected - for record in self.successes + for record in self.choice_successes if record.prediction is not None ), semantics=self.probability_semantics, # type: ignore[arg-type] @@ -267,7 +279,7 @@ def confidences(self) -> ConfidenceSeries: return ConfidenceSeries( values=tuple( record.prediction.confidence - for record in self.successes + for record in self.choice_successes if record.prediction is not None ) ) @@ -275,12 +287,12 @@ def confidences(self) -> ConfidenceSeries: def distributions(self) -> list[dict[str, float] | None]: return [ record.prediction.distribution - for record in self.successes + for record in self.choice_successes if record.prediction is not None ] def gold_labels(self) -> list[str]: - return [record.gold_label for record in self.successes] + return [record.gold_label for record in self.choice_successes] def to_jsonable(self) -> dict[str, Any]: return { diff --git a/src/plumbline/types.py b/src/plumbline/types.py index 4a59221..d7d4175 100644 --- a/src/plumbline/types.py +++ b/src/plumbline/types.py @@ -68,17 +68,21 @@ calibration target in the API is the bare probability. ``score`` - An ordinal level, such as 0 to 3. plumbline v0.1 has no ordinal support: - flattening levels into unordered options throws away the ordering, and - reporting rank-blind metrics on them would be worse than reporting nothing. - Such cases are loaded, marked, and excluded from scored results. + An ordinal level, such as 0 to 3. Its options are its levels, as integers. + It is answered as a distribution over the levels and read by rank-aware + figures of its own (``metrics/ordinal.py``), never by the rank-blind choice + figures, which would score wrong by one and wrong by three the same. """ QUESTION_TYPES: tuple[QuestionType, ...] = get_args(QuestionType) -#: Question types plumbline can score in v0.1. A case outside this set is loaded -#: and carried so that nothing is silently lost, and excluded from every metric. -SUPPORTED_QUESTION_TYPES: tuple[QuestionType, ...] = ("choice", "noul") +#: Question types plumbline can score. A case outside this set is loaded and +#: carried so that nothing is silently lost, and excluded from every metric. +SUPPORTED_QUESTION_TYPES: tuple[QuestionType, ...] = ("choice", "noul", "score") + +#: The question types the choice figures read: accuracy, ECE, Brier, and the +#: rest. A score row is read by the ordinal figures instead, never by these. +CHOICE_QUESTION_TYPES: tuple[QuestionType, ...] = ("choice", "noul") #: Option names read as "yes" and as "no". A yes/no question has to be pinned to #: its two outcomes before a bare P(yes) can be attached to either of them. @@ -187,7 +191,7 @@ class Case: @property def is_scoreable(self) -> bool: - """Whether v0.1 can turn this case into a number it is willing to report.""" + """Whether plumbline can turn this case into a number it is willing to report.""" return self.question_type in SUPPORTED_QUESTION_TYPES def __post_init__(self) -> None: diff --git a/tests/test_datasets.py b/tests/test_datasets.py index da985fa..a085214 100644 --- a/tests/test_datasets.py +++ b/tests/test_datasets.py @@ -281,25 +281,52 @@ def test_choice_rows_stay_choices() -> None: assert len([case for case in report.cases if case.question_type == "choice"]) == 67 -def test_score_rows_are_loaded_and_marked_rather_than_dropped() -> None: - """Nothing is lost silently: the rows are there, and they are labeled.""" +def test_score_rows_are_loaded_as_scoreable_with_their_rubric() -> None: + """A score row is scored by rank now, and its rubric describes its levels.""" report = loader.load_jevbench(PUBLIC_FIXTURE) score_cases = [case for case in report.cases if case.question_type == "score"] assert len(score_cases) == 6 assert report.row_count == 111 - assert all(not case.is_scoreable for case in score_cases) + assert all(case.is_scoreable for case in score_cases) + roster = next(case for case in score_cases if case.id == "hard-opus-a-temporal_numeric-12") + assert roster.label_descriptions == { + "0": "No violations", + "1": "Exactly one violation", + "2": "Exactly two violations", + "3": "Three or more violations", + } -def test_score_rows_are_kept_out_of_the_scoreable_set_and_the_notes_say_why() -> None: - """Flattening ordinal levels into unordered options throws the ordering away.""" +def test_the_notes_say_score_rows_are_read_by_rank_apart_from_the_choice_figures() -> None: report = loader.load_jevbench(PUBLIC_FIXTURE) - assert len(report.scoreable) == 105 - assert all(case.question_type != "score" for case in report.scoreable) - assert report.unsupported_by_type == {"score": 6} + assert len(report.scoreable) == 111 + assert report.unsupported_by_type == {} note = " ".join(report.notes) - assert "6" in note and "ordinal" in note and "excluded" in note + assert "6 rows ask for an ordinal score" in note and "by rank" in note + + +def test_a_score_row_whose_options_are_not_levels_is_refused(tmp_path: Path) -> None: + path = tmp_path / "d.jsonl" + path.write_text( + json.dumps( + { + "id": "s1", + "text": "rate it", + "labels": ["low", "high"], + "gold_label": "low", + "question_type": "score", + } + ) + + "\n", + encoding="utf-8", + ) + + report = loader.load_jsonl(path) + + assert not report.cases + assert "levels" in str(report.refusals[0]) def test_our_own_jsonl_can_declare_the_question_type(tmp_path: Path) -> None: @@ -340,7 +367,7 @@ def test_the_example_in_the_dataset_docs_loads_as_documented(tmp_path) -> None: assert not report.refusals kinds = sorted(case.question_type for case in report.cases) assert kinds == ["choice", "choice", "choice", "noul", "score"] - assert len(report.scoreable) == 4 # the score row loads and is held back + assert len(report.scoreable) == 5 # the score row is scored by rank # Validation the loaders were missing (#44, #45) diff --git a/tests/test_ordinal.py b/tests/test_ordinal.py new file mode 100644 index 0000000..1f1e9c6 --- /dev/null +++ b/tests/test_ordinal.py @@ -0,0 +1,171 @@ +"""Rank-aware figures for ordinal score rows (issue #6). + +Every choice metric is rank-blind: wrong by one level and wrong by three score +the same. These figures are not, and each is read against a null the way every +other figure is, or it is a number nobody can read. +""" + +from __future__ import annotations + +import numpy as np +import pytest + +from plumbline.metrics import ordinal +from plumbline.metrics.ordinal import ScoreAnswer + +LEVELS = (0, 1, 2, 3) + + +def point_mass(level: int, gold: int) -> ScoreAnswer: + probabilities = tuple(1.0 if value == level else 0.0 for value in LEVELS) + return ScoreAnswer(levels=LEVELS, probabilities=probabilities, expected=float(level), gold=gold) + + +def test_levels_are_read_as_ordered_integers() -> None: + assert ordinal.levels_of(("2", "0", "1", "3")) == (0, 1, 2, 3) + with pytest.raises(ValueError, match="integer"): + ordinal.levels_of(("low", "high")) + + +def test_an_answer_is_built_from_a_distribution_or_a_single_level() -> None: + from_distribution = ordinal.answer_from( + ("0", "1", "2", "3"), {"0": 0.1, "1": 0.2, "2": 0.3, "3": 0.4}, label="3", gold_label="1" + ) + assert from_distribution.expected == pytest.approx(0.2 + 0.6 + 1.2) + assert from_distribution.gold == 1 + + reported = ordinal.answer_from( + ("0", "1", "2", "3"), {"0": 0.5, "1": 0.5, "2": 0.0, "3": 0.0}, "1", "0", expected=0.5 + ) + assert reported.expected == 0.5 # the vendor's expected score is read as given + + level_only = ordinal.answer_from(("0", "1", "2"), None, label="2", gold_label="0") + assert level_only.probabilities is None and level_only.expected == 2.0 + + +def test_the_ranked_probability_score_charges_by_distance() -> None: + exact = ordinal.ranked_probability_score([point_mass(0, gold=0)]) + near = ordinal.ranked_probability_score([point_mass(1, gold=0)]) + far = ordinal.ranked_probability_score([point_mass(3, gold=0)]) + + assert exact == 0.0 + assert near == pytest.approx(1 / 3) + assert far == pytest.approx(1.0) + + +def test_each_row_is_one_event_per_threshold() -> None: + answers = [point_mass(1, gold=2), point_mass(2, gold=2)] + + predicted, observed = ordinal.cumulative_events(answers) + + assert len(predicted) == len(observed) == 2 * (len(LEVELS) - 1) + # Row one: all mass on level 1, so P(level <= k) is 0, 1, 1; gold 2 gives 0, 0, 1. + assert list(predicted[:3]) == [0.0, 1.0, 1.0] + assert list(observed[:3]) == [0.0, 0.0, 1.0] + + +def draws(n: int, seed: int, *, sharpen: float = 1.0) -> list[ScoreAnswer]: + """Answers whose gold is drawn from the true distribution; ``sharpen`` above 1 + makes the stated distribution overconfident relative to it.""" + rng = np.random.default_rng(seed) + answers = [] + for _ in range(n): + truth = rng.dirichlet(np.full(len(LEVELS), 0.8)) + gold = int(rng.choice(len(LEVELS), p=truth)) + stated = truth**sharpen + stated = stated / stated.sum() + expected = float(np.dot(stated, LEVELS)) + answers.append( + ScoreAnswer(levels=LEVELS, probabilities=tuple(stated), expected=expected, gold=gold) + ) + return answers + + +def test_a_calibrated_arm_clears_the_floors_about_one_time_in_twenty() -> None: + # A floor's 95th percentile is exceeded by a calibrated arm one time in + # twenty by design, so one seed proves nothing; the rate across seeds does. + over = {"rps": 0, "cumulative ece": 0} + for seed in range(20): + answers = draws(1000, seed=seed) + floors = ordinal.score_floor(answers, n_boot=300, seed=seed) + over["rps"] += ordinal.ranked_probability_score(answers) > floors["rps"].p95 + over["cumulative ece"] += ( + ordinal.cumulative_calibration_error(answers) > floors["cumulative ece"].p95 + ) + + assert over["rps"] <= 4, over + assert over["cumulative ece"] <= 4, over + + +def test_an_overconfident_arm_clears_both_floors() -> None: + answers = draws(2000, seed=1, sharpen=3.0) + + floors = ordinal.score_floor(answers, n_boot=400, seed=0) + + assert ordinal.ranked_probability_score(answers) > floors["rps"].p95 + assert ordinal.cumulative_calibration_error(answers) > floors["cumulative ece"].p95 + + +def test_an_informative_arm_beats_the_permutation_null_and_a_constant_one_does_not() -> None: + informative = draws(500, seed=2, sharpen=1.0) + middle = [ + ScoreAnswer(levels=a.levels, probabilities=a.probabilities, expected=1.5, gold=a.gold) + for a in informative + ] + + beats = ordinal.mean_absolute_error_figure(informative, n_boot=400, seed=0) + constant = ordinal.mean_absolute_error_figure(middle, n_boot=400, seed=0) + + assert beats.is_better_than_null + assert "better than the permutation null" in beats.statement() + assert not constant.is_better_than_null + assert "INCONCLUSIVE" in constant.statement() + + +def test_the_statements_carry_the_rows_and_the_floor() -> None: + lines = ordinal.score_statements(draws(300, seed=3), n_boot=200, seed=0) + + assert len(lines) == 3 + assert lines[0].startswith("Mean absolute error ") + assert lines[1].startswith("Ranked probability score ") + assert lines[2].startswith("Cumulative calibration error ") + assert all("over 300 rows" in line for line in lines) + assert "3 thresholds a row" in lines[2] + + +def test_an_arm_with_no_distributions_gets_the_error_only_and_says_why() -> None: + levels_only = [ + ScoreAnswer(levels=LEVELS, probabilities=None, expected=float(a.gold), gold=a.gold) + for a in draws(50, seed=4) + ] + + lines = ordinal.score_statements(levels_only, n_boot=100, seed=0) + + assert lines[0].startswith("Mean absolute error ") + assert len(lines) == 2 and "no distribution" in lines[1] + + +def test_the_report_reads_score_rows_by_rank_and_keeps_them_out_of_the_choice_figures( + tmp_path, +) -> None: + from pathlib import Path + + from typer.testing import CliRunner + + from plumbline import cli + + fixture = Path(__file__).resolve().parent.parent / "datasets/public/jevbench-hard.jsonl" + done = CliRunner().invoke( + cli.app, + ["run", str(fixture), "--format", "jevbench", "--results", str(tmp_path), "--boot", "100"], + catch_exceptions=False, + ) + + assert done.exit_code == 0, done.stderr + report = done.stdout + choice, score = report.split("#### Score questions", 1) + assert "ECE 0.0740 over 105 rows" in choice # the choice figures did not move + assert "Mean absolute error" not in choice + for figure in ("Mean absolute error", "Ranked probability score", "Cumulative calibration"): + assert f"{figure}" in score and "over 6 rows" in score + assert "not comparable with a choice figure" in score diff --git a/tests/test_pipeline_smoke.py b/tests/test_pipeline_smoke.py index 488c065..020f36a 100644 --- a/tests/test_pipeline_smoke.py +++ b/tests/test_pipeline_smoke.py @@ -41,11 +41,12 @@ def smoked(smoke: ModuleType, tmp_path_factory: pytest.TempPathFactory): def test_every_scoreable_row_of_the_public_fixture_reaches_the_runner(smoked) -> None: - """All 111 rows load; the 6 ordinal ones are held back rather than scored.""" + """All 111 rows load and all 111 run; the 6 ordinal ones are scored by rank.""" assert smoked.load.rows_read == 111 assert smoked.load.row_count == 111 - assert smoked.load.unsupported_by_type == {"score": 6} - assert len(smoked.result.records) == 105 + assert smoked.load.unsupported_by_type == {} + assert len(smoked.result.records) == 111 + assert sum(1 for record in smoked.result.records if record.question_type == "score") == 6 assert not smoked.result.failures @@ -83,9 +84,9 @@ def test_the_summary_refuses_to_let_a_mock_be_read_as_a_result(smoked) -> None: def test_the_artifact_lands_on_disk_with_the_rows_it_covered(smoked) -> None: stored = json.loads(smoked.artifact.read_text(encoding="utf-8")) - assert stored["dataset_rows"] == 105 + assert stored["dataset_rows"] == 111 assert stored["dataset_hash"] - assert len(stored["records"]) == 105 + assert len(stored["records"]) == 111 assert stored["probability_semantics"] == "calibrated_claim" diff --git a/tests/test_typesafe_wire.py b/tests/test_typesafe_wire.py index 787a942..f718ad5 100644 --- a/tests/test_typesafe_wire.py +++ b/tests/test_typesafe_wire.py @@ -12,7 +12,16 @@ from typing import Any import pytest -from typesafe_sdk import Choice, ChoiceAnswer, Noul, NoulAnswer, SystemOneResponse, Usage +from typesafe_sdk import ( + Choice, + ChoiceAnswer, + Noul, + NoulAnswer, + Score, + ScoreAnswer, + SystemOneResponse, + Usage, +) from plumbline.adapters import registry from plumbline.adapters.typesafe_wire import ( @@ -384,9 +393,61 @@ def test_a_choice_case_is_still_asked_as_a_choice() -> None: assert prediction.raw["asked_as"] == "choice" -def test_an_unsupported_question_type_is_refused_rather_than_asked_as_a_choice() -> None: - with pytest.raises(CaseRefusedError, match="score"): - an_adapter(a_response()).classify("a roster", ["0", "1", "2"], question_type="score") +def a_score_response( + *, score: float = 1.4, probabilities: dict[int, float] | None = None +) -> SystemOneResponse: + return SystemOneResponse( + model="jev-1.2", + usage=Usage(input_tokens=90, output_tokens=8), + answers={ + QUESTION_NAME: ScoreAnswer( + score=score, + confidence=0.6, + legend={0: "none", 1: "one", 2: "two"}, + probabilities=probabilities or {0: 0.1, 1: 0.4, 2: 0.5}, + ) + }, + ) + + +def test_a_score_row_is_asked_as_a_score_with_its_rubric_in_level_order() -> None: + client = FakeClient(a_score_response()) + adapter = TypeSafeWireAdapter(client=client) # type: ignore[arg-type] + + prediction = adapter.classify( + "a roster", + ["2", "0", "1"], + question_type="score", + descriptions={"0": "No violations", "1": "One violation", "2": "Two or more"}, + ) + + question = client.calls[0]["questions"][QUESTION_NAME] + assert isinstance(question, Score) + assert list(question.criteria) == ["No violations", "One violation", "Two or more"] + assert prediction.raw["asked_as"] == "score" + assert prediction.raw["expected_score"] == 1.4 + assert prediction.distribution == {"0": 0.1, "1": 0.4, "2": 0.5} + assert prediction.label == "1" # the level nearest the expected score, not the argmax + + +def test_a_score_answer_over_other_levels_is_a_contract_failure() -> None: + adapter = an_adapter(a_score_response(probabilities={0: 0.5, 1: 0.5})) + + with pytest.raises(WireContractError, match="different label set"): + adapter.classify("a roster", ["0", "1", "2"], question_type="score") + + +def test_an_expected_score_outside_the_levels_is_not_clamped() -> None: + adapter = an_adapter(a_score_response(score=2.5)) + + with pytest.raises(WireContractError, match="outside the levels"): + adapter.classify("a roster", ["0", "1", "2"], question_type="score") + + +def test_score_support_leaves_every_existing_cache_key_alone() -> None: + assert ( + "score_instructions" not in TypeSafeWireAdapter(client=FakeClient(a_response())).call_params + ) # type: ignore[arg-type] def test_the_sdk_client_is_built_with_its_own_retries_off(monkeypatch: pytest.MonkeyPatch) -> None: