From c0f9d98cca6dc8029b8073d07cfb7c8b031e5017 Mon Sep 17 00:00:00 2001 From: TMHSDigital <154358121+TMHSDigital@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:21:27 -0400 Subject: [PATCH] docs(methodology): the no-distribution penalty is a bias, and more rows show it plainer scripts/distribution_penalty.py repeats METHODOLOGY's one-size comparison from 500 to 20,000 rows. The top-line form's leftover miscalibration stays about 0.18 on an underconfident model and 0.02 on an overconfident one at every size while the floor falls, so its ratio to the floor grows from 2.9 to 17 times. The multiclass form inverts the injected temperature exactly, landing where the same mock does with no skew at all. Part of #8: whether a monotone fit such as isotonic regression does better on a scalar-only column is still open. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 6 +++ METHODOLOGY.md | 33 ++++++++++++++ scripts/distribution_penalty.py | 79 +++++++++++++++++++++++++++++++++ 3 files changed, 118 insertions(+) create mode 100644 scripts/distribution_penalty.py diff --git a/CHANGELOG.md b/CHANGELOG.md index f3814de..17f8c01 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -317,6 +317,12 @@ different event from one that moved because it was wrong. ### Documentation +- METHODOLOGY now measures how the penalty for reporting no distribution + changes with row count (#8). `scripts/distribution_penalty.py` repeats the + comparison from 500 to 20,000 rows: the top-line form's leftover + miscalibration stays about 0.18 on an underconfident model at every size, + while the floor falls, so its ratio to the floor grows from 2.9 to 17 times. + The penalty is a bias of the answer shape, not a small-sample artifact. - The load summary said every JevBench row "is asked as a one-of-n choice", which is not true of an adapter that asks yes/no rows as yes/no questions; it now says each row is asked as the question type it states, and the diff --git a/METHODOLOGY.md b/METHODOLOGY.md index f02fe49..b98e41e 100644 --- a/METHODOLOGY.md +++ b/METHODOLOGY.md @@ -53,6 +53,39 @@ underconfident case the multiclass form returns ECE to the noise floor and the binary form leaves it at roughly five times the floor, removing less than a fifth of the miscalibration present. +#### More rows do not close the gap + +That table is one sample size, so it could have been a small-sample artifact. +It is not. `scripts/distribution_penalty.py` repeats the comparison from 500 to +20,000 rows, averaged over five seeds of the same mock (accuracy 0.75, four +options). Each cell is post-scaling ECE on the held-out half, with its ratio to +that half's floor: + +| rows | injected skew | floor | multiclass | binary | +|---|---|---|---|---| +| 500 | T = 0.5 | 0.0489 | 0.0515 (1.05x) | 0.0438 (0.98x) | +| 2,000 | T = 0.5 | 0.0251 | 0.0306 (1.22x) | 0.0266 (1.18x) | +| 5,000 | T = 0.5 | 0.0158 | 0.0181 (1.14x) | 0.0238 (1.66x) | +| 20,000 | T = 0.5 | 0.0079 | 0.0113 (1.43x) | 0.0210 (2.92x) | +| 500 | T = 2.0 | 0.0489 | 0.0515 (1.05x) | 0.1815 (2.88x) | +| 2,000 | T = 2.0 | 0.0251 | 0.0306 (1.22x) | 0.1836 (5.78x) | +| 5,000 | T = 2.0 | 0.0158 | 0.0181 (1.14x) | 0.1756 (8.80x) | +| 20,000 | T = 2.0 | 0.0079 | 0.0113 (1.43x) | 0.1756 (17.23x) | + +The binary form's leftover miscalibration does not shrink with rows: about +0.18 on the underconfident case and 0.02 on the overconfident one, at every +size. Only its ratio to the floor grows, because the floor falls as rows are +added. So the penalty is a bias of the answer shape, not noise, and the more +rows a dataset has, the more plainly a top-line-only transport shows it. On the +overconfident case it is hidden at a few hundred rows and plain by a few +thousand. + +The multiclass form lands on identical figures for both skews, and on the same +figures again with no skew injected at all. The fit inverts the injected +temperature exactly, and what remains, including the drift to 1.43x at 20,000 +rows, is the mock's own base calibration rather than anything the correction +left behind. + So an adapter that reports a probability without a distribution is penalized twice: it loses a metric, and the correction plumbline can fit for it is weaker even in the easy case. This is a consequence of the API shape rather than of the diff --git a/scripts/distribution_penalty.py b/scripts/distribution_penalty.py new file mode 100644 index 0000000..35773a1 --- /dev/null +++ b/scripts/distribution_penalty.py @@ -0,0 +1,79 @@ +"""How the penalty for reporting no distribution changes with row count (issue #8). + +METHODOLOGY measures that an adapter reporting only its top-line probability +recalibrates worse than one reporting the full distribution, even when the +miscalibration is a pure temperature. This asks whether that is a small-sample +artifact that vanishes with more rows, or a property of the answer shape. + +The mock injects a known temperature into its log probabilities. Each run is +recalibrated twice on the same rows: once with the distributions (the +multiclass form) and once with the top-line probability alone (the binary +form). It prints the floor's mean on the held-out rows, then post-scaling ECE +for each form with its ratio to that floor, averaged over seeds; a ratio of +1.0 is as calibrated as sampling allows. + + uv run python scripts/distribution_penalty.py [seeds] + +METHODOLOGY quotes the table this prints at 5 seeds. +""" + +from __future__ import annotations + +import sys + +import numpy as np + +from plumbline.adapters.mock import MockAdapter +from plumbline.metrics.recalibration import recalibrate +from plumbline.runner import execute +from plumbline.types import Case, ProbabilitySeries + +LABELS = ("billing", "returns", "shipping", "other") +SEEDS = int(sys.argv[1]) if len(sys.argv) > 1 else 5 + + +def cases(n: int, seed: int) -> list[Case]: + rng = np.random.default_rng(seed) + gold = rng.integers(0, len(LABELS), size=n) + return [ + Case(id=f"r{index}", text=f"row {seed} {index}", labels=LABELS, gold_label=LABELS[g]) + for index, g in enumerate(gold) + ] + + +def scaled(n: int, temperature: float, seed: int, multiclass: bool) -> tuple[float, float]: + rows = cases(n, seed) + adapter = MockAdapter( + {case.text: case.gold_label for case in rows}, + accuracy=0.75, + calibration_temperature=temperature, + seed=seed, + ) + result = execute.run(adapter, rows, workers=1) + fit = recalibrate( + ProbabilitySeries( + values=tuple(p for p in result.probabilities().values if p is not None), + semantics="calibrated_claim", + ), + result.outcomes, + distributions=result.distributions() if multiclass else None, + gold_labels=result.gold_labels() if multiclass else None, + seed=seed, + n_boot_ci=50, + n_boot_floor=300, + ) + return fit.after.ece, fit.floor["ece"].mean + + +print(f"{'rows':>6} {'injected':>9} {'floor':>7} {'multiclass':>17} {'binary':>17}") +for temperature, name in ((0.5, "T = 0.5"), (2.0, "T = 2.0")): + for n in (500, 2000, 5000, 20000): + multi = np.array([scaled(n, temperature, seed, True) for seed in range(SEEDS)]) + binary = np.array([scaled(n, temperature, seed, False) for seed in range(SEEDS)]) + floor = multi[:, 1].mean() + print( + f"{n:>6} {name:>9} {floor:>7.4f} " + f"{multi[:, 0].mean():>8.4f} ({(multi[:, 0] / multi[:, 1]).mean():>5.2f}x) " + f"{binary[:, 0].mean():>8.4f} ({(binary[:, 0] / binary[:, 1]).mean():>5.2f}x)", + flush=True, + )