From 943ffb3e70083d32e0bdccbd5133de834ff1f690 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 16:22:30 +0000 Subject: [PATCH 1/3] Rename consolidated scheduler to v1; add consolidated_v2 v1: op_consolidated.py -> op_consolidated_v1.py, ConsolidatedV1Scheduler, strategy/flag consolidated_v1 (--consolidated kept as alias). v2: v1 posterior scored by mean + tau * max(0, draw - mean), tau 0.65. Tournament of all 41 op schedulers on 4 bandit_env environments picked the ingredients; v2 beats v1 on all four (+1.3-1.8%, 20 paired seeds, >=3 SE; v1-vs-v1 control within 2 SE). Leads no-Elo precedence. Also pins las_vegas in the fallback-precedence test; its missing ballot wiring is logged in TODO. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ape9xt5N7dHzpFqQcsNwk --- CHANGELOG.md | 5 + docs/DEEP_DIVE.md | 2 + docs/TODO.md | 4 +- src/fuzzer_tool/cli/commands.py | 23 ++- src/fuzzer_tool/core/schedulers/__init__.py | 6 +- ..._consolidated.py => op_consolidated_v1.py} | 16 +- .../core/schedulers/op_consolidated_v2.py | 134 +++++++++++++ src/fuzzer_tool/core/schedulers/op_corral.py | 4 +- src/fuzzer_tool/services/fuzzer.py | 51 +++-- src/fuzzer_tool/services/operators.py | 24 ++- tests/support/bandit_env.py | 2 +- tests/support/operator_env.py | 3 +- tests/test_bandit_env_fatigue.py | 2 +- tests/test_canary_scheduler.py | 4 +- ...r.py => test_consolidated_v1_scheduler.py} | 38 ++-- tests/test_consolidated_v2_scheduler.py | 189 ++++++++++++++++++ tests/test_corral_scheduler.py | 2 +- tests/test_regression_elo_all.py | 3 +- ...egression_scheduler_fallback_precedence.py | 19 +- ...est_regression_scheduler_operator_reach.py | 10 +- ...est_regression_track_op_effect_coverage.py | 3 +- tests/test_scheduler_convergence.py | 14 +- 22 files changed, 482 insertions(+), 76 deletions(-) rename src/fuzzer_tool/core/schedulers/{op_consolidated.py => op_consolidated_v1.py} (96%) create mode 100644 src/fuzzer_tool/core/schedulers/op_consolidated_v2.py rename tests/{test_consolidated_scheduler.py => test_consolidated_v1_scheduler.py} (83%) create mode 100644 tests/test_consolidated_v2_scheduler.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 8b9803c7..4e1b4082 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -37,6 +37,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **`--consolidated-v2`** (`core/schedulers/op_consolidated_v2.py`): consolidated v1 scored by an optimistic, tempered Thompson draw. +1.3-1.8% over v1 on all four `bandit_env` environments (20 paired seeds). Leads no-Elo precedence. - **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): per-seed ensembles of mutant hit counts; mean + SVD wall filter, CFAR χ² flow detection; flow power scales seed energy to `[1, max_mult]`. - **OS / network scheduler ports**: seed arms `mlfq`, `stride`, `eevdf`, `bfq`, `sfq`, `codel`, `aimd`, `p2c` (`--seed--scheduler`) and op arms `op_stride`, `op_p2c` (`--op-stride`, `--op-p2c`); `core/fair_queue.py` gains `Stride` and `EEVDF`. Elo arms, in `--hail-mary`; seed arms also run without `--elo`. Falsification and adversarial tests. No paired benchmark yet. - **Ten op mutators** for in-tree targets and text decoders: `json_mutate` (fuzzgoat), `sql_mutate` (sqlite SQL path), `ecdsa_field_mutate` (secp256k1), `recompress_lz4`, `recompress_png_idat` (format band, sniffer-gated); `encoding_wrap`, `escape_mutate`, `ascii_float` (structural); `utf16_transcode`, `nest_bomb` (radamsa). Tests: `tests/test_{json_mutate,sql_mutate,ecdsa_field_mutate,recompress_roundtrip,text_codec,nest_bomb}.py`. Discovery effect on fuzzgoat unmeasured. @@ -56,6 +57,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - **`core/group_testing.py`** + `tools/bench_group_testing.py`: non-adaptive pooled which-items-matter inference (COMP/DD, exact for any d) and a binary-splitting baseline. Not wired into `tmin`/colorizer (see handover section 6.5 result). - **`--second-order-blend W`** (default 0 = off): second-order operator chain `P(next | prev2, prev)` in `MonteCarloScheduler` (`core/op_chain2.py`, sparse, capped at 4096 contexts), backing off to `--pairwise-blend` on unseen contexts. Synthetic A/B only; real-target A/B not run. +### Changed + +- **`--consolidated` → `--consolidated-v1`** (`op_consolidated.py` → `op_consolidated_v1.py`, `ConsolidatedScheduler` → `ConsolidatedV1Scheduler`, strategy `consolidated` → `consolidated_v1`, stats keys `consolidated_v1_*`). `--consolidated` kept as an alias. Elo ratings saved under `consolidated` do not carry over. + ### Documentation - **Combinatorics gap analysis** (`docs/handover/handover_combinatorics_permutations_2026-09-02.md` section 6): ranked remaining candidates (FIC-style isolation over covering-array rows, covering-array extensions, orthogonal designs, rank/unrank, group testing) and what to skip. Docs only. diff --git a/docs/DEEP_DIVE.md b/docs/DEEP_DIVE.md index 1429c1c9..8593666f 100644 --- a/docs/DEEP_DIVE.md +++ b/docs/DEEP_DIVE.md @@ -220,6 +220,8 @@ For production and sensitive binaries using AFL family fuzzers is the best cours - **KL-UCB variants** (`--kl-ducb`, `--kl-ducb-gamma`, `--kl-swucb`, `--kl-swucb-window`): the same two policies with the Gaussian confidence width replaced by the empirical-Bernoulli KL upper bound (Cappé, Garivier, Maillard, Munos, Stoltz, 2013). The cost-adjusted surprisal rewards handed to `record()` are bounded in [0, 1] with a mass at zero — an operator that found no new coverage gets exactly 0 — so the true tail is heavier than the Gaussian one and the Gaussian width under-covers. The KL bound is the smallest q in [p, 1] with KL(p||q) >= xi*log(n)/N, solved by bisection in `core/schedulers/_kl_ucb.py`; it is a strict tightening of the Gaussian form, so it explores less and concentrates on the best arm. It is a separate scheduler rather than a flag on DUCB/SWUCB because the two forms are mutually exclusive and `--elo` arbitrates them as distinct strategies. `tools/measure_klucb_signal.py` runs all four side by side on a Bernoulli environment. Internally, D-UCB and KL-D-UCB share `DiscountedUCBBase` for discounted state and width computation; SW-UCB and KL-SW-UCB share `WindowedUCBBase` for windowed state — both in `core/schedulers/ucb_common.py`. - **MOSS** (`--moss`, `--moss-gamma`): Audibert & Bubeck's minimax-optimal UCB, anytime form (Degenne & Perchet 2016). The width is `c*sqrt(log+(t/(K*n_i))/n_i)`: zero once an operator holds its fair share t/K of pulls, so the policy stops re-opening every one of ~200 operators as t grows. On a 150-arm environment at fuzzing rates (0.002-0.08) it scored 3992 successes in 60k pulls against Consolidated 3928, D-UCB 756 and SW-UCB 659 (uniform ~560); the discounted/windowed UCBs keep too few pulls per arm at that K to separate good operators from bad. Undiscounted by default, so it adapts slowly to decaying yields (0.34 tail recovery on `DecayingBest`); `--moss-gamma 0.99995` raises that to 0.72 at a 22% cost on 150 arms. Numpy state, 21.5µs per select+record at K=200. Measurements in `core/schedulers/op_moss.py`. - **Bayes-UCB** (`--bayes-ucb`): Kaufmann et al. 2012. Scores an operator by the `1 - 1/(t (log t)^c)` quantile of its Beta posterior. Deterministic (no rng), `supports_priors = True` so format-operator priors bias the first quantiles. On the Elo ballot and in the fallback chain after MOSS. **Not A/B validated.** Details in `core/schedulers/op_bayes_ucb.py`. +- **Consolidated v1** (`--consolidated-v1`, alias `--consolidated`): flat Thompson over Beta evidence with a category-shrunk prior and a capped pseudocount (forgets by rescaling past 200). Only scheduler in the leading group on all four `bandit_env` environments. `core/schedulers/op_consolidated_v1.py`. +- **Consolidated v2** (`--consolidated-v2`): v1's posterior scored by `mean + tau * max(0, draw - mean)`, tau 0.65 — optimistic (a below-mean draw scores the mean) and tempered (less exploration, the edge MOSS/FPL/greedy had over v1). 20 paired seeds vs v1: +1.8% stationary, +1.6% decaying, +1.6% rotting, +1.3% Fatigue150, each ≥3 SE; a v1-vs-v1 control stays within 2 SE. +~5µs per select at K=200 (the Beta draw dominates). Leads the no-Elo precedence, ahead of v1. Synthetic only; real-target A/B pending. Tournament of all 41 operator schedulers in `core/schedulers/op_consolidated_v2.py`. - **BO-GP-UCB** (`--bo-gp-ucb`, `--bo-gp-length-scale`, `--bo-gp-noise`): Bayesian Optimization with Expected Improvement acquisition and noisy Gaussian Process posterior. EI(op) = (μ(op) - f_max)Φ(z) + σ(op)φ(z) where z = (μ(op) - f_max)/σ(op). Cholesky decomposition for stable GP posterior inference. The `noise` parameter adds σ² to the kernel diagonal, distinguishing observation noise from epistemic uncertainty — higher noise increases posterior variance and encourages exploration. Unlike UCB schedulers that balance exploration/exploitation via a β parameter, EI naturally balances both without a separate knob. `supports_priors = False` — GP posterior is computed from observations, not Beta priors. `tools/measure_bo_gp_ucb_signal.py` runs it side by side with other schedulers. - **Badness-indexed exploration floor** (`core/badness_floor.py`; `--op-katz`, `--op-kuramoto`): both schedulers floor each arm at `lambda(badness)/n`, `lambda` interpolating `explore_floor` (SUPERCRITICAL corpus) to `max_explore_floor` (SUBCRITICAL, stalled). Badness comes from `Fuzzer._current_scheduling_badness`; a raising source falls back to the static floor. - **PLL monitor** (`--pll`, `core/analyzers/analyzer_pll.py`, 2026-09-24): observation layer over `core/pll.py`. Exec times are pushed per execution (one array append), discovery-rate deltas at each stats tick; each series buffers 256 samples, bootstraps a `PhaseLockedLoop` from `detect_periodicity` (retries on a miss), and logs lock/unlock transitions with the stall-recovery flag. `stall_lift = P(stall | transition) / P(stall)`. Consumers: stats field `pll: t: d:…` and a "PLL" line per series under Spectral Diagnostics (tracked period next to the batch FFT). Read-only. Loops, counters and pending samples persist (`pll` state section; `PhaseLockedLoop.to_dict/from_dict`). fuzzgoat: exec-time series bootstraps at period ≈ 5, stays unlocked. diff --git a/docs/TODO.md b/docs/TODO.md index 583f70eb..d7a772f4 100644 --- a/docs/TODO.md +++ b/docs/TODO.md @@ -49,7 +49,9 @@ - [ ] **PLL: correlate lock transitions with stalls on real campaigns** (2026-09-24) — `--pll` logs transitions and `stall lift`; collect over ≥ 3 targets before any behavioural consumer or gain tuning. Exec-time periods near 5 samples sit outside the tuned 15-30 range. - [ ] **A/B `wfc_reorder_learned` now that it learns** (2026-09-24) — admission never called `notify_new_coverage`, so its tables were always empty in real runs and every prior `--wfc` measurement of it is void. Re-run paired `--wfc` cells on an isobmff/riff target (needs `ffmpeg_read`). - [ ] **A/B the badness floor on `--op-kuramoto`** (2026-09-24) — now wired like `--op-katz`, unmeasured. Re-run the `tests/support/bandit_env.py` lock-in harness with regime pinned SUBCRITICAL vs SUPERCRITICAL; then paired cells on fuzzgoat. -- [ ] **Re-run the scheduler tournament on `RottingArms` / `Fatigue150`** (2026-09-24) — both now in `tests/support/bandit_env.py`. `Fatigue150` is a reconstruction, not the original: `op_consolidated.py`'s 150-arm numbers are not comparable until re-measured. Then non_ucb step 3 (reward vs own pulls from the ablation CSV `operator` column). +- [x] **Re-run the scheduler tournament on `RottingArms` / `Fatigue150`** (2026-09-24, run 2026-10-01) — all 41 operator schedulers, 4 environments, 3 seeds: table in `core/schedulers/op_consolidated_v2.py`. Consolidated (now v1) leads overall; produced `--consolidated-v2`. Still open: non_ucb step 3 (reward vs own pulls from the ablation CSV `operator` column). +- [ ] **A/B `--consolidated-v2` vs `--consolidated-v1` on a real target** (2026-10-01) — +1.3-1.8% on all four synthetic environments; paired `bench_paired` cells on fuzzgoat (clang) next. Fatigue150 headroom is unlock detection (v1/v2 notice an unlock ~1500 rounds late; greedy oracle 1692 vs v2 1544); staleness decay and global discount both lost. +- [ ] **`--las-vegas` is never offered** (2026-10-01) — in `_FALLBACK_PRECEDENCE` and the dispatch chain but missing from `operator_strategy_pool()`, the `_track_op_effect` gate, `test_regression_scheduler_operator_reach` and `test_regression_track_op_effect_coverage._KWARGS`; the last two fail on main. Wire it like `--moss`. - [ ] **A/B `--op-tpe`** (2026-09-24) — BO-3 shipped unmeasured. Paired cells vs `--bo-gp-ucb` under `--elo`; sweep `gamma` (0.1/0.25/0.5) and window. - [ ] **A/B `--op-afl-det`** (2026-09-24) — T1-1 arm shipped unmeasured. Compare cold-start edge curves (first 5 min) on fuzzgoat/png with the arm on vs off; the case for it is cold start and plateaus, not steady state. - [ ] **Entropy LOO vs Shapley disagreement, then A/B `--entropy-loo`** (2026-09-24) — measure how often LOO and permutation-sampled Shapley rank seeds differently on real corpora (near-duplicate clusters are the known misprice) before building Shapley; then paired A/B. Also open: use the score to protect seeds in minimization. diff --git a/src/fuzzer_tool/cli/commands.py b/src/fuzzer_tool/cli/commands.py index 10dd92e7..d91f7457 100644 --- a/src/fuzzer_tool/cli/commands.py +++ b/src/fuzzer_tool/cli/commands.py @@ -457,7 +457,8 @@ def cmd_fuzz(args): args.successive_elim = True args.las_vegas = True args.canary_scheduler = True - args.consolidated = True + args.consolidated_v1 = True + args.consolidated_v2 = True args.moss = True args.bayes_ucb = True args.contextual = True @@ -724,7 +725,8 @@ def cmd_fuzz(args): shaped_reward_floor=getattr(args, "shaped_reward_floor", 0.0), continuum_reward=getattr(args, "continuum_reward", False), continuum_reward_floor=getattr(args, "continuum_reward_floor", 0.0), - consolidated=getattr(args, "consolidated", False), + consolidated_v1=getattr(args, "consolidated_v1", False), + consolidated_v2=getattr(args, "consolidated_v2", False), moss=getattr(args, "moss", False), moss_gamma=getattr(args, "moss_gamma", 1.0), bayes_ucb=getattr(args, "bayes_ucb", False), @@ -2058,7 +2060,8 @@ def cmd_sweep(args): "seed_p2c_scheduler", "softmax", "topk", - "consolidated", + "consolidated_v1", + "consolidated_v2", "moss", "bayes_ucb", "contextual", @@ -3452,12 +3455,22 @@ def main() -> int: ), ) fuzz_parser.add_argument( + "--consolidated-v1", "--consolidated", action="store_true", help=( - "Enable the consolidated operator scheduler: Thompson sampling with " + "Enable the consolidated_v1 operator scheduler: Thompson sampling with " "a category-shrunk prior and capped evidence. Takes precedence over " - "every other operator scheduler when Elo is off" + "every other operator scheduler but consolidated_v2 when Elo is off" + ), + ) + fuzz_parser.add_argument( + "--consolidated-v2", + action="store_true", + help=( + "Enable the consolidated_v2 operator scheduler: consolidated_v1 scored " + "by an optimistic, tempered Thompson draw. Takes precedence over every " + "other operator scheduler when Elo is off" ), ) fuzz_parser.add_argument( diff --git a/src/fuzzer_tool/core/schedulers/__init__.py b/src/fuzzer_tool/core/schedulers/__init__.py index 359a094e..3266c60e 100644 --- a/src/fuzzer_tool/core/schedulers/__init__.py +++ b/src/fuzzer_tool/core/schedulers/__init__.py @@ -5,7 +5,8 @@ from fuzzer_tool.core.schedulers.op_c2ucb import C2UCBScheduler from fuzzer_tool.core.schedulers.op_canary import CanaryScheduler from fuzzer_tool.core.schedulers.op_cmaes import CMAESScheduler -from fuzzer_tool.core.schedulers.op_consolidated import ConsolidatedScheduler +from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler +from fuzzer_tool.core.schedulers.op_consolidated_v2 import ConsolidatedV2Scheduler from fuzzer_tool.core.schedulers.op_contextual import ContextualLinUCBScheduler from fuzzer_tool.core.schedulers.op_corral import CorralScheduler from fuzzer_tool.core.schedulers.op_cucb import CUCBScheduler @@ -43,7 +44,8 @@ "WhittleIndexScheduler", "FPLScheduler", "CMAESScheduler", - "ConsolidatedScheduler", + "ConsolidatedV1Scheduler", + "ConsolidatedV2Scheduler", "C2UCBScheduler", "MonteCarloScheduler", "MOptScheduler", diff --git a/src/fuzzer_tool/core/schedulers/op_consolidated.py b/src/fuzzer_tool/core/schedulers/op_consolidated_v1.py similarity index 96% rename from src/fuzzer_tool/core/schedulers/op_consolidated.py rename to src/fuzzer_tool/core/schedulers/op_consolidated_v1.py index 18891cbc..c975f2c1 100644 --- a/src/fuzzer_tool/core/schedulers/op_consolidated.py +++ b/src/fuzzer_tool/core/schedulers/op_consolidated_v1.py @@ -1,4 +1,4 @@ -"""ConsolidatedScheduler: one operator scheduler built from what measured best. +"""ConsolidatedV1Scheduler: one operator scheduler built from what measured best. Why one scheduler and not a portfolio ------------------------------------- @@ -94,7 +94,7 @@ _MIN_PARAM = 1e-3 -class ConsolidatedScheduler: +class ConsolidatedV1Scheduler: """Flat Thompson sampling with a category-shrunk prior and capped evidence. Args: @@ -115,6 +115,9 @@ class ConsolidatedScheduler: #: target_profiler.format_operator_priors(); see init_arm. supports_priors = True + #: bandit_stats() key prefix; subclasses report under their own name. + _STATS_PREFIX = "consolidated_v1" + def __init__( self, prior_strength: float = 4.0, @@ -275,9 +278,10 @@ def bandit_stats(self) -> dict: means = a / np.maximum(a + b, _MIN_PARAM) order = np.argsort(-means)[:5] top = [(self._names[i], round(float(means[i]), 4)) for i in order] + p = self._STATS_PREFIX return { - "consolidated_pulls": self._total_pulls, - "consolidated_successes": round(self._total_successes, 3), - "consolidated_arms": len(self._names), - "consolidated_top": top, + f"{p}_pulls": self._total_pulls, + f"{p}_successes": round(self._total_successes, 3), + f"{p}_arms": len(self._names), + f"{p}_top": top, } diff --git a/src/fuzzer_tool/core/schedulers/op_consolidated_v2.py b/src/fuzzer_tool/core/schedulers/op_consolidated_v2.py new file mode 100644 index 00000000..aec714d6 --- /dev/null +++ b/src/fuzzer_tool/core/schedulers/op_consolidated_v2.py @@ -0,0 +1,134 @@ +"""ConsolidatedV2Scheduler: v1 with an optimistic, tempered Thompson draw. + +Where it comes from +------------------- +Every operator scheduler in ``core/schedulers/`` (41 classes) was run on the +four ``tests/support/bandit_env.py`` environments, 3 seeds each (mean +successes; 6k rounds stationary, 20k decaying/rotting, 60k Fatigue150; +``--``: not finished): + + ================== ========== ======== ======= ========== + scheduler stationary decaying rotting fatigue150 + ================== ========== ======== ======= ========== + Consolidated v1 1696 4652 3553 1522 + Hierarchical 1693 **4658** 3573 1432 + KL-SW-UCB 1710 4294 3124 1442 + Bayes-UCB 1676 3700 3498 1457 + Corral 1616 3594 3497 **1536** + MOSS 1715 3532 3494 1435 + FPL **1723** 3343 2715 1409 + TopK (greedy) 1058 3633 **3633** -- + ================== ========== ======== ======= ========== + +v1 is the only one in the leading group on all four, so v2 keeps all of it: +the category-shrunk prior and the capped pseudocount (both from +Hierarchical), flat Thompson over Beta evidence (MonteCarlo), fractional +Bernoulli rewards. What the others beat it with is *less exploration*: +MOSS stops exploring at an arm's fair share, FPL's perturbation shrinks as +1/sqrt(t), and plain greedy wins the rotting environment, where exploiting +the current best is near-optimal. Thompson's symmetric draw spends half its +samples *below* an arm's mean -- pessimism that only ever demotes arms. + +The change +---------- +Score each candidate by ``mean + tau * max(0, draw - mean)``: + +- **Optimistic** (May et al., *Optimistic Bayesian Sampling in Contextual- + Bandit Problems*, JMLR 2012): a below-mean draw scores the mean, so an + arm is never ranked below its own expectation. +- **Tempered**: an above-mean draw keeps ``tau`` of its excess -- the + exploration bonus scaled down, as MOSS/FPL do with their widths. + +``tau = 1`` is optimistic Thompson; ``tau -> 0`` is greedy on the +posterior mean. A sweep over tau in {0.5, 0.65, 0.75} (12 seeds) put all +three above v1 on every environment; 0.65 is the middle. + +Measured +-------- +Mean successes, 20 paired seeds (same environment stream per seed), real +classes. ``v1 control`` is v1 again under a different scheduler RNG seed: +the gap it shows is sampler noise alone. + + ========== ==== =========== ==== =========== ========= + env v1 v1 control v2 v2 - v1 v2 > v1 + ========== ==== =========== ==== =========== ========= + stationary 1729 1730 1759 +30 (+1.8%) 19/20 + decaying 4612 4617 4685 +73 (+1.6%) 20/20 + rotting 3502 3485 3558 +56 (+1.6%) 17/20 + fatigue150 1524 1520 1544 +20 (+1.3%) 16/20 + ========== ==== =========== ==== =========== ========= + +Every v2 gain is at least 3 standard errors of the paired difference +(6-9); every control gap is within 2. + +What did not help (12 seeds, v1 as the base): a global discount +(gamma 0.9999-0.99999: -2% to -9% on Fatigue150), per-arm staleness decay +(half-life 2k-20k: -4% to -17%), caps of 50/100/300, category caps of 300, +prior strengths of 2/8 -- all at or below v1. Re-opening stale arms to catch +Fatigue150's late unlocks costs more on its ~140 dead arms than it gains. + +Synthetic environments again: the paired A/B (bench_paired) on a real +target is still the claim that matters. + +Rewards and fan-out are v1's: off-policy-safe, one bounded observation per +record. +""" + +from __future__ import annotations + +import math + +import numpy as np + +from fuzzer_tool.core.rand_pool import RandPool +from fuzzer_tool.core.schedulers.op_consolidated_v1 import _MIN_PARAM, ConsolidatedV1Scheduler + + +class ConsolidatedV2Scheduler(ConsolidatedV1Scheduler): + """v1's posterior, scored by an optimistic, tempered Thompson draw. + + Args: + tau: Fraction of a draw's excess over the mean that counts, in + (0, 1]. 0.65 measured; see the module docstring. + prior_strength, max_pseudocount, category_max_pseudocount, rng: + As ``ConsolidatedV1Scheduler``. + """ + + #: init_arm() is v1's: (prior_alpha, prior_beta) become initial evidence. + supports_priors = True + + _STATS_PREFIX = "consolidated_v2" + + def __init__( + self, + tau: float = 0.65, + prior_strength: float = 4.0, + max_pseudocount: float = 200.0, + category_max_pseudocount: float = 1000.0, + rng: RandPool | None = None, + ) -> None: + if math.isnan(tau) or not 0.0 < tau <= 1.0: + raise ValueError(f"tau must be in (0, 1], got {tau!r}") + super().__init__(prior_strength, max_pseudocount, category_max_pseudocount, rng) + self.tau = float(tau) + + def select_op(self, ops: list[str]) -> str: + """Largest ``mean + tau * max(0, draw - mean)`` wins.""" + if not ops: + return "" + if len(ops) == 1: + return ops[0] + + idx = self._indices(ops) + pa, pb = self._prior(idx) + a = np.maximum(pa + self._alpha[idx], _MIN_PARAM) + b = np.maximum(pb + self._beta[idx], _MIN_PARAM) + mean = a / (a + b) + + # Below-mean draws score the mean; above-mean keep tau of the excess. + # Computed in place: the Beta draw already dominates the cost. + score = self._rng.betavariate_array(a, b) - mean + np.maximum(score, 0.0, out=score) + score *= self.tau + score += mean + return ops[int(np.argmax(score))] diff --git a/src/fuzzer_tool/core/schedulers/op_corral.py b/src/fuzzer_tool/core/schedulers/op_corral.py index 15c6a70b..8733be0e 100644 --- a/src/fuzzer_tool/core/schedulers/op_corral.py +++ b/src/fuzzer_tool/core/schedulers/op_corral.py @@ -14,7 +14,7 @@ The existing twenty schedulers are UCB confidence widths (``ucb_common``, ``ducb``, ``swucb``, ``kl_*``, ``cucb``, ``c2ucb``, ``moss``, ``gp_ucb``, ``cusum_ucb``, ``contextual``), posterior sampling (``monte_carlo``, -``consolidated``), exponential weights (``exp3``, ``fpl``), population +``consolidated_v1``), exponential weights (``exp3``, ``fpl``), population methods (``cmaes``, ``mopt``, ``replicator``) and tree search (``mcts``). Log-barrier OMD is none of those, and the difference is not cosmetic: @@ -57,7 +57,7 @@ round. Whoever played is punished hardest, which is churn, not learning, and it is the same pathology that makes the Elo fan-out prefer whoever played least recently (measured in -``core/schedulers/op_consolidated.py``). Unbiasedness does not save it; the +``core/schedulers/op_consolidated_v1.py``). Unbiasedness does not save it; the variance is what does the damage. The fix is the standard loss shift. With running mean loss ``b``, diff --git a/src/fuzzer_tool/services/fuzzer.py b/src/fuzzer_tool/services/fuzzer.py index 26df1947..5234b7af 100644 --- a/src/fuzzer_tool/services/fuzzer.py +++ b/src/fuzzer_tool/services/fuzzer.py @@ -68,7 +68,8 @@ C2UCBScheduler, CanaryScheduler, CMAESScheduler, - ConsolidatedScheduler, + ConsolidatedV1Scheduler, + ConsolidatedV2Scheduler, ContextualLinUCBScheduler, CorralScheduler, CUCBScheduler, @@ -142,7 +143,8 @@ # Strategy names pre-registered with the Elo tracker (single source of truth # for the pre-registration loop and the meta-scheduler log line). _OPERATOR_STRATEGY_NAMES = ( - "consolidated", + "consolidated_v2", + "consolidated_v1", "replicator", "bandit", "mopt", @@ -1312,7 +1314,8 @@ def __init__( shaped_reward_floor=0.0, continuum_reward=False, continuum_reward_floor=0.0, - consolidated=False, + consolidated_v1=False, + consolidated_v2=False, moss=False, moss_gamma=1.0, bayes_ucb=False, @@ -3396,13 +3399,21 @@ def __init__( # Consolidated: flat Thompson with a category-shrunk prior and capped # evidence -- the single learner meant to replace the Elo portfolio - # (see core/schedulers/op_consolidated.py for the measurements). - self._use_consolidated = consolidated - self._consolidated = None - if consolidated: - self._consolidated = ConsolidatedScheduler(rng=self._rng) + # (see core/schedulers/op_consolidated_v1.py for the measurements). + self._use_consolidated_v1 = consolidated_v1 + self._consolidated_v1 = None + if consolidated_v1: + self._consolidated_v1 = ConsolidatedV1Scheduler(rng=self._rng) log.info("Consolidated operator scheduler enabled") + # Consolidated v2: v1 scored by an optimistic, tempered Thompson draw + # (see core/schedulers/op_consolidated_v2.py for the measurements). + self._use_consolidated_v2 = consolidated_v2 + self._consolidated_v2 = None + if consolidated_v2: + self._consolidated_v2 = ConsolidatedV2Scheduler(rng=self._rng) + log.info("Consolidated v2 operator scheduler enabled") + # MOSS: UCB whose exploration bonus ends at an arm's fair share t/K, # built for many low-yield operators (see core/schedulers/op_moss.py). self._use_moss = moss @@ -3670,7 +3681,8 @@ def __init__( or self._kl_ducb or self._corral or self._kl_swucb - or self._consolidated + or self._consolidated_v1 + or self._consolidated_v2 or self._moss or self._bayes_ucb or self._cucb @@ -3887,8 +3899,10 @@ def _init(op): _register_arms(self._las_vegas, _format_priors) if self._op_kuramoto: _register_arms(self._op_kuramoto) - if self._consolidated: - _register_arms(self._consolidated, _format_priors) + if self._consolidated_v1: + _register_arms(self._consolidated_v1, _format_priors) + if self._consolidated_v2: + _register_arms(self._consolidated_v2, _format_priors) if self._moss: _register_arms(self._moss) if self._bayes_ucb: @@ -7086,7 +7100,8 @@ def _op_success(op: str) -> bool: self._gradient, self._whittle, self._successive_elim, - self._consolidated, + self._consolidated_v1, + self._consolidated_v2, self._moss, self._bayes_ucb, self._canary, @@ -8785,8 +8800,10 @@ def _selected_schedulers_str(self) -> str: parts.append(f"power={self._power_schedule}") ops = [] - if getattr(self, "_consolidated", False): - ops.append("consolidated") + if getattr(self, "_consolidated_v2", False): + ops.append("consolidated_v2") + if getattr(self, "_consolidated_v1", False): + ops.append("consolidated_v1") if self.mc_bandit: ops.append("bandit") if self.mc_cem: @@ -9215,8 +9232,10 @@ def _print_enabled_features(self) -> None: groups["Scheduling"].append(f"power-schedule={sched_pol}") ops = [] - if getattr(self, "_consolidated", False): - ops.append("consolidated") + if getattr(self, "_consolidated_v2", False): + ops.append("consolidated_v2") + if getattr(self, "_consolidated_v1", False): + ops.append("consolidated_v1") if self.mc_bandit: ops.append("bandit") if self.mc_cem: diff --git a/src/fuzzer_tool/services/operators.py b/src/fuzzer_tool/services/operators.py index 8d7eb007..f1552c74 100644 --- a/src/fuzzer_tool/services/operators.py +++ b/src/fuzzer_tool/services/operators.py @@ -623,13 +623,14 @@ def _det_interesting(data, scratch, positions, quota): #: No-Elo selection order, highest first: with Elo off, the first enabled #: scheduler here selects every operator. ``cem`` and ``invasion`` are #: absent on purpose -- both ride on ``f.mc`` and ``mc_bandit``, which the -#: ``bandit`` entry already claims ahead of them. ``consolidated`` leads: -#: it is the scheduler built to be the one that runs (see -#: core/schedulers/op_consolidated.py), so enabling it alongside others without -#: Elo means it selects. Pinned by +#: ``bandit`` entry already claims ahead of them. ``consolidated_v2`` leads, +#: then ``consolidated_v1``: they are the schedulers built to be the one that +#: runs (see core/schedulers/op_consolidated_v1.py, op_consolidated_v2.py), so +#: enabling one alongside others without Elo means it selects. Pinned by #: test_regression_scheduler_fallback_precedence. _FALLBACK_PRECEDENCE = ( - "consolidated", + "consolidated_v2", + "consolidated_v1", "replicator", "mopt", "bandit", @@ -709,8 +710,10 @@ def operator_strategy_pool(f) -> list[str]: while nothing is rated yet, so it is not cosmetic. """ available = [] - if f._use_consolidated and f._consolidated: - available.append("consolidated") + if f._use_consolidated_v2 and f._consolidated_v2: + available.append("consolidated_v2") + if f._use_consolidated_v1 and f._consolidated_v1: + available.append("consolidated_v1") if f._use_replicator and f._replicator: available.append("replicator") if f.mc and f.mc_bandit: @@ -5078,8 +5081,11 @@ def select_op(self, ops: list[str]) -> str: if f._use_elo and f._elo and strategy: f._meta_strategy_used.add(strategy) - if strategy == "consolidated" and f._consolidated: - op = f._consolidated.select_op(ops) + if strategy == "consolidated_v2" and f._consolidated_v2: + op = f._consolidated_v2.select_op(ops) + f._last_mopt_particles.append(None) + elif strategy == "consolidated_v1" and f._consolidated_v1: + op = f._consolidated_v1.select_op(ops) f._last_mopt_particles.append(None) elif strategy == "replicator" and f._replicator: op = f._replicator.select_op(ops) diff --git a/tests/support/bandit_env.py b/tests/support/bandit_env.py index 30b407bb..69434645 100644 --- a/tests/support/bandit_env.py +++ b/tests/support/bandit_env.py @@ -34,7 +34,7 @@ count, not the global round. Separates rested (per-arm) forgetters from round-indexed ones; DecayingBest cannot. * :class:`Fatigue150` -- 150 arms, rare heavy-tailed yields, fatigue on - success, periodic unlocks: the environment ``op_consolidated.py`` was tuned + success, periodic unlocks: the environment ``op_consolidated_v1.py`` was tuned on, reconstructed. Stateful environments (the last two) expose ``reset()`` and diff --git a/tests/support/operator_env.py b/tests/support/operator_env.py index b6a0ef9b..b5e6552d 100644 --- a/tests/support/operator_env.py +++ b/tests/support/operator_env.py @@ -36,7 +36,8 @@ "c2ucb", "canary", "cmaes", - "consolidated", + "consolidated_v1", + "consolidated_v2", "contextual", "cucb", "cusum_ucb", diff --git a/tests/test_bandit_env_fatigue.py b/tests/test_bandit_env_fatigue.py index 344d826d..5337a301 100644 --- a/tests/test_bandit_env_fatigue.py +++ b/tests/test_bandit_env_fatigue.py @@ -2,7 +2,7 @@ RottingArms decays an arm by its *own* pulls, so it separates rested forgetters from round-indexed ones, which DecayingBest cannot. -Fatigue150 reconstructs op_consolidated.py's 150-arm environment. +Fatigue150 reconstructs op_consolidated_v1.py's 150-arm environment. """ from __future__ import annotations diff --git a/tests/test_canary_scheduler.py b/tests/test_canary_scheduler.py index 48a81c2c..2613cbd0 100644 --- a/tests/test_canary_scheduler.py +++ b/tests/test_canary_scheduler.py @@ -14,7 +14,7 @@ from fuzzer_tool.core.analyzers.analyzer_elo import BayesianEloTracker from fuzzer_tool.core.schedulers.op_canary import CanaryScheduler -from fuzzer_tool.core.schedulers.op_consolidated import ConsolidatedScheduler +from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler ARMS = ["bit_flip", "byte_flip", "havoc"] @@ -130,7 +130,7 @@ def run(select, record, rounds=4000): canary_reward = run(canary.select_op, canary.record) rng = random.Random(1234) # same draw stream for a fair comparison - consolidated = ConsolidatedScheduler() + consolidated = ConsolidatedV1Scheduler() consolidated_reward = run(consolidated.select_op, consolidated.record) assert canary_reward < consolidated_reward diff --git a/tests/test_consolidated_scheduler.py b/tests/test_consolidated_v1_scheduler.py similarity index 83% rename from tests/test_consolidated_scheduler.py rename to tests/test_consolidated_v1_scheduler.py index d9132f69..ff4cca32 100644 --- a/tests/test_consolidated_scheduler.py +++ b/tests/test_consolidated_v1_scheduler.py @@ -1,4 +1,4 @@ -"""ConsolidatedScheduler: the properties it is made of, one at a time. +"""ConsolidatedV1Scheduler: the properties it is made of, one at a time. Convergence and decay recovery live in test_scheduler_convergence (RELIABLE and RECOVERS). These pin the mechanisms: category sharing reaches unsampled @@ -15,7 +15,7 @@ from fuzzer_tool.core.operator_categories import OPERATOR_CATEGORIES, category_of from fuzzer_tool.core.rand_pool import RandPool -from fuzzer_tool.core.schedulers import ConsolidatedScheduler +from fuzzer_tool.core.schedulers import ConsolidatedV1Scheduler def _two_categories(): @@ -31,7 +31,7 @@ def test_unsampled_arm_inherits_its_categorys_rate(): """The point of the shrinkage prior: evidence on two operators of a category moves the third, never pulled, above an arm of a dead one.""" (a1, a2, a_unseen), b = _two_categories() - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) for op in (a1, a2, a_unseen, b): s.init_arm(op) for _ in range(100): @@ -46,7 +46,7 @@ def test_unsampled_arm_inherits_its_categorys_rate(): def test_without_prior_strength_categories_are_ignored(): (a1, a2, a_unseen), _ = _two_categories() - s = ConsolidatedScheduler(prior_strength=0.0, rng=RandPool(1)) + s = ConsolidatedV1Scheduler(prior_strength=0.0, rng=RandPool(1)) for _ in range(100): s.record(a1, True) s.record(a2, True) @@ -54,7 +54,7 @@ def test_without_prior_strength_categories_are_ignored(): def test_cap_bounds_evidence_and_keeps_the_mean(): - s = ConsolidatedScheduler(max_pseudocount=50.0, rng=RandPool(1)) + s = ConsolidatedV1Scheduler(max_pseudocount=50.0, rng=RandPool(1)) op = "bit_flip" for i in range(1000): s.record(op, i % 4 == 0) @@ -66,7 +66,7 @@ def test_cap_bounds_evidence_and_keeps_the_mean(): def test_cap_lets_a_dead_arm_be_left(): """After the cap, 200 failures move the mean most of the way down -- an uncapped posterior with 5000 prior successes would barely move.""" - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) op = "bit_flip" for _ in range(5000): s.record(op, True) @@ -76,7 +76,7 @@ def test_cap_lets_a_dead_arm_be_left(): def test_format_prior_is_a_one_time_nudge(): - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) s.init_arm("bit_flip", 2.0, 1.0) aid = s._index["bit_flip"] assert (s._alpha[aid], s._beta[aid]) == (1.0, 0.0) @@ -90,7 +90,7 @@ def test_format_prior_is_a_one_time_nudge(): ("success", "weight", "alpha"), [(True, 15.0, 1.0), (True, 0.25, 0.25), (False, 1.0, 0.0)] ) def test_one_record_is_one_bounded_observation(success, weight, alpha): - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) s.record("bit_flip", success, weight=weight) aid = s._index["bit_flip"] assert s._alpha[aid] == pytest.approx(alpha) @@ -101,7 +101,7 @@ def test_seeded_rng_reproduces_the_campaign(): ops = ["bit_flip", "byte_flip", "arith_inc", "havoc"] def campaign(seed): - s = ConsolidatedScheduler(rng=RandPool(seed)) + s = ConsolidatedV1Scheduler(rng=RandPool(seed)) out = [] for i in range(300): op = s.select_op(ops) @@ -114,30 +114,30 @@ def campaign(seed): def test_degenerate_candidate_lists(): - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) assert s.select_op([]) == "" assert s.select_op(["havoc"]) == "havoc" def test_unknown_operator_is_registered_lazily(): - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) assert s.select_op(["not_a_real_op", "bit_flip"]) in ("not_a_real_op", "bit_flip") s.record("also_unknown", True) assert "also_unknown" in s._index def test_bandit_stats_shape(): - s = ConsolidatedScheduler(rng=RandPool(1)) + s = ConsolidatedV1Scheduler(rng=RandPool(1)) s.record("bit_flip", True) stats = s.bandit_stats() - assert stats["consolidated_pulls"] == 1 - assert stats["consolidated_top"][0][0] == "bit_flip" + assert stats["consolidated_v1_pulls"] == 1 + assert stats["consolidated_v1_top"][0][0] == "bit_flip" @pytest.mark.parametrize("kwargs", [{"prior_strength": -1.0}, {"max_pseudocount": 0.0}]) def test_rejects_invalid_parameters(kwargs): with pytest.raises(ValueError): - ConsolidatedScheduler(**kwargs) + ConsolidatedV1Scheduler(**kwargs) _TARGET = Path(__file__).resolve().parent.parent / "targets" / "test_target" @@ -159,10 +159,10 @@ def test_fuzzer_wiring_selects_and_learns(tmp_path): crashes_dir=str(crashes), max_len=4096, use_coverage=True, - consolidated=True, + consolidated_v1=True, mc_bandit=True, ) - assert isinstance(f._consolidated, ConsolidatedScheduler) + assert isinstance(f._consolidated_v1, ConsolidatedV1Scheduler) assert f._track_op_effect def bandit_mass(): @@ -173,7 +173,7 @@ def bandit_mass(): for i in range(40): f.fuzz_one(bytes([65 + i % 26]) * 16) selectors.add(f._op_selector) - assert "consolidated" in selectors + assert "consolidated_v1" in selectors assert "bandit" not in selectors - assert f._consolidated.bandit_stats()["consolidated_pulls"] > 0 + assert f._consolidated_v1.bandit_stats()["consolidated_v1_pulls"] > 0 assert bandit_mass() > before, "the bandit stopped learning from rounds it did not select" diff --git a/tests/test_consolidated_v2_scheduler.py b/tests/test_consolidated_v2_scheduler.py new file mode 100644 index 00000000..88ec4c9b --- /dev/null +++ b/tests/test_consolidated_v2_scheduler.py @@ -0,0 +1,189 @@ +"""ConsolidatedV2Scheduler: v1 plus an optimistic, tempered Thompson draw. + +The score is ``mean + tau * max(0, draw - mean)``. These pin that rule with +scripted draws, its limits (tau=1, all-pessimistic draws), and that v1's +category prior, cap and reward handling are inherited unchanged. Convergence +and decay recovery live in test_scheduler_convergence. +""" + +from __future__ import annotations + +from pathlib import Path + +import numpy as np +import pytest + +from fuzzer_tool.core.rand_pool import RandPool +from fuzzer_tool.core.schedulers import ConsolidatedV1Scheduler, ConsolidatedV2Scheduler + +# Two operators in different categories, so neither's evidence moves the +# other's prior. +_STRONG = "bit_flip" +_WEAK = "havoc" + + +class _ScriptedBeta: + """RandPool seam: ``betavariate_array`` returns the scripted draws in order.""" + + def __init__(self, *draws): + self._draws = [np.asarray(d, dtype=float) for d in draws] + + def betavariate_array(self, alphas, betas): + assert len(alphas) == len(betas) == len(self._draws[0]) + return self._draws.pop(0) + + +def _trained(cls, rng, **kwargs): + """_STRONG at ~0.3 over 100 pulls, _WEAK at 0 over 20.""" + s = cls(rng=rng, **kwargs) + for i in range(100): + s.record(_STRONG, i % 10 < 3) + for _ in range(20): + s.record(_WEAK, False) + return s + + +def _means(s): + return s.posterior_mean(_STRONG), s.posterior_mean(_WEAK) + + +def test_pessimistic_draw_is_floored_at_the_mean(): + """_STRONG draws below its mean, _WEAK draws above its own mean but below + _STRONG's. Raw Thompson (v1) takes _WEAK; v2 scores _STRONG at its mean + and _WEAK at mean + tau * excess, which stays below it.""" + probe = _trained(ConsolidatedV2Scheduler, RandPool(1)) + mu_s, mu_w = _means(probe) + assert mu_w < mu_s - 0.1 + draws = [mu_s - 0.2, mu_s - 0.1] + + v1 = _trained(ConsolidatedV1Scheduler, _ScriptedBeta(draws)) + v2 = _trained(ConsolidatedV2Scheduler, _ScriptedBeta(draws)) + + # Control: the same draws make raw Thompson pick the other arm, so the + # assertion below can fail. + assert v1.select_op([_STRONG, _WEAK]) == _WEAK + assert v2.select_op([_STRONG, _WEAK]) == _STRONG + + +def test_tempering_shrinks_the_optimistic_excess(): + """Both arms draw above their means. Score = mean + tau * excess, so the + winner flips at the tau where the two scores cross.""" + probe = _trained(ConsolidatedV2Scheduler, RandPool(1)) + mu_s, mu_w = _means(probe) + excess_s, excess_w = 0.01, (mu_s - mu_w) + 0.05 + draws = [mu_s + excess_s, mu_w + excess_w] + + # mu_s + tau*excess_s == mu_w + tau*excess_w + tau_cross = (mu_s - mu_w) / (excess_w - excess_s) + lo = _trained(ConsolidatedV2Scheduler, _ScriptedBeta(draws), tau=tau_cross * 0.9) + hi = _trained(ConsolidatedV2Scheduler, _ScriptedBeta(draws), tau=min(1.0, tau_cross * 1.1)) + + assert lo.select_op([_STRONG, _WEAK]) == _STRONG + assert hi.select_op([_STRONG, _WEAK]) == _WEAK + + +def test_all_pessimistic_draws_select_the_best_mean(): + """Adversarial for the sampler: every draw lands far below its mean, in + reversed order. v2 degrades to greedy on the posterior mean.""" + ops = ["bit_flip", "byte_flip", "arith_inc", "havoc"] + rates = [0.1, 0.4, 0.2, 0.05] + s = ConsolidatedV2Scheduler(rng=RandPool(1)) + for op, p in zip(ops, rates, strict=True): + for i in range(100): + s.record(op, i < p * 100) + means = [s.posterior_mean(op) for op in ops] + draws = [1e-6 * (len(ops) - k) for k in range(len(ops))] # argmax would be ops[0] + + s._rng = _ScriptedBeta(draws) + assert s.select_op(ops) == ops[int(np.argmax(means))] + + +def test_tau_one_is_optimistic_thompson(): + """tau=1: the score is max(draw, mean) -- an above-mean draw counts in full.""" + probe = _trained(ConsolidatedV2Scheduler, RandPool(1)) + mu_s, mu_w = _means(probe) + draws = [mu_s - 0.2, mu_s + 0.01] + s = _trained(ConsolidatedV2Scheduler, _ScriptedBeta(draws), tau=1.0) + assert s.select_op([_STRONG, _WEAK]) == _WEAK + + +@pytest.mark.parametrize("tau", [0.0, -0.5, 1.5, float("nan")]) +def test_rejects_tau_outside_unit_interval(tau): + with pytest.raises(ValueError): + ConsolidatedV2Scheduler(tau=tau) + + +def test_inherits_v1_learning(): + """Same records, same posterior: v2 changes selection only.""" + v1 = _trained(ConsolidatedV1Scheduler, RandPool(1)) + v2 = _trained(ConsolidatedV2Scheduler, RandPool(1)) + for op in (_STRONG, _WEAK, "arith_inc"): + assert v2.posterior_mean(op) == pytest.approx(v1.posterior_mean(op)) + + +def test_seeded_rng_reproduces_the_campaign(): + ops = ["bit_flip", "byte_flip", "arith_inc", "havoc"] + + def campaign(seed): + s = ConsolidatedV2Scheduler(rng=RandPool(seed)) + out = [] + for i in range(300): + op = s.select_op(ops) + s.record(op, (i * 7 + len(op)) % 5 == 0) + out.append(op) + return out + + assert campaign(3) == campaign(3) + assert campaign(3) != campaign(4) + + +def test_degenerate_candidate_lists(): + s = ConsolidatedV2Scheduler(rng=RandPool(1)) + assert s.select_op([]) == "" + assert s.select_op(["havoc"]) == "havoc" + + +def test_bandit_stats_shape(): + s = ConsolidatedV2Scheduler(rng=RandPool(1)) + s.record("bit_flip", True) + stats = s.bandit_stats() + assert stats["consolidated_v2_pulls"] == 1 + assert stats["consolidated_v2_top"][0][0] == "bit_flip" + assert not any(k.startswith("consolidated_v1") for k in stats) + + +def test_declares_supports_priors(): + assert ConsolidatedV2Scheduler.supports_priors is True + + +_TARGET = Path(__file__).resolve().parent.parent / "targets" / "test_target" + + +@pytest.mark.skipif(not _TARGET.exists(), reason="targets/test_target not built") +def test_fuzzer_wiring_selects_ahead_of_v1(tmp_path): + """--consolidated-v2 builds it and, without Elo, it selects ahead of v1; + v1 still learns from the rounds through the shared fan-out.""" + from fuzzer_tool.services.fuzzer import Fuzzer + + corpus, crashes = tmp_path / "c", tmp_path / "k" + corpus.mkdir() + crashes.mkdir() + f = Fuzzer( + target=str(_TARGET), + corpus_dir=str(corpus), + crashes_dir=str(crashes), + max_len=4096, + use_coverage=True, + consolidated_v1=True, + consolidated_v2=True, + ) + assert isinstance(f._consolidated_v2, ConsolidatedV2Scheduler) + assert f._track_op_effect + + selectors = set() + for i in range(40): + f.fuzz_one(bytes([65 + i % 26]) * 16) + selectors.add(f._op_selector) + assert selectors == {"consolidated_v2"} + assert f._consolidated_v2.bandit_stats()["consolidated_v2_pulls"] > 0 + assert f._consolidated_v1.bandit_stats()["consolidated_v1_pulls"] > 0 diff --git a/tests/test_corral_scheduler.py b/tests/test_corral_scheduler.py index b8097674..dc819114 100644 --- a/tests/test_corral_scheduler.py +++ b/tests/test_corral_scheduler.py @@ -273,7 +273,7 @@ def test_a_productive_operator_gains_probability(): def test_concentrates_at_fuzzing_realistic_rates(): """10% against 1% -- the regime where the loss shift matters. - ``core/schedulers/op_consolidated.py`` records that the Elo fan-out gave a + ``core/schedulers/op_consolidated_v1.py`` records that the Elo fan-out gave a ten-times-more-productive arm only 54% of the picks. This is the bar that motivated the family. """ diff --git a/tests/test_regression_elo_all.py b/tests/test_regression_elo_all.py index 05e20fc4..bdbf1e91 100644 --- a/tests/test_regression_elo_all.py +++ b/tests/test_regression_elo_all.py @@ -51,7 +51,8 @@ def fake_fuzzer(**kwargs): "cusum_ucb", "c2ucb", "fpl", - "consolidated", + "consolidated_v1", + "consolidated_v2", "moss", "bayes_ucb", "contextual", diff --git a/tests/test_regression_scheduler_fallback_precedence.py b/tests/test_regression_scheduler_fallback_precedence.py index 1244ce84..8812ce12 100644 --- a/tests/test_regression_scheduler_fallback_precedence.py +++ b/tests/test_regression_scheduler_fallback_precedence.py @@ -30,7 +30,8 @@ # Fallback precedence, highest first. "random" is the terminal fallback (the # chain's else-branch, represented by _rng.choice, not a scheduler). _FALLBACK_PRECEDENCE = [ - "consolidated", + "consolidated_v2", + "consolidated_v1", "replicator", "mopt", "bandit", @@ -53,6 +54,7 @@ "bayes_ucb", "fpl", "successive_elim", + "las_vegas", "round_robin", ] @@ -116,7 +118,8 @@ class _FakeFuzzer: """Minimal fuzzer stand-in exposing exactly the attrs select_op() reads.""" _SCHEDULER_ATTRS = { - "consolidated": ("_use_consolidated", "_consolidated"), + "consolidated_v2": ("_use_consolidated_v2", "_consolidated_v2"), + "consolidated_v1": ("_use_consolidated_v1", "_consolidated_v1"), "replicator": ("_use_replicator", "_replicator"), "mopt": ("_use_mopt", "_mopt"), "exp3": ("_use_exp3", "_exp3"), @@ -138,6 +141,7 @@ class _FakeFuzzer: "fpl": ("_use_fpl", "_fpl"), "exp4": ("_use_exp4", "_exp4"), "successive_elim": ("_use_successive_elim", "_successive_elim"), + "las_vegas": ("_use_las_vegas", "_las_vegas"), "round_robin": ("_use_round_robin", "_round_robin"), # canary, op_katz, op_kuramoto, op_tang, gradient, whittle, corral # are deliberately absent from _FALLBACK_PRECEDENCE (see @@ -231,10 +235,19 @@ def test_highest_priority_enabled_wins(self): if name != first: assert fake.calls == 0, f"{name} consulted despite {first} enabled" + def test_v2_absent_falls_to_v1(self): + """consolidated_v2 leads; without it consolidated_v1 selects.""" + f = _FakeFuzzer() + fakes = {name: f.enable(name) for name in _FALLBACK_PRECEDENCE if name != "consolidated_v2"} + assert OperatorEngine(f).select_op(["bit_flip", "byte_flip"]) == "op_consolidated_v1" + assert fakes["consolidated_v1"].calls == 1 + assert fakes["replicator"].calls == 0 + def test_everything_but_consolidated_falls_to_replicator(self): """The pre-consolidated order is unchanged underneath it.""" f = _FakeFuzzer() - fakes = {name: f.enable(name) for name in _FALLBACK_PRECEDENCE if name != "consolidated"} + consolidated = ("consolidated_v1", "consolidated_v2") + fakes = {name: f.enable(name) for name in _FALLBACK_PRECEDENCE if name not in consolidated} assert OperatorEngine(f).select_op(["bit_flip", "byte_flip"]) == "op_replicator" assert fakes["replicator"].calls == 1 diff --git a/tests/test_regression_scheduler_operator_reach.py b/tests/test_regression_scheduler_operator_reach.py index 0a32f6f0..32edae26 100644 --- a/tests/test_regression_scheduler_operator_reach.py +++ b/tests/test_regression_scheduler_operator_reach.py @@ -132,11 +132,17 @@ def _ctx(_op): ("CUCBScheduler", cucb, cucb.select_op, cucb.record), ("CUSUM_UCBScheduler", cusum := S.CUSUM_UCBScheduler(), cusum.select_op, cusum.record), ( - "ConsolidatedScheduler", - consolidated := S.ConsolidatedScheduler(), + "ConsolidatedV1Scheduler", + consolidated := S.ConsolidatedV1Scheduler(), consolidated.select_op, consolidated.record, ), + ( + "ConsolidatedV2Scheduler", + consolidated_v2 := S.ConsolidatedV2Scheduler(), + consolidated_v2.select_op, + consolidated_v2.record, + ), ("MOSSScheduler", moss := S.MOSSScheduler(), moss.select_op, moss.record), ("BayesUCBScheduler", bayes := S.BayesUCBScheduler(), bayes.select_op, bayes.record), # RoundRobin is stateless in terms of rewards; record() is no-op. diff --git a/tests/test_regression_track_op_effect_coverage.py b/tests/test_regression_track_op_effect_coverage.py index 3491e86e..82953045 100644 --- a/tests/test_regression_track_op_effect_coverage.py +++ b/tests/test_regression_track_op_effect_coverage.py @@ -29,7 +29,8 @@ #: Ballot name -> Fuzzer kwargs that enable it. _KWARGS = { - "consolidated": {"consolidated": True}, + "consolidated_v1": {"consolidated_v1": True}, + "consolidated_v2": {"consolidated_v2": True}, "moss": {"moss": True}, "bayes_ucb": {"bayes_ucb": True}, "replicator": {"replicator": True}, diff --git a/tests/test_scheduler_convergence.py b/tests/test_scheduler_convergence.py index f6012035..11609486 100644 --- a/tests/test_scheduler_convergence.py +++ b/tests/test_scheduler_convergence.py @@ -62,7 +62,8 @@ from fuzzer_tool.core.rand_pool import RandPool from fuzzer_tool.core.schedulers import ( CMAESScheduler, - ConsolidatedScheduler, + ConsolidatedV1Scheduler, + ConsolidatedV2Scheduler, ContextualLinUCBScheduler, CorralScheduler, CUCBScheduler, @@ -110,7 +111,12 @@ RELIABLE = { # Floors below the observed minimum over 40 seeds at ROUNDS: share # 0.947, slope max 0.611 (median 0.267). - "Consolidated": (lambda seed: ConsolidatedScheduler(rng=RandPool(seed)), 0.90, 0.70), + "Consolidated": (lambda seed: ConsolidatedV1Scheduler(rng=RandPool(seed)), 0.90, 0.70), + # Over 40 seeds: share min 0.968, slope median 0.184, max 0.947 on seed + # 12 -- whose total regret (33) is the lowest of the 40, so its slope is + # fitted on near-zero increments. The highest-regret seeds have the + # lowest slopes (<0.02). At FIXED_SEED: 0.996 / 0.107. + "ConsolidatedV2": (lambda seed: ConsolidatedV2Scheduler(rng=RandPool(seed)), 0.93, 0.70), "ContextualLinUCB": (lambda seed: ContextualLinUCBScheduler(dim=4), 0.90, 0.40), "EpsilonGreedy": (lambda seed: EpsilonGreedyScheduler(), 0.85, 0.45), "Exp3": (lambda seed: Exp3Scheduler(), 0.78, 0.70), @@ -362,7 +368,9 @@ def test_hierarchical_category_starvation_is_rare(): "Hierarchical": (HierarchicalBanditScheduler, 0.90), # Minimum over 20 seeds 0.946: the same pseudocount cap as Hierarchical, # without its category-first starvation. - "Consolidated": (lambda: ConsolidatedScheduler(rng=RandPool(FIXED_SEED)), 0.90), + "Consolidated": (lambda: ConsolidatedV1Scheduler(rng=RandPool(FIXED_SEED)), 0.90), + # Minimum over 20 seeds 0.984 (v1: 0.946). + "ConsolidatedV2": (lambda: ConsolidatedV2Scheduler(rng=RandPool(FIXED_SEED)), 0.95), # The three schedulers built for this regime. Floors sit below the # observed minimum over 20 seeds at 20k rounds: D-UCB 0.739, # SW-UCB 0.806, CUCB 0.910. From 6a8cc2d9407534287fefc286fc442016d29fe37b Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 16:29:58 +0000 Subject: [PATCH 2/3] Keep ConsolidatedScheduler importable as a v1 alias ImpactGuard flagged the rename as removing ConsolidatedScheduler. The old module path and package name now alias ConsolidatedV1Scheduler. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ape9xt5N7dHzpFqQcsNwk --- src/fuzzer_tool/core/schedulers/__init__.py | 3 +++ src/fuzzer_tool/core/schedulers/op_consolidated.py | 10 ++++++++++ 2 files changed, 13 insertions(+) create mode 100644 src/fuzzer_tool/core/schedulers/op_consolidated.py diff --git a/src/fuzzer_tool/core/schedulers/__init__.py b/src/fuzzer_tool/core/schedulers/__init__.py index 3266c60e..77f2fa61 100644 --- a/src/fuzzer_tool/core/schedulers/__init__.py +++ b/src/fuzzer_tool/core/schedulers/__init__.py @@ -5,6 +5,9 @@ from fuzzer_tool.core.schedulers.op_c2ucb import C2UCBScheduler from fuzzer_tool.core.schedulers.op_canary import CanaryScheduler from fuzzer_tool.core.schedulers.op_cmaes import CMAESScheduler +from fuzzer_tool.core.schedulers.op_consolidated import ( + ConsolidatedScheduler, # noqa: F401 pre-v2 name +) from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler from fuzzer_tool.core.schedulers.op_consolidated_v2 import ConsolidatedV2Scheduler from fuzzer_tool.core.schedulers.op_contextual import ContextualLinUCBScheduler diff --git a/src/fuzzer_tool/core/schedulers/op_consolidated.py b/src/fuzzer_tool/core/schedulers/op_consolidated.py new file mode 100644 index 00000000..5ca3f075 --- /dev/null +++ b/src/fuzzer_tool/core/schedulers/op_consolidated.py @@ -0,0 +1,10 @@ +"""Moved to ``op_consolidated_v1.py``; kept so pre-v2 imports still resolve.""" + +from __future__ import annotations + +from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler + +#: Pre-v2 name of :class:`ConsolidatedV1Scheduler`. +ConsolidatedScheduler = ConsolidatedV1Scheduler + +__all__ = ["ConsolidatedScheduler"] From 143f1e80a147c484fb27244ab4920429f24c434a Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 17:25:20 +0000 Subject: [PATCH 3/3] Keep pre-v2 consolidated API intact Fuzzer keeps `consolidated` in its positional slot (pre-v2 name of v1); `consolidated_v1`/`consolidated_v2` are appended. ConsolidatedScheduler is exported and reports the pre-v2 stats keys. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_019ape9xt5N7dHzpFqQcsNwk --- src/fuzzer_tool/core/schedulers/__init__.py | 5 +- .../core/schedulers/op_consolidated.py | 8 +++- src/fuzzer_tool/services/fuzzer.py | 8 +++- tests/test_consolidated_v1_scheduler.py | 48 +++++++++++++++++++ ...est_regression_scheduler_operator_reach.py | 6 +++ 5 files changed, 68 insertions(+), 7 deletions(-) diff --git a/src/fuzzer_tool/core/schedulers/__init__.py b/src/fuzzer_tool/core/schedulers/__init__.py index 77f2fa61..40be6311 100644 --- a/src/fuzzer_tool/core/schedulers/__init__.py +++ b/src/fuzzer_tool/core/schedulers/__init__.py @@ -5,9 +5,7 @@ from fuzzer_tool.core.schedulers.op_c2ucb import C2UCBScheduler from fuzzer_tool.core.schedulers.op_canary import CanaryScheduler from fuzzer_tool.core.schedulers.op_cmaes import CMAESScheduler -from fuzzer_tool.core.schedulers.op_consolidated import ( - ConsolidatedScheduler, # noqa: F401 pre-v2 name -) +from fuzzer_tool.core.schedulers.op_consolidated import ConsolidatedScheduler from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler from fuzzer_tool.core.schedulers.op_consolidated_v2 import ConsolidatedV2Scheduler from fuzzer_tool.core.schedulers.op_contextual import ContextualLinUCBScheduler @@ -47,6 +45,7 @@ "WhittleIndexScheduler", "FPLScheduler", "CMAESScheduler", + "ConsolidatedScheduler", "ConsolidatedV1Scheduler", "ConsolidatedV2Scheduler", "C2UCBScheduler", diff --git a/src/fuzzer_tool/core/schedulers/op_consolidated.py b/src/fuzzer_tool/core/schedulers/op_consolidated.py index 5ca3f075..58cda3ff 100644 --- a/src/fuzzer_tool/core/schedulers/op_consolidated.py +++ b/src/fuzzer_tool/core/schedulers/op_consolidated.py @@ -4,7 +4,11 @@ from fuzzer_tool.core.schedulers.op_consolidated_v1 import ConsolidatedV1Scheduler -#: Pre-v2 name of :class:`ConsolidatedV1Scheduler`. -ConsolidatedScheduler = ConsolidatedV1Scheduler + +class ConsolidatedScheduler(ConsolidatedV1Scheduler): + """Pre-v2 name of ConsolidatedV1Scheduler; reports the pre-v2 stats keys.""" + + _STATS_PREFIX = "consolidated" + __all__ = ["ConsolidatedScheduler"] diff --git a/src/fuzzer_tool/services/fuzzer.py b/src/fuzzer_tool/services/fuzzer.py index 5234b7af..1f719ff7 100644 --- a/src/fuzzer_tool/services/fuzzer.py +++ b/src/fuzzer_tool/services/fuzzer.py @@ -1314,8 +1314,7 @@ def __init__( shaped_reward_floor=0.0, continuum_reward=False, continuum_reward_floor=0.0, - consolidated_v1=False, - consolidated_v2=False, + consolidated=False, moss=False, moss_gamma=1.0, bayes_ucb=False, @@ -1562,6 +1561,10 @@ def __init__( seed_p2c_scheduler=False, op_stride=False, op_p2c=False, + # Versioned consolidated schedulers; `consolidated` above is the + # pre-v2 name of v1. Appended: positional signature. + consolidated_v1=False, + consolidated_v2=False, ): # Snapshot os.environ before anything below (or later in run()) can # write __AFL_DIST_SHM_ID / __AFL_SHM_ID / AFL_MAP_SIZE / LD_PRELOAD / @@ -3400,6 +3403,7 @@ def __init__( # Consolidated: flat Thompson with a category-shrunk prior and capped # evidence -- the single learner meant to replace the Elo portfolio # (see core/schedulers/op_consolidated_v1.py for the measurements). + consolidated_v1 = consolidated_v1 or consolidated self._use_consolidated_v1 = consolidated_v1 self._consolidated_v1 = None if consolidated_v1: diff --git a/tests/test_consolidated_v1_scheduler.py b/tests/test_consolidated_v1_scheduler.py index ff4cca32..a29bc14c 100644 --- a/tests/test_consolidated_v1_scheduler.py +++ b/tests/test_consolidated_v1_scheduler.py @@ -177,3 +177,51 @@ def bandit_mass(): assert "bandit" not in selectors assert f._consolidated_v1.bandit_stats()["consolidated_v1_pulls"] > 0 assert bandit_mass() > before, "the bandit stopped learning from rounds it did not select" + + +# -- pre-v2 compatibility (Copilot review on daedalus/fuzzer#53) ------------- + + +def test_regression_alias_keeps_legacy_stats_keys(): + """ConsolidatedScheduler behaves as v1 but reports its pre-v2 keys.""" + from fuzzer_tool.core.schedulers import ConsolidatedScheduler + + s = ConsolidatedScheduler(rng=RandPool(1)) + assert isinstance(s, ConsolidatedV1Scheduler) + s.record("bit_flip", True) + stats = s.bandit_stats() + assert stats["consolidated_pulls"] == 1 + assert stats["consolidated_top"][0][0] == "bit_flip" + assert not any(k.startswith("consolidated_v1") for k in stats) + + +def test_regression_alias_is_exported(): + import fuzzer_tool.core.schedulers as S + + assert "ConsolidatedScheduler" in S.__all__ + + +def test_regression_fuzzer_signature_keeps_positional_slots(): + """`consolidated` keeps its pre-v2 slot (just before `moss`); the + versioned flags are appended, so positional callers are not shifted.""" + import inspect + + from fuzzer_tool.services.fuzzer import Fuzzer + + params = list(inspect.signature(Fuzzer.__init__).parameters) + assert params.index("moss") == params.index("consolidated") + 1 + assert params[-2:] == ["consolidated_v1", "consolidated_v2"] + + +@pytest.mark.skipif(not _TARGET.exists(), reason="targets/test_target not built") +def test_regression_legacy_consolidated_kwarg_builds_v1(tmp_path): + from fuzzer_tool.services.fuzzer import Fuzzer + + corpus, crashes = tmp_path / "c", tmp_path / "k" + corpus.mkdir() + crashes.mkdir() + f = Fuzzer( + target=str(_TARGET), corpus_dir=str(corpus), crashes_dir=str(crashes), consolidated=True + ) + assert isinstance(f._consolidated_v1, ConsolidatedV1Scheduler) + assert f._use_consolidated_v1 diff --git a/tests/test_regression_scheduler_operator_reach.py b/tests/test_regression_scheduler_operator_reach.py index 32edae26..7c5313e6 100644 --- a/tests/test_regression_scheduler_operator_reach.py +++ b/tests/test_regression_scheduler_operator_reach.py @@ -137,6 +137,12 @@ def _ctx(_op): consolidated.select_op, consolidated.record, ), + ( + "ConsolidatedScheduler", + consolidated_legacy := S.ConsolidatedScheduler(), + consolidated_legacy.select_op, + consolidated_legacy.record, + ), ( "ConsolidatedV2Scheduler", consolidated_v2 := S.ConsolidatedV2Scheduler(),