From bf91186e6ecde9f0205efd6ff2e24cf8e3f6529b Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 00:47:40 +0000 Subject: [PATCH] Fix review findings from PRs #11, #45, #46 - #11: --mod-solving trace clobbered `targets`, dropping Katz. - #45: Doppler frames close for any corpus size (admission, not LRU eviction); flow ids compact and capped; keys per target; falls back to base without SHM; compile failure fails the wiring test. - #46: seed-arm ledgers gate on live-corpus membership, not seed_meta; EEVDF docs state O(n + log n) pick cost. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Rvj2gdMAa4HA8MGHULfDJE --- CHANGELOG.md | 5 + docs/DEEP_DIVE.md | 2 +- docs/TODO.md | 4 +- src/fuzzer_tool/core/fair_queue.py | 8 +- src/fuzzer_tool/core/power_doppler.py | 106 +++++++++++--- src/fuzzer_tool/services/fuzzer.py | 54 ++++++- tests/test_os_net_scheduler_wiring.py | 26 ++++ tests/test_power_doppler.py | 135 +++++++++++++++++- tests/test_regression_katz_resume_contract.py | 37 +++++ 9 files changed, 341 insertions(+), 36 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 63170af3..c3968339 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **`--mod-solving trace` clobbered `targets`** (`services/fuzzer.py`): the trace block reassigned the directed-targets parameter, disabling the Katz channel and directing at the fuzz target. Renamed the local. +- **Doppler never scored corpora > 64 seeds picked in turn** (`core/power_doppler.py`): LRU evicted every partial frame. New seeds now wait for a slot; abandoned frames are scored early and freed. Flow-edge ids are int64 arrays under a global cap (frozensets could reach hundreds of MiB). +- **Doppler mixed targets' edges** (`services/fuzzer.py`): multi-target frames are keyed per target. Without SHM, `--schedule doppler` now falls back to `base` with a warning instead of reporting enabled. +- **Seed-arm ledgers recorded standalone-QEA parents**: `seed_meta` is not corpus membership. Gated on the seed picker's cached corpus key map, memoized per parent. + - **EEVDF pick scanned ineligible flows** (`core/fair_queue.py`): one deadline heap popped every flow with an earlier deadline but `ve > V` (5000 pops at 5000 flows). Now a `ve` heap feeds a deadline heap; amortized O(log n). Test: `test_eevdf_pick_does_not_scan_ineligible_flows`. - **Seed-arm ledgers grew without bound**: non-corpus (Markov) parents were recorded, and departed seeds never left `ArmCounts`. Only corpus parents are recorded now; ledgers trim to 2x the live corpus. - **Round robin was O(n^2) per pick** (`seed_round_robin`, `op_round_robin`): `x in list` per registered arm. Set membership now: 103 ms -> 0.7 ms per pick at 5000 seeds. diff --git a/docs/DEEP_DIVE.md b/docs/DEEP_DIVE.md index a27ac00b..a275f894 100644 --- a/docs/DEEP_DIVE.md +++ b/docs/DEEP_DIVE.md @@ -85,7 +85,7 @@ For production and sensitive binaries using AFL family fuzzers is the best cours - **AFLGo SHM-tail distance channel**: compiled into EVERY shim-linked target since `__AFL_DISTANCE_MODE` defaulted to 1 (2026-08-24; `-D__AFL_DISTANCE_MODE=0` opts out). Inert unless directed mode uploads a distance table (`DistanceTableShm`/`__AFL_DIST_SHM_ID`): without one, sum/count stay 0 and every reader takes the Python-side path. `build_targets.sh --distance` additionally builds the trace-pc-instrumented `*_dist.so`/`*_dist_asan.so` variants, where the shim accumulates per-block distances in `__sanitizer_cov_trace_pc()` — the PC (relative to the dladdr-derived object base) probes an open-addressing table of `{key, dist}` entries (packed 12-byte layout; the 4-byte header holds the slot *capacity*, a power of two ≥ 2×entries so empty slots exist, and the builder hash-inserts at `key % capacity` with linear probing to mirror the shim's probe — uploaded by the fuzzer at startup via `DistanceTableShm`/`__AFL_DIST_SHM_ID`), accumulating sum/count into the 16-byte SHM **tail** (after the edge table: `u64 dist_sum`, `u64 dist_count`), written at reset, at process exit (subprocess runs never call reset), and per-iteration in in-process modes via `__afl_dist_flush` (direct_lite has no process boundary, so the runner flushes the tail after each `run_one`). Per-execution `avg_distance = sum/count/100` is read straight from the tail and preferred over Python-side computation; blocks without a table entry don't count (AFLGo semantics). The table's PC keys are recovered by scanning text for `call __sanitizer_cov_trace_pc` sites (modern clang emits no `__sancov_pcs` for trace-pc) mapped to valued blocks via the CFGs — `TargetDistance.pc_distance_table()`. The shim's sanitizer-coverage callbacks are hidden-visibility so a libasan LD_PRELOAD cannot interpose over them in PIE builds. `tools/gen_distance_table.py` emits the table as C or text for inspection. Without the table (count==0) everything degrades to the Python-side path. Works in subprocess, direct_lite, and persistent modes (ASAN direct_lite requires libasan preloaded at fuzzer-process start — the `use_direct_lite` gate). The periodic stats line shows live distance when directed mode is active: `dist: avg: min: max:` (or `no-data`). `build_targets.sh --distance` builds both `*_dist.so` (no-ASAN) and `*_dist_asan.so` (ASAN) variants with the cmplog shim linked in, so `--cmplog` keeps them in direct_lite mode. Startup reports `[*] Distance instrumentation: detected` when the target carries the channel (the shim's `__afl_dist_flush` or a defined `__sanitizer_cov_trace_pc`), mirroring the AFL-instrumentation check. With `--elo` in directed mode, `aflgo` joins the Elo-arbitrated seed-strategy pool — a distance-pure arm picking `P(seed) ∝ exp(-2·norm_dist)` (distinct from the generic `weighted` arm, which blends distance with speed/size/entropy). - **AFLGo distance-annealed schedule** (`--schedule go`, requires `--target-functions`): wires the precomputed `avg_distance` (per-seed distance to directed targets) and `_anneal_progress` (exploration/exploitation annealing variable) into `SeedScorer.score()` for mutation budget scaling. During exploration phase (`anneal_progress` ≈ 0): uniform energy. During exploitation phase (`anneal_progress` → 1): `energy *= exp(β · (1 - norm_dist))` where `β = anneal_progress * 5`, capping at 100x. Seeds near the target get exponentially more mutations as the campaign matures. Previously these metrics only influenced seed selection but not mutation intensity. - **Power schedules** (`--schedule base|fast|coe|rare|mopt|lin|quad|go|aflgo|entropic|doppler`): AFL++ power schedules ported to control mutation budget per seed via `SeedScorer`. Each schedule modifies a base score (100) by frequency-based factors. Honggfuzz-style novelty decay, density, fertility, freshness, and entropy factors are applied multiplicatively on top. `entropic` (libFuzzer `-entropic`) scales energy by `1 + log2(1 + rare)`, where `rare` is the larger of `rare_edge_count`/`tc_ref` already collected for RARE/honggfuzz scoring — an approximation of libFuzzer's feature-frequency Shannon entropy using signal the fuzzer already tracks. -- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB), 4096 scores (LRU). SHM coverage only; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. +- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed. Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. - **Favored set / cull_queue** (`core/schedules.py`, `services/fuzzer.py`): AFL-style `top_rated` minimal-set-cover selection. For each edge, the cheapest seed covering it is selected, then a greedy cover builds the favored set. FAST and COE schedules apply energy bonuses to favored seeds (`_fast_factor`, `_coe_factor`, `coe_skip`). `_cull_queue()` runs periodically during fuzzing and updates `self._favored`; the score call site passes `favored=(seed_key in self._favored)` so the scheduler actually uses it. - **Bayesian seed quality** (`--bayesian`): `BayesianSeedQuality` (`core/seed_quality.py`) maintains a Beta-Bernoulli posterior per seed over `P(outcome = new_coverage)`. Thompson sampling naturally balances explore/exploit without a manual temperature knob — unexplored seeds have high posterior variance and get sampled. The `record_outcome()` feedback loop is now wired in `fuzz_one()` (was previously a dead code path with all posteriors stuck at Beta(1,1)). State is persisted to `seed_quality.json` and restored on resume. diff --git a/docs/TODO.md b/docs/TODO.md index 01d002b5..334c2206 100644 --- a/docs/TODO.md +++ b/docs/TODO.md @@ -25,7 +25,7 @@ - [ ] **ptrace breakpoints only on dominator-tree leaves** (2026-09-26) — a hit block implies its dominators ran, so `ptrace_coverage.py` could place int3 on leaves only and infer the rest. Edges `(prev, curr)` are not implied the same way; measure breakpoint count and edge-set loss on fuzzgoat first. See `docs/learnings/2026-09-26-dominators-chk-worst-case.md`. ## Scheduling -- [ ] **A/B `--schedule doppler`** (2026-10-01) — power Doppler seed energy (`core/power_doppler.py`) is unit-tested and wired; never measured. Paired `bench_paired.py` vs `fast` on clang-built fuzzgoat (Hard Rule 52). Open: (a) ~100 µs/exec at 2k edges, mostly `get_edge_counts()` dict → numpy; a numpy accessor on `ShmCoverage` (exposes `_active_columns`, Hard Rule 34 — needs approval) removes it; (b) ensembles span picks, so a seed needs 32 picks to score — tune `ensemble` vs pick rate; (c) unused: `flow_edges()` (input-sensitive edge set) could feed position/operator targeting; (d) scores not persisted across `--resume`. +- [ ] **A/B `--schedule doppler`** (2026-10-01) — power Doppler seed energy (`core/power_doppler.py`) is unit-tested and wired; never measured. Paired `bench_paired.py` vs `fast` on clang-built fuzzgoat (Hard Rule 52). Open: (a) ~100 µs/exec at 2k edges, mostly `get_edge_counts()` dict → numpy; a numpy accessor on `ShmCoverage` (exposes `_active_columns`, Hard Rule 34 — needs approval) removes it; (b) ensembles span picks, so a seed needs 32 picks to score — tune `ensemble` vs pick rate; (c) unused: `flow_edges()` (input-sensitive edge set) could feed position/operator targeting; (d) scores not persisted across `--resume`; (e) `STALE_FRAMES = 4` (abandoned-frame window) and early-close scoring of partial frames are untuned — log `stats()['refused']` on fuzzgoat. - [ ] **A/B the OS / network scheduler ports** (2026-09-30) — `mlfq`, `stride`, `eevdf`, `bfq`, `sfq`, `codel`, `aimd`, `p2c` seed arms and `op_stride`, `op_p2c` are unit- and wiring-tested only. Run `tools/lib/bench_paired.py` per arm vs `seed_round_robin` / `drr` on fuzzgoat (clang, ASAN). Untuned: MLFQ allotment/boost, BFQ budget range, CoDel target/interval, AIMD alpha/beta/loss run. Open: (a) `sfq` groups only direct siblings (`parent_key`), and only under `--lineage`; a lineage-root flow would group whole families; (b) no arm persists state. - [ ] **A/B the effector/token/chunk/changed/rare_mask position arms** (2026-09-30) — wired and unit-tested; a 3k-exec fuzzgoat run confirms live signal (`changed`: 673 moved / 96 unmoved / 2231 unmeasured rounds; `rare_mask`: target on 328/3000 rounds) but fuzzgoat cannot rank position arms (see `pos_fibonacci` entry). Run `tools/lib/bench_paired.py` `pos-arena-{token,chunk,changed,rare-mask}` vs `pos-arena-uniform` on png_read (`chunk` needs a container corpus). Open: (a) `effector` had no drained map in 3k execs (SkipDet skipped 111/112 seeds) -- measure its reach on long runs before A/B, it is not subset-testable (needs the det stage); (b) `changed` credits only ~25% of rounds: parents without a recorded path hash (spliced/generated inputs) -- record one at admission or accept; (c) neither `changed` nor `rare_mask` persists state. - [ ] **Pre-existing, found 2026-09-30:** `tools/build_targets.sh` ASAN variants fail in the cloud container (non-ASAN builds fine). @@ -91,7 +91,7 @@ - [ ] **Point in-script cross-references at `tools/benchmark.py`** (2026-09-26) — harness docstrings and `bench.sh`/`bench_sweep.sh` headers still name each other by path (`tools/lib/bench_paired.py`, `tools/lib/bench_diff.py`). Correct, but a reader never learns the entry point from them. - [ ] **Three operators are still unexamined by the enumeration harness** (2026-09-12) — `avif_chunk_mutate`, `golomb` and `pgs_chunk_mutate` report `too_deep` once `_walk_operator` samples them in spread order, where the lexicographic walk called them `over_budget` and so hid the fact that they exceed `max_depth=16`. Raising the harness depth admits them; the depth cap exists so a truncated path is reported rather than silently walked, so raise it deliberately and re-measure the census rather than removing it. - [ ] **ffmpeg campaign throughput is admission-bound** (2026-09-26) — Docker `ffmpeg` image, full build, 4 cores: raw target 841 eps, campaign 3 eps with `-c --elo all --lineage-backtrack`, 6 eps with `-c` alone; 55 s to first exec. Not Docker (`--shm-size=2g` unchanged). Profile with `--profile-hotpath` on the same image. -- [ ] **`--mod-solving trace` clobbers `targets`** (2026-09-26) — `Fuzzer.__init__` reassigns the directed-mode `targets` parameter to `multi_targets or [target]` inside the SMT trace block, which then disables the Katz channel and feeds `_distance_targets` the fuzz target itself. Rename the local. +- [x] **`--mod-solving trace` clobbers `targets`** (2026-09-26, fixed 2026-10-01: local renamed `div_targets`; `test_regression_trace_mode_keeps_katz`) — `Fuzzer.__init__` reassigns the directed-mode `targets` parameter to `multi_targets or [target]` inside the SMT trace block, which then disables the Katz channel and feeds `_distance_targets` the fuzz target itself. Rename the local. - [ ] **Pre-existing red tests** (2026-09-25) — `test_katz_channel::TestScores::test_ensure_scores_caches_until_dirty_and_due`, `test_regression_build_lib_pairing::test_targets_are_actually_built` (red on base 2026-09-26); `test_regression_track_op_effect_coverage::test_every_ballot_name_is_mapped` (no kwargs for `fewa`, `softmax`, `topk`); `test_kruskal_count` / `test_seed_round_robin` `TestFuzzerWiring::test_constructor_flag_is_*last*` (constructor order); `test_regression_no_op_mutations::test_every_selectable_operator_is_reachable`. Red on master before the LRU/bayes-ucb change. ## Scheduling diff --git a/src/fuzzer_tool/core/fair_queue.py b/src/fuzzer_tool/core/fair_queue.py index 14e9df32..5a7bd684 100644 --- a/src/fuzzer_tool/core/fair_queue.py +++ b/src/fuzzer_tool/core/fair_queue.py @@ -15,8 +15,9 @@ Stride counts O(log n) deterministic tickets, seeds/ops EEVDF cost O(log n)* lag-bounded, new flows join at V -(*) amortized: each flow crosses from the ve heap to the deadline heap once -per service. +(*) heap work only, amortized: each flow crosses from the ve heap to the +deadline heap once per service. The whole pick is O(n + log n): it also +compares the caller's flow list against the last one. Picks are O(n) because callers hand over the live flow set every call; the flow sets here (targets, corpus) are rebuilt per pick by the caller anyway. @@ -336,7 +337,8 @@ class EEVDF: a new flow joins at ``V`` (lag 0), so it neither catches up nor waits. Flat cost and weight are plain round robin. - Two heaps keep a pick amortized O(log n): deadline order is not + Two heaps keep a pick's heap work amortized O(log n) (the pick itself + stays O(n + log n), see module note): deadline order is not eligibility order, so one deadline heap would pop every ineligible flow with an earlier deadline. ``pending`` holds flows by ``ve``; those that fall at or below ``V`` move to ``ready``, ordered by deadline:: diff --git a/src/fuzzer_tool/core/power_doppler.py b/src/fuzzer_tool/core/power_doppler.py index e47e3cea..4e5f4900 100644 --- a/src/fuzzer_tool/core/power_doppler.py +++ b/src/fuzzer_tool/core/power_doppler.py @@ -36,8 +36,12 @@ DEFAULT_MAX_SEEDS = 64 # Edge columns per ensemble; 64 x 32 x 2048 x 4 B bounds the matrices at 16 MiB. DEFAULT_MAX_EDGES = 2048 -# Closed-ensemble scores kept (two floats and a frozenset each). +# Closed-ensemble scores kept (power and flow-edge ids each). DEFAULT_MAX_SCORES = 4096 +# Flow-edge ids kept across all scores: 8 B each bounds them at 8 MiB. +DEFAULT_MAX_FLOW_IDS = 1 << 20 +# Open frame untouched for this many full frames of samples is abandoned. +STALE_FRAMES = 4 # CFAR false-alarm probability per edge. FALSE_ALARM = 1e-3 # A component spread over at least this share of the seed's edges is a flash. @@ -126,7 +130,7 @@ def doppler_power(x: np.ndarray) -> tuple[float, np.ndarray, int]: class _Ensemble: """One seed's open slow-time matrix with an edge-id -> column map.""" - __slots__ = ("x", "n", "m", "cap", "ids", "skeys", "sslots", "last") + __slots__ = ("x", "n", "m", "cap", "ids", "skeys", "sslots", "last", "touched") def __init__(self, length: int, cap: int) -> None: cols = min(INITIAL_COLS, cap) @@ -140,6 +144,7 @@ def __init__(self, length: int, cap: int) -> None: # Previous sample's (ids, columns, kept): mutants mostly replay the # seed's path in the same SHM order, so the lookup is usually reused. self.last: tuple[np.ndarray, np.ndarray, np.ndarray] | None = None + self.touched = 0 # PowerDoppler tick of the last sample def _grow(self, need: int) -> None: cols = self.x.shape[1] @@ -198,9 +203,12 @@ class PowerDoppler: Args: ensemble: Mutants per closed ensemble. - max_seeds: Open ensembles kept (LRU). + max_seeds: Open ensembles kept. When full, new seeds wait for a + slot rather than evict partial frames: LRU eviction never closed + a frame once more than max_seeds seeds were picked in turn. max_edges: Edge columns per ensemble; extra edges are dropped. max_scores: Closed-ensemble scores kept (LRU). + max_flow_ids: Flow-edge ids kept across all scores (LRU). """ def __init__( @@ -209,23 +217,36 @@ def __init__( max_seeds: int = DEFAULT_MAX_SEEDS, max_edges: int = DEFAULT_MAX_EDGES, max_scores: int = DEFAULT_MAX_SCORES, + max_flow_ids: int = DEFAULT_MAX_FLOW_IDS, ) -> None: if ensemble < _MIN_ENSEMBLE: raise ValueError(f"ensemble must be >= {_MIN_ENSEMBLE}, got {ensemble}") + if max_seeds < 1: + raise ValueError(f"max_seeds must be >= 1, got {max_seeds}") if max_edges < 1: raise ValueError(f"max_edges must be >= 1, got {max_edges}") + if max_scores < 1: + raise ValueError(f"max_scores must be >= 1, got {max_scores}") + if max_flow_ids < 0: + raise ValueError(f"max_flow_ids must be >= 0, got {max_flow_ids}") self._ensemble = ensemble + self._max_seeds = max_seeds self._max_edges = max_edges - self._open: LRUCache = LRUCache(max_seeds) - self._scores: LRUCache = LRUCache(max_scores, on_evict=self._evicted) + self._max_scores = max_scores + self._max_flow_ids = max_flow_ids + self._stale_after = max_seeds * ensemble * STALE_FRAMES + # Insertion order is recency order; capacity is enforced by hand + # (_admit, _trim) so evicted frames can be scored / ids uncounted. + self._open: dict[str, _Ensemble] = {} + self._scores: LRUCache = LRUCache(max_scores + 1) + self._flow_ids = 0 self._max_power = 0.0 self._max_stale = False self._closed = 0 self._dropped = 0 + self._refused = 0 self._samples = 0 - - def _evicted(self, _key) -> None: - self._max_stale = True + self._ticks = 0 def observe(self, seed_key: str, hits: Mapping[int, int]) -> None: """Add one mutant execution of *seed_key*: ``{edge_id: hit count}``.""" @@ -233,13 +254,19 @@ def observe(self, seed_key: str, hits: Mapping[int, int]) -> None: if not n: return - ids = np.fromiter(hits.keys(), dtype=np.int64, count=n) - counts = np.fromiter(hits.values(), dtype=np.int64, count=n) - ens = self._open.get(seed_key) + self._ticks += 1 + ens = self._open.pop(seed_key, None) + if ens is None: + ens = self._admit() if ens is None: - ens = _Ensemble(self._ensemble, self._max_edges) - self._open[seed_key] = ens + self._refused += 1 + return + # Re-insert: most recent last. + self._open[seed_key] = ens + ens.touched = self._ticks + ids = np.fromiter(hits.keys(), dtype=np.int64, count=n) + counts = np.fromiter(hits.values(), dtype=np.int64, count=n) cols, kept = ens.columns(ids) self._dropped += int(ids.size - cols.size) ens.x[ens.n, cols] = np.log2(1.0 + counts[kept]) @@ -252,14 +279,53 @@ def observe(self, seed_key: str, hits: Mapping[int, int]) -> None: del self._open[seed_key] self._close(seed_key, ens) + def _admit(self) -> _Ensemble | None: + """Fresh frame if a slot is free or the oldest frame is abandoned. + + Waiting instead of evicting keeps open frames progressing whatever + the corpus size, e.g. 200 seeds picked in turn into 64 slots: the + first 64 fill and close, then the next 64 get their slots. + """ + if len(self._open) >= self._max_seeds: + key = next(iter(self._open)) + old = self._open[key] + if self._ticks - old.touched < self._stale_after: + return None + + # Abandoned: score what it has rather than throw it away. + del self._open[key] + if old.n >= _MIN_ENSEMBLE: + self._close(key, old) + return _Ensemble(self._ensemble, self._max_edges) + def _close(self, seed_key: str, ens: _Ensemble) -> None: - power, flow, _ = doppler_power(ens.x[:, : ens.m].astype(np.float64)) - old = self._scores.get(seed_key) - if old is not None and old[0] >= self._max_power: - self._max_stale = True - self._scores[seed_key] = (power, frozenset(ens.ids[: ens.m][flow].tolist())) + power, flow, _ = doppler_power(ens.x[: ens.n, : ens.m].astype(np.float64)) + old = self._scores.pop(seed_key, None) + if old is not None: + self._drop(old) + flow_ids = np.sort(ens.ids[: ens.m][flow]) + self._scores[seed_key] = (power, flow_ids) + self._flow_ids += flow_ids.size self._max_power = max(self._max_power, power) self._closed += 1 + self._trim() + + def _drop(self, score: tuple[float, np.ndarray]) -> None: + """Forget *score*'s ids; the peak is recomputed if it was the max.""" + self._flow_ids -= score[1].size + if score[0] >= self._max_power: + self._max_stale = True + + def _trim(self) -> None: + """Evict LRU scores past the count or flow-id budget. + + Flow ids live in compact int64 arrays (8 B per id; a frozenset of + ints costs ~70 B), and their total is capped as well as the count. + """ + scores = self._scores + while len(scores) > self._max_scores or self._flow_ids > self._max_flow_ids: + _, score = scores.popitem(last=False) + self._drop(score) def _peak(self) -> float: if self._max_stale: @@ -278,7 +344,7 @@ def energy(self, seed_key: str) -> float: def flow_edges(self, seed_key: str) -> frozenset[int]: """Input-sensitive edges of the seed's last closed ensemble.""" score = self._scores.get(seed_key) - return frozenset() if score is None else score[1] + return frozenset() if score is None else frozenset(score[1].tolist()) def stats(self) -> dict[str, float]: """Diagnostics for reports.""" @@ -288,5 +354,7 @@ def stats(self) -> dict[str, float]: "scored": len(self._scores), "ensembles": self._closed, "dropped_edges": self._dropped, + "refused": self._refused, + "flow_ids": self._flow_ids, "max_power": self._peak(), } diff --git a/src/fuzzer_tool/services/fuzzer.py b/src/fuzzer_tool/services/fuzzer.py index 230e7716..211de7c4 100644 --- a/src/fuzzer_tool/services/fuzzer.py +++ b/src/fuzzer_tool/services/fuzzer.py @@ -2001,10 +2001,10 @@ def __init__( if mod_solving == "trace": from fuzzer_tool.core.elf import extract_div_constants - targets = multi_targets or [target] + div_targets = multi_targets or [target] div_map: dict[int, int] = {} weak_set: set[int] = set() - for t in targets: + for t in div_targets: try: d, w = extract_div_constants(t) div_map.update(d) @@ -2336,6 +2336,12 @@ def __init__( self._perf_counters = None self.hw_perf = False + # Doppler samples SHM hit counts only: without SHM it would score + # nothing and silently act as 'base' while reported as enabled. + if schedule == "doppler" and self.shm_cov is None: + print("[!] Doppler schedule: needs AFL SHM coverage; falling back to 'base'") + schedule = "base" + # Seed-level energy multiplier: scales mutations_per_input per seed self._seed_scorer = SeedScorer( schedule=schedule or "base", @@ -2739,6 +2745,8 @@ def __init__( ) if arm is not None ) + # (parent, corpus size, in corpus?): one membership test per pick. + self._parent_memo: tuple[bytes, int, bool] | None = None if self._seed_os_arms: log.info("OS/network seed arms enabled: %d", len(self._seed_os_arms)) # LST override: no seed waits more than lst_revisit seconds between @@ -4521,24 +4529,52 @@ def _seed_residual_outcomes(self) -> dict[str, float]: def _record_seed_os_arms(self, parent: bytes, success: bool, weight: float) -> None: """Feed one corpus parent's outcome to every enabled OS / network seed arm. - Parents outside the corpus (Markov-generated inputs) are skipped: they - are never candidates, and each would leave a permanent ledger entry. + Parents outside the corpus (Markov-generated inputs, standalone-QEA + collapses that live only in seed_meta) are skipped: they are never + candidates, and each would leave a permanent ledger entry. Off-policy like the canary feed: every arm sees every parent, whichever strategy picked it, since MLFQ demotion, CoDel staleness and AIMD windows are properties of the seed, not of the picker. """ - if not self._seed_os_arms or parent not in self.seed_meta: + if not self._seed_os_arms or not self._in_corpus(parent): return key = self._seed_key(parent) for arm in self._seed_os_arms: arm.record(key, success=success, weight=weight) + def _in_corpus(self, parent: bytes) -> bool: + """Live-corpus membership of *parent*, memoized across its executions. + + Checked against the seed picker's cached key map (rebuilt only on + corpus change), not ``seed_meta``. The memo is keyed on corpus size + too, so a parent admitted mid-pick is seen. + """ + n = len(self.corpus) + memo = self._parent_memo + if memo is not None and memo[0] is parent and memo[1] == n: + return memo[2] + + key_to_seed, _ = self._seed_picker._corpus_keys() + hit = self._seed_key(parent) in key_to_seed + self._parent_memo = (parent, n, hit) + return hit + def _seed_key(self, data: bytes) -> str: """Return content hash for *data*.""" return self._corpus_manager.seed_key(data) + def _doppler_key(self, seed_key: str) -> str: + """Doppler frame key: SHM edge ids are per target, so namespace them. + + Multi-target: ``"a.bin\0"`` and ``"b.bin\0"`` are two + frames; mixing them would merge unrelated edges sharing an id. + """ + if not self.multi_targets: + return seed_key + return f"{self.target}\0{seed_key}" + def _boost_corpus_sizes(self) -> None: """Resize each corpus seed to a target size drawn from N(boost_mean, boost_std), clamped to [1, corpus_boost]. Target sizes are shuffled to avoid ordering bias @@ -6377,7 +6413,9 @@ def fuzz_one(self, data: bytes) -> bool: and not is_crash and not is_timeout ): - self._doppler.observe(self._seed_key(data), scanned_shm.get_edge_counts()) + self._doppler.observe( + self._doppler_key(self._seed_key(data)), scanned_shm.get_edge_counts() + ) # Performance novelty: an edge whose trip count grew substantially # past anything seen before. The hit-count buckets saturate (129 and @@ -9799,7 +9837,9 @@ def run(self, iterations=0, max_execs=0): else 0.0 ), doppler_energy=( - self._doppler.energy(seed_key) if self._doppler is not None else 0.0 + self._doppler.energy(self._doppler_key(seed_key)) + if self._doppler is not None + else 0.0 ), **hf_kwargs, ) diff --git a/tests/test_os_net_scheduler_wiring.py b/tests/test_os_net_scheduler_wiring.py index 5e9ada1e..d8fca0ca 100644 --- a/tests/test_os_net_scheduler_wiring.py +++ b/tests/test_os_net_scheduler_wiring.py @@ -240,6 +240,7 @@ def test_seed_outcome_reaches_every_arm(self, tmp_path): """Falsification: one recorded corpus outcome lands in each enabled arm's ledger.""" kwargs = {kw: True for kw, _a, _c in SEED_ARMS.values()} f = _real_fuzzer(tmp_path, **kwargs) + f.corpus.append(SEED_A) f.seed_meta[SEED_A] = {} f._record_seed_os_arms(SEED_A, success=True, weight=1.0) @@ -256,6 +257,31 @@ def test_regression_non_corpus_parent_not_recorded(self, tmp_path): for arm in f._seed_os_arms: assert arm.bandit_stats() == {} + def test_regression_seed_meta_without_corpus_not_recorded(self, tmp_path): + """Falsification (PR #46 review): standalone QEA fills seed_meta, not the corpus.""" + kwargs = {kw: True for kw, _a, _c in SEED_ARMS.values()} + f = _real_fuzzer(tmp_path, **kwargs) + f.seed_meta[SEED_B] = {} + assert SEED_B not in f.corpus + + f._record_seed_os_arms(SEED_B, success=True, weight=1.0) + + for arm in f._seed_os_arms: + assert arm.bandit_stats() == {} + + def test_adversarial_parent_admitted_later_is_recorded(self, tmp_path): + """A cached 'not in corpus' verdict must not outlive the parent's admission.""" + kwargs = {kw: True for kw, _a, _c in SEED_ARMS.values()} + f = _real_fuzzer(tmp_path, **kwargs) + f._record_seed_os_arms(SEED_C, success=True, weight=1.0) + + f.corpus.append(SEED_C) + f._record_seed_os_arms(SEED_C, success=True, weight=1.0) + + key = f._seed_key(SEED_C) + for arm in f._seed_os_arms: + assert arm.bandit_stats() == {key: (1.0, 0.0)} + def test_op_arms_get_priors_registration_and_records(self): """The op arms sit in _register_arms and in fuzz_one's shared record loop.""" from fuzzer_tool.services import fuzzer as fz diff --git a/tests/test_power_doppler.py b/tests/test_power_doppler.py index 1e75d5ab..9218b8ba 100644 --- a/tests/test_power_doppler.py +++ b/tests/test_power_doppler.py @@ -15,7 +15,12 @@ import numpy as np import pytest -from fuzzer_tool.core.power_doppler import PowerDoppler, _components, doppler_power +from fuzzer_tool.core.power_doppler import ( + STALE_FRAMES, + PowerDoppler, + _components, + doppler_power, +) from fuzzer_tool.core.schedules import SeedScorer ROOT = Path(__file__).resolve().parent.parent @@ -170,6 +175,67 @@ def test_memory_bounded_by_edge_cap(self): assert pd.stats()["dropped_edges"] == PATH - cap + def test_regression_cyclic_corpus_beyond_seed_cap_closes_frames(self): + # Falsification: 3x more seeds than open slots, one mutant per pick. + # LRU eviction of partial frames left every seed unscored forever. + cap, n_seeds = 4, 12 + pd = PowerDoppler(ensemble=N, max_seeds=cap) + ids = list(range(PATH)) + for _ in range(N): + for k in range(n_seeds): + _feed(pd, f"s{k}", _static(n=1), ids) + + assert pd.stats()["ensembles"] >= cap + assert pd.stats()["open"] <= cap + + def test_regression_abandoned_frame_yields_its_slot(self): + # Adversarial: a seed never picked again must not pin its slot; its + # partial frame (>= 3 samples) is scored on the way out. + pd = PowerDoppler(ensemble=N, max_seeds=1) + ids = list(range(PATH + 1)) + flow = _with_flow(_static(), _toggle()) + _feed(pd, "gone", flow[:3], ids) + stale = N * STALE_FRAMES + + _feed(pd, "next", _static(n=stale + N), ids[:PATH]) + + assert pd.flow_edges("gone") == frozenset({PATH}) + assert pd.stats()["ensembles"] == 2 + + def test_adversarial_tiny_partial_frame_not_scored(self): + # Fewer samples than a valid ensemble: dropped, never scored. + pd = PowerDoppler(ensemble=N, max_seeds=1) + ids = list(range(PATH)) + _feed(pd, "gone", _static(n=2), ids) + + _feed(pd, "next", _static(n=N * STALE_FRAMES + N), ids) + + assert pd.stats()["ensembles"] == 1 + assert pd.energy("gone") == 0.0 + + def test_regression_flow_ids_bounded_in_total(self): + # Falsification: retained flow-edge ids stay under the global budget, + # however many seeds are scored. + budget = 3 + pd = PowerDoppler(ensemble=N, max_flow_ids=budget) + rows = np.column_stack([_static(m=PATH)] + [_toggle()] * 2) + ids = list(range(PATH + 2)) + for k in range(5): + _feed(pd, f"s{k}", rows, ids) + + assert pd.stats()["flow_ids"] <= budget + assert pd.flow_edges("s4") == frozenset({PATH, PATH + 1}) + + def test_adversarial_rescore_does_not_leak_flow_budget(self): + # Re-closing the same seed replaces, not adds to, its retained ids. + pd = PowerDoppler(ensemble=N) + rows = _with_flow(_static(), _toggle()) + ids = list(range(PATH + 1)) + for _ in range(4): + _feed(pd, "s", rows, ids) + + assert pd.stats()["flow_ids"] == 1 + def test_bad_parameters_rejected(self): with pytest.raises(ValueError): PowerDoppler(ensemble=2) @@ -177,6 +243,8 @@ def test_bad_parameters_rejected(self): PowerDoppler(max_seeds=0) with pytest.raises(ValueError): PowerDoppler(max_edges=0) + with pytest.raises(ValueError): + PowerDoppler(max_flow_ids=-1) class TestDopplerSchedule: @@ -236,8 +304,8 @@ def test_fuzz_loop_feeds_doppler(tmp_path): capture_output=True, text=True, ) - if r.returncode != 0: - pytest.skip(f"driver failed to build: {r.stderr[:300]}") + # clang presence is the skipif above; a failed build is a shim regression. + assert r.returncode == 0, r.stderr[:300] corpus = tmp_path / "corpus" corpus.mkdir() (corpus / "a").write_bytes(b"hello world") @@ -259,5 +327,64 @@ def test_fuzz_loop_feeds_doppler(tmp_path): f.run(iterations=100_000, max_execs=600) stats = f._doppler.stats() - assert stats["samples"] >= 500 + # Every execution reaches Doppler: accepted, or waiting for a free slot. + assert stats["samples"] + stats["refused"] >= 500 assert stats["ensembles"] > 0 + + +def _fuzzer_no_target(tmp_path, **kw): + from fuzzer_tool.services.fuzzer import Fuzzer + + corpus, crashes = tmp_path / "c", tmp_path / "k" + corpus.mkdir() + crashes.mkdir() + return Fuzzer( + target="targets/test_target", + corpus_dir=str(corpus), + crashes_dir=str(crashes), + max_len=64, + schedule="doppler", + **kw, + ) + + +def test_regression_doppler_without_shm_is_disabled(tmp_path, capsys): + """Falsification: no SHM means no samples; disable and say so.""" + f = _fuzzer_no_target(tmp_path, use_coverage=False) + + assert f._doppler is None + assert f._seed_scorer.schedule == "base" + assert "doppler" in capsys.readouterr().out.lower() + + +def test_doppler_with_shm_stays_enabled(tmp_path, monkeypatch): + """Adversarial: the gate must not disable Doppler when SHM is live.""" + monkeypatch.setattr("fuzzer_tool.core.elf.sancov_guard_status", lambda _t: "present") + monkeypatch.setattr("fuzzer_tool.core.elf.detect_ctx_bits", lambda _t: 4) + f = _fuzzer_no_target(tmp_path, use_coverage=True) + + assert f.shm_cov is not None + assert f._doppler is not None + assert f._seed_scorer.schedule == "doppler" + + +def _key_of(target, multi): + from types import SimpleNamespace + + from fuzzer_tool.services.fuzzer import Fuzzer + + owner = SimpleNamespace(target=target, multi_targets=multi) + return Fuzzer._doppler_key(owner, "seed") + + +def test_regression_doppler_key_namespaced_per_target(): + """Falsification: SHM edge ids are per target; one seed, two targets, two frames.""" + multi = ["a.bin", "b.bin"] + + assert _key_of("a.bin", multi) != _key_of("b.bin", multi) + assert _key_of("a.bin", multi) == _key_of("a.bin", multi) + + +def test_doppler_key_single_target_is_seed_key(): + """Adversarial: single-target runs keep the plain seed key.""" + assert _key_of("a.bin", None) == "seed" diff --git a/tests/test_regression_katz_resume_contract.py b/tests/test_regression_katz_resume_contract.py index 0ddc1de3..88b176cf 100644 --- a/tests/test_regression_katz_resume_contract.py +++ b/tests/test_regression_katz_resume_contract.py @@ -82,3 +82,40 @@ def test_regression_resume_without_katz_refused(tmp_path, monkeypatch): _with_katz(monkeypatch, None) with pytest.raises(RuntimeError, match="node_channel"): _fuzzer(tmp_path, resume=True) + + +class _FakeZ3: + """Stands in for Z3Solver: no z3 install needed.""" + + _available = True + + def __init__(self, **_kw): + pass + + +def _trace_mode(monkeypatch) -> None: + monkeypatch.setattr("fuzzer_tool.core.smt_solver.Z3Solver", _FakeZ3) + monkeypatch.setattr("fuzzer_tool.core.elf.extract_div_constants", lambda _t: ({}, set())) + + +def test_regression_trace_mode_keeps_katz(tmp_path, monkeypatch): + """Falsification: --mod-solving trace must not clobber `targets` and drop Katz.""" + _trace_mode(monkeypatch) + _with_katz(monkeypatch, _FakeKatz()) + + f = _fuzzer(tmp_path, enable_smt_z3=True, mod_solving="trace") + + assert f._katz_channel is not None + assert f._distance_targets is None + + +def test_regression_trace_mode_keeps_directed(tmp_path, monkeypatch): + """Adversarial: directed targets survive trace mode and still exclude Katz.""" + _trace_mode(monkeypatch) + _with_katz(monkeypatch, _FakeKatz()) + directed = ["0x401000"] + + f = _fuzzer(tmp_path, enable_smt_z3=True, mod_solving="trace", targets=directed) + + assert f._katz_channel is None + assert f._distance_targets == directed