From 0b2ef4443d5096ad5bb6561c397a847f949a9532 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 14:23:47 +0000 Subject: [PATCH] Fix PR #48 review finding: Doppler horizon compounding Each returning dropped seed doubled one shared horizon (N returns -> 2^N), disabling abandonment. The dropped-key memory (max_seeds keys) also forgot keys before large corpora cycled back, so the horizon never widened at all. Now horizon = max(horizon, 2 x measured revisit gap), with 2^15 remembered keys. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Rvj2gdMAa4HA8MGHULfDJE --- CHANGELOG.md | 2 ++ docs/DEEP_DIVE.md | 2 +- src/fuzzer_tool/core/power_doppler.py | 28 +++++++++++++++++++-------- tests/test_power_doppler.py | 18 ++++++++++++++++- 4 files changed, 40 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d1c75c27..9b4b6719 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **Doppler horizon compounded per returning seed** (`core/power_doppler.py`): each return doubled one shared horizon (N returns → 2^N), disabling abandonment; and the dropped-key memory (`max_seeds` keys) forgot keys before large corpora cycled back. Horizon is now `max(horizon, 2 × measured gap)`; memory is 2¹⁵ keys. + - **Doppler horizon thrashed on slow corpus cycles** (`core/power_doppler.py`): a cycle longer than the fixed abandon horizon dropped every frame one tick before its seed returned. The horizon now doubles when a dropped seed comes back. - **Stale corpus-membership memo** (`services/fuzzer.py`): keyed on corpus length, so an in-place trim of the parent kept a "member" verdict. Hits are now re-validated by slot identity; misses are not memoized. diff --git a/docs/DEEP_DIVE.md b/docs/DEEP_DIVE.md index c71a0cb8..7847d729 100644 --- a/docs/DEEP_DIVE.md +++ b/docs/DEEP_DIVE.md @@ -85,7 +85,7 @@ For production and sensitive binaries using AFL family fuzzers is the best cours - **AFLGo SHM-tail distance channel**: compiled into EVERY shim-linked target since `__AFL_DISTANCE_MODE` defaulted to 1 (2026-08-24; `-D__AFL_DISTANCE_MODE=0` opts out). Inert unless directed mode uploads a distance table (`DistanceTableShm`/`__AFL_DIST_SHM_ID`): without one, sum/count stay 0 and every reader takes the Python-side path. `build_targets.sh --distance` additionally builds the trace-pc-instrumented `*_dist.so`/`*_dist_asan.so` variants, where the shim accumulates per-block distances in `__sanitizer_cov_trace_pc()` — the PC (relative to the dladdr-derived object base) probes an open-addressing table of `{key, dist}` entries (packed 12-byte layout; the 4-byte header holds the slot *capacity*, a power of two ≥ 2×entries so empty slots exist, and the builder hash-inserts at `key % capacity` with linear probing to mirror the shim's probe — uploaded by the fuzzer at startup via `DistanceTableShm`/`__AFL_DIST_SHM_ID`), accumulating sum/count into the 16-byte SHM **tail** (after the edge table: `u64 dist_sum`, `u64 dist_count`), written at reset, at process exit (subprocess runs never call reset), and per-iteration in in-process modes via `__afl_dist_flush` (direct_lite has no process boundary, so the runner flushes the tail after each `run_one`). Per-execution `avg_distance = sum/count/100` is read straight from the tail and preferred over Python-side computation; blocks without a table entry don't count (AFLGo semantics). The table's PC keys are recovered by scanning text for `call __sanitizer_cov_trace_pc` sites (modern clang emits no `__sancov_pcs` for trace-pc) mapped to valued blocks via the CFGs — `TargetDistance.pc_distance_table()`. The shim's sanitizer-coverage callbacks are hidden-visibility so a libasan LD_PRELOAD cannot interpose over them in PIE builds. `tools/gen_distance_table.py` emits the table as C or text for inspection. Without the table (count==0) everything degrades to the Python-side path. Works in subprocess, direct_lite, and persistent modes (ASAN direct_lite requires libasan preloaded at fuzzer-process start — the `use_direct_lite` gate). The periodic stats line shows live distance when directed mode is active: `dist: avg: min: max:` (or `no-data`). `build_targets.sh --distance` builds both `*_dist.so` (no-ASAN) and `*_dist_asan.so` (ASAN) variants with the cmplog shim linked in, so `--cmplog` keeps them in direct_lite mode. Startup reports `[*] Distance instrumentation: detected` when the target carries the channel (the shim's `__afl_dist_flush` or a defined `__sanitizer_cov_trace_pc`), mirroring the AFL-instrumentation check. With `--elo` in directed mode, `aflgo` joins the Elo-arbitrated seed-strategy pool — a distance-pure arm picking `P(seed) ∝ exp(-2·norm_dist)` (distinct from the generic `weighted` arm, which blends distance with speed/size/entropy). - **AFLGo distance-annealed schedule** (`--schedule go`, requires `--target-functions`): wires the precomputed `avg_distance` (per-seed distance to directed targets) and `_anneal_progress` (exploration/exploitation annealing variable) into `SeedScorer.score()` for mutation budget scaling. During exploration phase (`anneal_progress` ≈ 0): uniform energy. During exploitation phase (`anneal_progress` → 1): `energy *= exp(β · (1 - norm_dist))` where `β = anneal_progress * 5`, capping at 100x. Seeds near the target get exponentially more mutations as the campaign matures. Previously these metrics only influenced seed selection but not mutation intensity. - **Power schedules** (`--schedule base|fast|coe|rare|mopt|lin|quad|go|aflgo|entropic|doppler`): AFL++ power schedules ported to control mutation budget per seed via `SeedScorer`. Each schedule modifies a base score (100) by frequency-based factors. Honggfuzz-style novelty decay, density, fertility, freshness, and entropy factors are applied multiplicatively on top. `entropic` (libFuzzer `-entropic`) scales energy by `1 + log2(1 + rare)`, where `rare` is the larger of `rare_edge_count`/`tc_ref` already collected for RARE/honggfuzz scoring — an approximation of libFuzzer's feature-frequency Shannon entropy using signal the fuzzer already tracks. -- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; that horizon doubles whenever a dropped seed returns (slow, not abandoned), so it converges past the corpus revisit time. Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. +- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; when a dropped seed returns (slow, not abandoned) the horizon rises to 2× its measured revisit gap — tracking the slowest real gap, never compounding per return. Dropped keys remembered: 2¹⁵ (LRU, ~5 MiB). Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. - **Favored set / cull_queue** (`core/schedules.py`, `services/fuzzer.py`): AFL-style `top_rated` minimal-set-cover selection. For each edge, the cheapest seed covering it is selected, then a greedy cover builds the favored set. FAST and COE schedules apply energy bonuses to favored seeds (`_fast_factor`, `_coe_factor`, `coe_skip`). `_cull_queue()` runs periodically during fuzzing and updates `self._favored`; the score call site passes `favored=(seed_key in self._favored)` so the scheduler actually uses it. - **Bayesian seed quality** (`--bayesian`): `BayesianSeedQuality` (`core/seed_quality.py`) maintains a Beta-Bernoulli posterior per seed over `P(outcome = new_coverage)`. Thompson sampling naturally balances explore/exploit without a manual temperature knob — unexplored seeds have high posterior variance and get sampled. The `record_outcome()` feedback loop is now wired in `fuzz_one()` (was previously a dead code path with all posteriors stuck at Beta(1,1)). State is persisted to `seed_quality.json` and restored on resume. diff --git a/src/fuzzer_tool/core/power_doppler.py b/src/fuzzer_tool/core/power_doppler.py index fd61cb2e..837a78fa 100644 --- a/src/fuzzer_tool/core/power_doppler.py +++ b/src/fuzzer_tool/core/power_doppler.py @@ -41,8 +41,13 @@ # Flow-edge ids kept across all scores: 8 B each bounds them at 8 MiB. DEFAULT_MAX_FLOW_IDS = 1 << 20 # Open frame untouched for this many full frames of samples is abandoned -# (initial horizon; doubles whenever a dropped seed comes back). +# (initial horizon; widened to REVISIT_MARGIN x any gap a dropped seed proves). STALE_FRAMES = 4 +# Horizon headroom over the slowest observed revisit gap. +REVISIT_MARGIN = 2 +# Dropped keys remembered (key -> last-touch tick); must outlast a corpus +# cycle's worth of drops. ~150 B each bounds it near 5 MiB. +DEFAULT_MAX_DROPPED = 1 << 15 # CFAR false-alarm probability per edge. FALSE_ALARM = 1e-3 # A component spread over at least this share of the seed's edges is a flash. @@ -210,6 +215,7 @@ class PowerDoppler: max_edges: Edge columns per ensemble; extra edges are dropped. max_scores: Closed-ensemble scores kept (LRU). max_flow_ids: Flow-edge ids kept across all scores (LRU). + max_dropped: Dropped-frame keys remembered (LRU) to measure revisit gaps. """ def __init__( @@ -219,6 +225,7 @@ def __init__( max_edges: int = DEFAULT_MAX_EDGES, max_scores: int = DEFAULT_MAX_SCORES, max_flow_ids: int = DEFAULT_MAX_FLOW_IDS, + max_dropped: int = DEFAULT_MAX_DROPPED, ) -> None: if ensemble < _MIN_ENSEMBLE: raise ValueError(f"ensemble must be >= {_MIN_ENSEMBLE}, got {ensemble}") @@ -230,14 +237,16 @@ def __init__( raise ValueError(f"max_scores must be >= 1, got {max_scores}") if max_flow_ids < 0: raise ValueError(f"max_flow_ids must be >= 0, got {max_flow_ids}") + if max_dropped < 1: + raise ValueError(f"max_dropped must be >= 1, got {max_dropped}") self._ensemble = ensemble self._max_seeds = max_seeds self._max_edges = max_edges self._max_scores = max_scores self._max_flow_ids = max_flow_ids self._stale_after = max_seeds * ensemble * STALE_FRAMES - # Keys whose frames were dropped as abandoned; bounded like _open. - self._dropped_keys: LRUCache = LRUCache(max_seeds) + # Dropped-as-abandoned key -> its frame's last-touch tick. + self._dropped_keys: LRUCache = LRUCache(max_dropped) # Insertion order is recency order; capacity is enforced by hand # (_admit, _trim) so evicted frames can be scored / ids uncounted. self._open: dict[str, _Ensemble] = {} @@ -288,13 +297,16 @@ def _revisit(self, seed_key: str) -> None: A fixed horizon thrashes once the corpus cycle outlasts it, e.g. 13 seeds in turn vs a 12-tick horizon: every frame is dropped one tick - before its seed returns. Doubling converges past the revisit time. + before its seed returns. The horizon rises to a margin over the + measured gap, never per return: 64 returns of a 200-tick cycle give + 400, where doubling per key gave 2^64. """ - if seed_key not in self._dropped_keys: + touched = self._dropped_keys.pop(seed_key, None) + if touched is None: return - del self._dropped_keys[seed_key] - self._stale_after *= 2 + gap = self._ticks - touched + self._stale_after = max(self._stale_after, REVISIT_MARGIN * gap) def _admit(self) -> _Ensemble | None: """Fresh frame if a slot is free or the oldest frame is abandoned. @@ -311,7 +323,7 @@ def _admit(self) -> _Ensemble | None: # Abandoned: score what it has rather than throw it away. del self._open[key] - self._dropped_keys[key] = True + self._dropped_keys[key] = old.touched if old.n >= _MIN_ENSEMBLE: self._close(key, old) return _Ensemble(self._ensemble, self._max_edges) diff --git a/tests/test_power_doppler.py b/tests/test_power_doppler.py index 6aea19a9..51b51a9e 100644 --- a/tests/test_power_doppler.py +++ b/tests/test_power_doppler.py @@ -214,10 +214,24 @@ def test_regression_slow_cycle_beyond_horizon_still_scores(self): assert pd.stats()["ensembles"] > 0 + def test_regression_many_returns_do_not_compound_horizon(self): + # Falsification (PR #48 review): every returning dropped key doubled + # one shared horizon, so N returns inflated it by 2^N and abandonment + # never fired again. It must track the revisit gap, not the count. + cap, ens, n_seeds = 4, 3, 200 + pd = PowerDoppler(ensemble=ens, max_seeds=cap) + ids = list(range(PATH)) + for _ in range(ens * 6): + for k in range(n_seeds): + _feed(pd, f"s{k}", _static(n=1), ids) + + assert pd.stats()["stale_after"] <= 4 * n_seeds + assert pd.stats()["ensembles"] > 0 + def test_adversarial_abandoned_keys_do_not_grow_horizon(self): # Keys that never return keep the horizon; the dropped-key memory stays bounded. cap = 2 - pd = PowerDoppler(ensemble=N, max_seeds=cap) + pd = PowerDoppler(ensemble=N, max_seeds=cap, max_dropped=cap) horizon = pd.stats()["stale_after"] ids = list(range(PATH)) for k in range(cap * 8): @@ -269,6 +283,8 @@ def test_bad_parameters_rejected(self): PowerDoppler(max_edges=0) with pytest.raises(ValueError): PowerDoppler(max_flow_ids=-1) + with pytest.raises(ValueError): + PowerDoppler(max_dropped=0) class TestDopplerSchedule: