From c56ff767d81f21f6fd2b411db21f6cbef20512f5 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 14:55:45 +0000 Subject: [PATCH 1/2] Fix PR #50 review finding: Doppler drop memory churn The fixed 2^15-key LRU forgot every dropped key once the corpus cycle outgrew it, so revisits were never seen and frames never closed. Remember a crc32-sampled subset instead, halving the sample on overflow: sampled keys are never churned out, memory stays bounded. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Rvj2gdMAa4HA8MGHULfDJE --- CHANGELOG.md | 2 ++ docs/DEEP_DIVE.md | 2 +- src/fuzzer_tool/core/power_doppler.py | 41 +++++++++++++++++++++++---- tests/test_power_doppler.py | 28 ++++++++++++++++++ 4 files changed, 67 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9b4b6719..339490ee 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **Doppler drop memory churned on huge corpora** (`core/power_doppler.py`): the 2¹⁵-key LRU forgot every dropped key before a larger cycle returned, so the horizon never widened. Now a crc32-sampled subset (halved on overflow) is kept, which survives any cycle length within the same bound. + - **Doppler horizon compounded per returning seed** (`core/power_doppler.py`): each return doubled one shared horizon (N returns → 2^N), disabling abandonment; and the dropped-key memory (`max_seeds` keys) forgot keys before large corpora cycled back. Horizon is now `max(horizon, 2 × measured gap)`; memory is 2¹⁵ keys. - **Doppler horizon thrashed on slow corpus cycles** (`core/power_doppler.py`): a cycle longer than the fixed abandon horizon dropped every frame one tick before its seed returned. The horizon now doubles when a dropped seed comes back. diff --git a/docs/DEEP_DIVE.md b/docs/DEEP_DIVE.md index 7847d729..12b5f041 100644 --- a/docs/DEEP_DIVE.md +++ b/docs/DEEP_DIVE.md @@ -85,7 +85,7 @@ For production and sensitive binaries using AFL family fuzzers is the best cours - **AFLGo SHM-tail distance channel**: compiled into EVERY shim-linked target since `__AFL_DISTANCE_MODE` defaulted to 1 (2026-08-24; `-D__AFL_DISTANCE_MODE=0` opts out). Inert unless directed mode uploads a distance table (`DistanceTableShm`/`__AFL_DIST_SHM_ID`): without one, sum/count stay 0 and every reader takes the Python-side path. `build_targets.sh --distance` additionally builds the trace-pc-instrumented `*_dist.so`/`*_dist_asan.so` variants, where the shim accumulates per-block distances in `__sanitizer_cov_trace_pc()` — the PC (relative to the dladdr-derived object base) probes an open-addressing table of `{key, dist}` entries (packed 12-byte layout; the 4-byte header holds the slot *capacity*, a power of two ≥ 2×entries so empty slots exist, and the builder hash-inserts at `key % capacity` with linear probing to mirror the shim's probe — uploaded by the fuzzer at startup via `DistanceTableShm`/`__AFL_DIST_SHM_ID`), accumulating sum/count into the 16-byte SHM **tail** (after the edge table: `u64 dist_sum`, `u64 dist_count`), written at reset, at process exit (subprocess runs never call reset), and per-iteration in in-process modes via `__afl_dist_flush` (direct_lite has no process boundary, so the runner flushes the tail after each `run_one`). Per-execution `avg_distance = sum/count/100` is read straight from the tail and preferred over Python-side computation; blocks without a table entry don't count (AFLGo semantics). The table's PC keys are recovered by scanning text for `call __sanitizer_cov_trace_pc` sites (modern clang emits no `__sancov_pcs` for trace-pc) mapped to valued blocks via the CFGs — `TargetDistance.pc_distance_table()`. The shim's sanitizer-coverage callbacks are hidden-visibility so a libasan LD_PRELOAD cannot interpose over them in PIE builds. `tools/gen_distance_table.py` emits the table as C or text for inspection. Without the table (count==0) everything degrades to the Python-side path. Works in subprocess, direct_lite, and persistent modes (ASAN direct_lite requires libasan preloaded at fuzzer-process start — the `use_direct_lite` gate). The periodic stats line shows live distance when directed mode is active: `dist: avg: min: max:` (or `no-data`). `build_targets.sh --distance` builds both `*_dist.so` (no-ASAN) and `*_dist_asan.so` (ASAN) variants with the cmplog shim linked in, so `--cmplog` keeps them in direct_lite mode. Startup reports `[*] Distance instrumentation: detected` when the target carries the channel (the shim's `__afl_dist_flush` or a defined `__sanitizer_cov_trace_pc`), mirroring the AFL-instrumentation check. With `--elo` in directed mode, `aflgo` joins the Elo-arbitrated seed-strategy pool — a distance-pure arm picking `P(seed) ∝ exp(-2·norm_dist)` (distinct from the generic `weighted` arm, which blends distance with speed/size/entropy). - **AFLGo distance-annealed schedule** (`--schedule go`, requires `--target-functions`): wires the precomputed `avg_distance` (per-seed distance to directed targets) and `_anneal_progress` (exploration/exploitation annealing variable) into `SeedScorer.score()` for mutation budget scaling. During exploration phase (`anneal_progress` ≈ 0): uniform energy. During exploitation phase (`anneal_progress` → 1): `energy *= exp(β · (1 - norm_dist))` where `β = anneal_progress * 5`, capping at 100x. Seeds near the target get exponentially more mutations as the campaign matures. Previously these metrics only influenced seed selection but not mutation intensity. - **Power schedules** (`--schedule base|fast|coe|rare|mopt|lin|quad|go|aflgo|entropic|doppler`): AFL++ power schedules ported to control mutation budget per seed via `SeedScorer`. Each schedule modifies a base score (100) by frequency-based factors. Honggfuzz-style novelty decay, density, fertility, freshness, and entropy factors are applied multiplicatively on top. `entropic` (libFuzzer `-entropic`) scales energy by `1 + log2(1 + rare)`, where `rare` is the larger of `rare_edge_count`/`tc_ref` already collected for RARE/honggfuzz scoring — an approximation of libFuzzer's feature-frequency Shannon entropy using signal the fuzzer already tracks. -- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; when a dropped seed returns (slow, not abandoned) the horizon rises to 2× its measured revisit gap — tracking the slowest real gap, never compounding per return. Dropped keys remembered: 2¹⁵ (LRU, ~5 MiB). Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. +- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; when a dropped seed returns (slow, not abandoned) the horizon rises to 2× its measured revisit gap — tracking the slowest real gap, never compounding per return. Dropped keys remembered: ≤ 2¹⁵ (~5 MiB), a crc32-sampled subset that halves on overflow — never churned out, so revisits are measured at any corpus size. Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. - **Favored set / cull_queue** (`core/schedules.py`, `services/fuzzer.py`): AFL-style `top_rated` minimal-set-cover selection. For each edge, the cheapest seed covering it is selected, then a greedy cover builds the favored set. FAST and COE schedules apply energy bonuses to favored seeds (`_fast_factor`, `_coe_factor`, `coe_skip`). `_cull_queue()` runs periodically during fuzzing and updates `self._favored`; the score call site passes `favored=(seed_key in self._favored)` so the scheduler actually uses it. - **Bayesian seed quality** (`--bayesian`): `BayesianSeedQuality` (`core/seed_quality.py`) maintains a Beta-Bernoulli posterior per seed over `P(outcome = new_coverage)`. Thompson sampling naturally balances explore/exploit without a manual temperature knob — unexplored seeds have high posterior variance and get sampled. The `record_outcome()` feedback loop is now wired in `fuzz_one()` (was previously a dead code path with all posteriors stuck at Beta(1,1)). State is persisted to `seed_quality.json` and restored on resume. diff --git a/src/fuzzer_tool/core/power_doppler.py b/src/fuzzer_tool/core/power_doppler.py index 837a78fa..584df976 100644 --- a/src/fuzzer_tool/core/power_doppler.py +++ b/src/fuzzer_tool/core/power_doppler.py @@ -23,6 +23,7 @@ import functools import math +import zlib from collections.abc import Mapping import numpy as np @@ -45,8 +46,8 @@ STALE_FRAMES = 4 # Horizon headroom over the slowest observed revisit gap. REVISIT_MARGIN = 2 -# Dropped keys remembered (key -> last-touch tick); must outlast a corpus -# cycle's worth of drops. ~150 B each bounds it near 5 MiB. +# Dropped keys remembered (key -> last-touch tick), a hash-sampled subset +# so any corpus cycle is observable. ~150 B each bounds it near 5 MiB. DEFAULT_MAX_DROPPED = 1 << 15 # CFAR false-alarm probability per edge. FALSE_ALARM = 1e-3 @@ -63,6 +64,7 @@ # First column allocation; grows by doubling up to max_edges. INITIAL_COLS = 64 +_CRC_BITS = 32 _MIN_ENSEMBLE = 3 # mean removal + one clutter rank still leaves a dof @@ -215,7 +217,8 @@ class PowerDoppler: max_edges: Edge columns per ensemble; extra edges are dropped. max_scores: Closed-ensemble scores kept (LRU). max_flow_ids: Flow-edge ids kept across all scores (LRU). - max_dropped: Dropped-frame keys remembered (LRU) to measure revisit gaps. + max_dropped: Dropped-frame keys remembered (hash-sampled) to measure + revisit gaps. """ def __init__( @@ -246,7 +249,9 @@ def __init__( self._max_flow_ids = max_flow_ids self._stale_after = max_seeds * ensemble * STALE_FRAMES # Dropped-as-abandoned key -> its frame's last-touch tick. - self._dropped_keys: LRUCache = LRUCache(max_dropped) + self._dropped_keys: dict[str, int] = {} + self._max_dropped = max_dropped + self._sample_bits = 0 # remember keys with crc32 % 2**bits == 0 # Insertion order is recency order; capacity is enforced by hand # (_admit, _trim) so evicted frames can be scored / ids uncounted. self._open: dict[str, _Ensemble] = {} @@ -308,6 +313,32 @@ def _revisit(self, seed_key: str) -> None: gap = self._ticks - touched self._stale_after = max(self._stale_after, REVISIT_MARGIN * gap) + def _sampled(self, key: str) -> bool: + """Key is in the remembered subset: low *sample_bits* of crc32 zero.""" + mask = (1 << self._sample_bits) - 1 + return not zlib.crc32(key.encode()) & mask + + def _remember(self, key: str, touched: int) -> None: + """Record a dropped key, halving the sample when memory is full. + + LRU memory forgot every key once the corpus cycle outgrew it (32k+ + seeds in turn: each key evicted just before it returned). A hash + subset is never churned out, so its keys survive any cycle length:: + + bits 0: all keys bits 1: crc even bits 2: crc % 4 == 0 ... + """ + if not self._sampled(key): + return + + self._dropped_keys[key] = touched + while len(self._dropped_keys) > self._max_dropped and self._sample_bits < _CRC_BITS: + self._sample_bits += 1 + self._dropped_keys = {k: t for k, t in self._dropped_keys.items() if self._sampled(k)} + + # Only crc32 == 0 keys left and still over: keep the bound, drop this one. + if len(self._dropped_keys) > self._max_dropped: + self._dropped_keys.pop(key, None) + def _admit(self) -> _Ensemble | None: """Fresh frame if a slot is free or the oldest frame is abandoned. @@ -323,7 +354,7 @@ def _admit(self) -> _Ensemble | None: # Abandoned: score what it has rather than throw it away. del self._open[key] - self._dropped_keys[key] = old.touched + self._remember(key, old.touched) if old.n >= _MIN_ENSEMBLE: self._close(key, old) return _Ensemble(self._ensemble, self._max_edges) diff --git a/tests/test_power_doppler.py b/tests/test_power_doppler.py index 51b51a9e..5b80ec44 100644 --- a/tests/test_power_doppler.py +++ b/tests/test_power_doppler.py @@ -10,6 +10,7 @@ import shutil import subprocess +import zlib from pathlib import Path import numpy as np @@ -228,6 +229,33 @@ def test_regression_many_returns_do_not_compound_horizon(self): assert pd.stats()["stale_after"] <= 4 * n_seeds assert pd.stats()["ensembles"] > 0 + def test_regression_cycle_beyond_drop_memory_still_scores(self): + # Falsification (PR #50 review): 200 seeds in turn vs 8 remembered + # drops. LRU memory forgot every key just before it returned. + cap, ens, n_seeds = 4, 3, 200 + pd = PowerDoppler(ensemble=ens, max_seeds=cap, max_dropped=8) + ids = list(range(PATH)) + for _ in range(ens * 6): + for k in range(n_seeds): + _feed(pd, f"s{k}", _static(n=1), ids) + + assert pd.stats()["ensembles"] > 0 + assert pd.stats()["stale_after"] <= 4 * n_seeds + assert len(pd._dropped_keys) <= 8 + + def test_adversarial_drop_memory_halving_keeps_hash_subset(self): + # Overflow halves the sample: memory stays bounded and holds exactly + # the keys whose crc32 low bits are zero, so they cannot be churned out. + cap = 2 + pd = PowerDoppler(ensemble=N, max_dropped=cap) + for k in range(100): + pd._remember(f"k{k}", k) + + mask = (1 << pd._sample_bits) - 1 + assert len(pd._dropped_keys) <= cap + assert pd._sample_bits > 0 + assert all(not zlib.crc32(k.encode()) & mask for k in pd._dropped_keys) + def test_adversarial_abandoned_keys_do_not_grow_horizon(self): # Keys that never return keep the horizon; the dropped-key memory stays bounded. cap = 2 From d7f1efd71bc180b8cddd0be25201ac08fffebc40 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 1 Oct 2026 15:00:29 +0000 Subject: [PATCH 2/2] Doppler drop memory: bottom-k crc32 reservoir A halving crc threshold subset emptied out when every key had an odd crc, so no revisit widened the horizon again. Keep the k smallest crc32s instead: a fixed, never-empty subset of any cycle. Use the repo's crc32_ieee; test with a cycle of masked-out keys. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01Rvj2gdMAa4HA8MGHULfDJE --- CHANGELOG.md | 2 +- docs/DEEP_DIVE.md | 2 +- src/fuzzer_tool/core/power_doppler.py | 64 ++++++++++++++++----------- tests/test_power_doppler.py | 45 ++++++++++++++----- 4 files changed, 76 insertions(+), 37 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 339490ee..8b9803c7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,7 +9,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed -- **Doppler drop memory churned on huge corpora** (`core/power_doppler.py`): the 2¹⁵-key LRU forgot every dropped key before a larger cycle returned, so the horizon never widened. Now a crc32-sampled subset (halved on overflow) is kept, which survives any cycle length within the same bound. +- **Doppler drop memory churned on huge corpora** (`core/power_doppler.py`): the 2¹⁵-key LRU forgot every dropped key before a larger cycle returned, so the horizon never widened. Now the bottom-k keys by `crc32_ieee` are kept: a fixed, never-empty subset of any cycle within the same bound (a halving crc threshold could empty out on all-odd crcs). - **Doppler horizon compounded per returning seed** (`core/power_doppler.py`): each return doubled one shared horizon (N returns → 2^N), disabling abandonment; and the dropped-key memory (`max_seeds` keys) forgot keys before large corpora cycled back. Horizon is now `max(horizon, 2 × measured gap)`; memory is 2¹⁵ keys. diff --git a/docs/DEEP_DIVE.md b/docs/DEEP_DIVE.md index 12b5f041..1429c1c9 100644 --- a/docs/DEEP_DIVE.md +++ b/docs/DEEP_DIVE.md @@ -85,7 +85,7 @@ For production and sensitive binaries using AFL family fuzzers is the best cours - **AFLGo SHM-tail distance channel**: compiled into EVERY shim-linked target since `__AFL_DISTANCE_MODE` defaulted to 1 (2026-08-24; `-D__AFL_DISTANCE_MODE=0` opts out). Inert unless directed mode uploads a distance table (`DistanceTableShm`/`__AFL_DIST_SHM_ID`): without one, sum/count stay 0 and every reader takes the Python-side path. `build_targets.sh --distance` additionally builds the trace-pc-instrumented `*_dist.so`/`*_dist_asan.so` variants, where the shim accumulates per-block distances in `__sanitizer_cov_trace_pc()` — the PC (relative to the dladdr-derived object base) probes an open-addressing table of `{key, dist}` entries (packed 12-byte layout; the 4-byte header holds the slot *capacity*, a power of two ≥ 2×entries so empty slots exist, and the builder hash-inserts at `key % capacity` with linear probing to mirror the shim's probe — uploaded by the fuzzer at startup via `DistanceTableShm`/`__AFL_DIST_SHM_ID`), accumulating sum/count into the 16-byte SHM **tail** (after the edge table: `u64 dist_sum`, `u64 dist_count`), written at reset, at process exit (subprocess runs never call reset), and per-iteration in in-process modes via `__afl_dist_flush` (direct_lite has no process boundary, so the runner flushes the tail after each `run_one`). Per-execution `avg_distance = sum/count/100` is read straight from the tail and preferred over Python-side computation; blocks without a table entry don't count (AFLGo semantics). The table's PC keys are recovered by scanning text for `call __sanitizer_cov_trace_pc` sites (modern clang emits no `__sancov_pcs` for trace-pc) mapped to valued blocks via the CFGs — `TargetDistance.pc_distance_table()`. The shim's sanitizer-coverage callbacks are hidden-visibility so a libasan LD_PRELOAD cannot interpose over them in PIE builds. `tools/gen_distance_table.py` emits the table as C or text for inspection. Without the table (count==0) everything degrades to the Python-side path. Works in subprocess, direct_lite, and persistent modes (ASAN direct_lite requires libasan preloaded at fuzzer-process start — the `use_direct_lite` gate). The periodic stats line shows live distance when directed mode is active: `dist: avg: min: max:` (or `no-data`). `build_targets.sh --distance` builds both `*_dist.so` (no-ASAN) and `*_dist_asan.so` (ASAN) variants with the cmplog shim linked in, so `--cmplog` keeps them in direct_lite mode. Startup reports `[*] Distance instrumentation: detected` when the target carries the channel (the shim's `__afl_dist_flush` or a defined `__sanitizer_cov_trace_pc`), mirroring the AFL-instrumentation check. With `--elo` in directed mode, `aflgo` joins the Elo-arbitrated seed-strategy pool — a distance-pure arm picking `P(seed) ∝ exp(-2·norm_dist)` (distinct from the generic `weighted` arm, which blends distance with speed/size/entropy). - **AFLGo distance-annealed schedule** (`--schedule go`, requires `--target-functions`): wires the precomputed `avg_distance` (per-seed distance to directed targets) and `_anneal_progress` (exploration/exploitation annealing variable) into `SeedScorer.score()` for mutation budget scaling. During exploration phase (`anneal_progress` ≈ 0): uniform energy. During exploitation phase (`anneal_progress` → 1): `energy *= exp(β · (1 - norm_dist))` where `β = anneal_progress * 5`, capping at 100x. Seeds near the target get exponentially more mutations as the campaign matures. Previously these metrics only influenced seed selection but not mutation intensity. - **Power schedules** (`--schedule base|fast|coe|rare|mopt|lin|quad|go|aflgo|entropic|doppler`): AFL++ power schedules ported to control mutation budget per seed via `SeedScorer`. Each schedule modifies a base score (100) by frequency-based factors. Honggfuzz-style novelty decay, density, fertility, freshness, and entropy factors are applied multiplicatively on top. `entropic` (libFuzzer `-entropic`) scales energy by `1 + log2(1 + rare)`, where `rare` is the larger of `rare_edge_count`/`tc_ref` already collected for RARE/honggfuzz scoring — an approximation of libFuzzer's feature-frequency Shannon entropy using signal the fuzzer already tracks. -- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; when a dropped seed returns (slow, not abandoned) the horizon rises to 2× its measured revisit gap — tracking the slowest real gap, never compounding per return. Dropped keys remembered: ≤ 2¹⁵ (~5 MiB), a crc32-sampled subset that halves on overflow — never churned out, so revisits are measured at any corpus size. Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. +- **Power Doppler schedule** (`--schedule doppler`, `core/power_doppler.py`): ultrasound power Doppler on coverage. Slow time = successive mutants of one seed (32 per ensemble); pixel = edge; sample = `log2(1 + hits)` (raw SHM counts, not buckets). Wall filter: mean removal (static path), then SVD components whose participation ratio spans ≥ half the seed's edges (and ≥ 4) are dropped as clutter — an early reject moving the whole path at once ("flash"). CFAR: residual power per edge vs `σ² · χ²_{dof}(1 − 10⁻³)`, `σ²` = median residual variance (floor 10⁻³). Seed power = summed flow power / dof; energy = `log1p(p)/log1p(max p)` ∈ [0, 1], scaled to `[1, max_mult]` like `katz`. Unscored or static seeds stay 1×. Gram eigendecomposition (n×n) replaces the full SVD (~6× faster). Bounded: 64 open ensembles × 2048 edges (float32, ≤16 MiB); when full, new seeds wait instead of evicting partial frames (LRU eviction never closed a frame once >64 seeds were picked in turn), and a frame untouched for 4 frames' worth of samples is scored early (≥ 3 samples) and freed; when a dropped seed returns (slow, not abandoned) the horizon rises to 2× its measured revisit gap — tracking the slowest real gap, never compounding per return. Dropped keys remembered: ≤ 2¹⁵ (~5 MiB), the bottom-k by `crc32_ieee` — a fixed, never-empty subset of any cycle, never churned out, so revisits are measured at any corpus size. Scores: 4096 (LRU), flow-edge ids in int64 arrays capped at 2²⁰ total (8 MiB). Multi-target frames are keyed per target (edge ids are per-target). SHM coverage only — without it the schedule falls back to `base` with a warning; cost ~100 µs/exec at 2k live edges (dict→array conversion dominates), zero when off. Unmeasured — see `docs/TODO.md`. - **Favored set / cull_queue** (`core/schedules.py`, `services/fuzzer.py`): AFL-style `top_rated` minimal-set-cover selection. For each edge, the cheapest seed covering it is selected, then a greedy cover builds the favored set. FAST and COE schedules apply energy bonuses to favored seeds (`_fast_factor`, `_coe_factor`, `coe_skip`). `_cull_queue()` runs periodically during fuzzing and updates `self._favored`; the score call site passes `favored=(seed_key in self._favored)` so the scheduler actually uses it. - **Bayesian seed quality** (`--bayesian`): `BayesianSeedQuality` (`core/seed_quality.py`) maintains a Beta-Bernoulli posterior per seed over `P(outcome = new_coverage)`. Thompson sampling naturally balances explore/exploit without a manual temperature knob — unexplored seeds have high posterior variance and get sampled. The `record_outcome()` feedback loop is now wired in `fuzz_one()` (was previously a dead code path with all posteriors stuck at Beta(1,1)). State is persisted to `seed_quality.json` and restored on resume. diff --git a/src/fuzzer_tool/core/power_doppler.py b/src/fuzzer_tool/core/power_doppler.py index 584df976..8fc70614 100644 --- a/src/fuzzer_tool/core/power_doppler.py +++ b/src/fuzzer_tool/core/power_doppler.py @@ -22,13 +22,14 @@ from __future__ import annotations import functools +import heapq import math -import zlib from collections.abc import Mapping import numpy as np from fuzzer_tool.core.chi_squared import chi_squared_critical_value +from fuzzer_tool.core.crc32 import crc32_ieee from fuzzer_tool.core.lru import LRUCache # Mutants per ensemble (pulses per Doppler frame). @@ -46,8 +47,8 @@ STALE_FRAMES = 4 # Horizon headroom over the slowest observed revisit gap. REVISIT_MARGIN = 2 -# Dropped keys remembered (key -> last-touch tick), a hash-sampled subset -# so any corpus cycle is observable. ~150 B each bounds it near 5 MiB. +# Dropped keys remembered (key -> last-touch tick): the bottom-k by crc32, +# so any corpus cycle keeps witnesses. ~150 B each bounds it near 5 MiB. DEFAULT_MAX_DROPPED = 1 << 15 # CFAR false-alarm probability per edge. FALSE_ALARM = 1e-3 @@ -64,7 +65,6 @@ # First column allocation; grows by doubling up to max_edges. INITIAL_COLS = 64 -_CRC_BITS = 32 _MIN_ENSEMBLE = 3 # mean removal + one clutter rank still leaves a dof @@ -217,8 +217,8 @@ class PowerDoppler: max_edges: Edge columns per ensemble; extra edges are dropped. max_scores: Closed-ensemble scores kept (LRU). max_flow_ids: Flow-edge ids kept across all scores (LRU). - max_dropped: Dropped-frame keys remembered (hash-sampled) to measure - revisit gaps. + max_dropped: Dropped-frame keys remembered (bottom-k crc32) to + measure revisit gaps. """ def __init__( @@ -251,7 +251,8 @@ def __init__( # Dropped-as-abandoned key -> its frame's last-touch tick. self._dropped_keys: dict[str, int] = {} self._max_dropped = max_dropped - self._sample_bits = 0 # remember keys with crc32 % 2**bits == 0 + # Max-heap (-crc, key) over remembered keys; stale entries lazily skipped. + self._crc_heap: list[tuple[int, str]] = [] # Insertion order is recency order; capacity is enforced by hand # (_admit, _trim) so evicted frames can be scored / ids uncounted. self._open: dict[str, _Ensemble] = {} @@ -313,31 +314,44 @@ def _revisit(self, seed_key: str) -> None: gap = self._ticks - touched self._stale_after = max(self._stale_after, REVISIT_MARGIN * gap) - def _sampled(self, key: str) -> bool: - """Key is in the remembered subset: low *sample_bits* of crc32 zero.""" - mask = (1 << self._sample_bits) - 1 - return not zlib.crc32(key.encode()) & mask - def _remember(self, key: str, touched: int) -> None: - """Record a dropped key, halving the sample when memory is full. + """Record a dropped key; keep the *max_dropped* smallest crc32s. - LRU memory forgot every key once the corpus cycle outgrew it (32k+ - seeds in turn: each key evicted just before it returned). A hash - subset is never churned out, so its keys survive any cycle length:: + LRU memory forgot every key once the corpus cycle outgrew it, and a + crc threshold subset could empty out (all-odd crcs after one halving). + Bottom-k is a fixed subset of any cycle, never churned, never empty:: - bits 0: all keys bits 1: crc even bits 2: crc % 4 == 0 ... + drops (crc): a(9) b(2) c(7) d(1), k=2 -> remember {d, b} """ - if not self._sampled(key): + if key in self._dropped_keys: + self._dropped_keys[key] = touched return - self._dropped_keys[key] = touched - while len(self._dropped_keys) > self._max_dropped and self._sample_bits < _CRC_BITS: - self._sample_bits += 1 - self._dropped_keys = {k: t for k, t in self._dropped_keys.items() if self._sampled(k)} + crc = crc32_ieee(key.encode()) + heap = self._crc_heap + if len(self._dropped_keys) >= self._max_dropped: + top = self._heap_top() + if crc >= -top[0]: + return - # Only crc32 == 0 keys left and still over: keep the bound, drop this one. - if len(self._dropped_keys) > self._max_dropped: - self._dropped_keys.pop(key, None) + heapq.heappop(heap) + del self._dropped_keys[top[1]] + + self._dropped_keys[key] = touched + heapq.heappush(heap, (-crc, key)) + # Revisits leave stale (and, on re-drop, duplicate) entries: rebuild + # one live entry per key once the heap doubles past the bound. + if len(heap) > 2 * self._max_dropped: + live = {e[1]: e for e in heap if e[1] in self._dropped_keys} + self._crc_heap = list(live.values()) + heapq.heapify(self._crc_heap) + + def _heap_top(self) -> tuple[int, str]: + """Largest-crc live entry; pops entries of keys already revisited.""" + heap = self._crc_heap + while heap[0][1] not in self._dropped_keys: + heapq.heappop(heap) + return heap[0] def _admit(self) -> _Ensemble | None: """Fresh frame if a slot is free or the oldest frame is abandoned. diff --git a/tests/test_power_doppler.py b/tests/test_power_doppler.py index 5b80ec44..1d245969 100644 --- a/tests/test_power_doppler.py +++ b/tests/test_power_doppler.py @@ -10,12 +10,12 @@ import shutil import subprocess -import zlib from pathlib import Path import numpy as np import pytest +from fuzzer_tool.core.crc32 import crc32_ieee from fuzzer_tool.core.power_doppler import ( STALE_FRAMES, PowerDoppler, @@ -243,18 +243,43 @@ def test_regression_cycle_beyond_drop_memory_still_scores(self): assert pd.stats()["stale_after"] <= 4 * n_seeds assert len(pd._dropped_keys) <= 8 - def test_adversarial_drop_memory_halving_keeps_hash_subset(self): - # Overflow halves the sample: memory stays bounded and holds exactly - # the keys whose crc32 low bits are zero, so they cannot be churned out. + def test_regression_masked_out_cycle_keeps_a_witness(self): + # Falsification (PR #51 review): every key has an odd crc32, so the + # first halving emptied the threshold subset and no revisit was + # ever seen again. Bottom-k always keeps the k smallest crcs. + cap, ens, n_seeds = 4, 3, 200 + keys = [k for k in (f"s{i}" for i in range(4 * n_seeds)) if crc32_ieee(k.encode()) & 1] + keys = keys[:n_seeds] + pd = PowerDoppler(ensemble=ens, max_seeds=cap, max_dropped=2) + ids = list(range(PATH)) + for _ in range(ens * 6): + for k in keys: + _feed(pd, k, _static(n=1), ids) + + assert pd.stats()["ensembles"] > 0 + + def test_adversarial_drop_memory_is_bottom_k_crc(self): + # Overflow keeps exactly the k smallest crcs: bounded, never empty. + cap = 3 + pd = PowerDoppler(ensemble=N, max_dropped=cap) + keys = [f"k{i}" for i in range(100)] + for t, k in enumerate(keys): + pd._remember(k, t) + + expected = sorted(keys, key=lambda k: crc32_ieee(k.encode()))[:cap] + assert set(pd._dropped_keys) == set(expected) + + def test_adversarial_redropped_key_keeps_heap_bounded(self): + # Drop -> revisit -> drop of one key, many times: heap stays O(k). cap = 2 pd = PowerDoppler(ensemble=N, max_dropped=cap) - for k in range(100): - pd._remember(f"k{k}", k) + for t in range(100): + pd._remember("a", t) + pd._remember("b", t) + pd._revisit("a") - mask = (1 << pd._sample_bits) - 1 - assert len(pd._dropped_keys) <= cap - assert pd._sample_bits > 0 - assert all(not zlib.crc32(k.encode()) & mask for k in pd._dropped_keys) + assert len(pd._crc_heap) <= 2 * cap + assert set(pd._dropped_keys) == {"b"} def test_adversarial_abandoned_keys_do_not_grow_horizon(self): # Keys that never return keep the horizon; the dropped-key memory stays bounded.