diff --git a/src/fuzzer_tool/cli/commands.py b/src/fuzzer_tool/cli/commands.py index f7527205..55092cbf 100644 --- a/src/fuzzer_tool/cli/commands.py +++ b/src/fuzzer_tool/cli/commands.py @@ -830,6 +830,7 @@ def cmd_fuzz(args): dedup_execs=not getattr(args, "no_dedup_execs", False), seed_calibration=not getattr(args, "no_calibration", False), exec_dedup_backend=getattr(args, "exec_dedup_backend", "bloom"), + cuckoo_seed_filter=getattr(args, "cuckoo_seed_filter", False), perf_novelty=not getattr(args, "no_perf_novelty", False), reject_code=getattr(args, "reject_code", None), op_span_reverse=getattr(args, "op_span_reverse", False), @@ -4041,6 +4042,19 @@ def main() -> int: "contract, so the choice is opt-in and the default is unchanged." ), ) + fuzz_parser.add_argument( + "--cuckoo-seed-filter", + action="store_true", + default=False, + help=( + "Track pruned seeds in a CuckooFilter so their mutations are " + "skipped at dedup time. At startup, all seeds under corpus/seeds/pruned/ " + "are added to the filter. When a seed is pruned during minimization, " + "its hash is added. In _dedup_mutate(), if a mutation's hash matches " + "a pruned seed, the mutation is skipped (original data returned). " + "Disabled by default." + ), + ) fuzz_parser.add_argument( "--reject-code", type=int, diff --git a/src/fuzzer_tool/services/corpus_manager.py b/src/fuzzer_tool/services/corpus_manager.py index 9aae5d4b..a8e49551 100644 --- a/src/fuzzer_tool/services/corpus_manager.py +++ b/src/fuzzer_tool/services/corpus_manager.py @@ -1755,6 +1755,10 @@ def _recover_uncovered(self, unique: list[bytes], mandatory: set[int]) -> None: mandatory.add(id(seed)) if f.corpus_dir: self._promote_seed(seed) + # Remove recovered seed from cuckoo filter if present + if f.cuckoo_seed_filter is not None: + h = self.seed_key(seed) + f.cuckoo_seed_filter.remove(h) recovered_count += 1 if recovered_count: log.warning( @@ -1772,6 +1776,13 @@ def _commit_minimize(self, unique: list[bytes], removed: int, stale_ratio: float self._prune_files(kept_set) del kept_set # free kept hashes after file pruning + # Add pruned seeds to the cuckoo seed filter if enabled + if f.cuckoo_seed_filter is not None: + for seed in f.corpus: + if seed not in unique: + h = _hash(seed) + f.cuckoo_seed_filter.add(h) + f.corpus = unique self.rebuild_entropy() new_meta = {} diff --git a/src/fuzzer_tool/services/fuzzer.py b/src/fuzzer_tool/services/fuzzer.py index 03c016ef..5e3169ce 100644 --- a/src/fuzzer_tool/services/fuzzer.py +++ b/src/fuzzer_tool/services/fuzzer.py @@ -1352,6 +1352,12 @@ def __init__( # reset_on_full) contract that _dedup_mutate drives, so the # branch lives in one place: here, at construction. exec_dedup_backend="bloom", + # Cuckoo filter for pruned seed dedup. When enabled, seeds under + # corpus/seeds/pruned/ are loaded at startup and their hashes added + # to the filter. When a seed is pruned during minimization, its hash + # is added. In _dedup_mutate(), if a mutation's hash is in the + # filter, the mutation is skipped (original data returned). + cuckoo_seed_filter=False, fluctuation=False, fluctuation_beta=1.0, fluctuation_window=1000, @@ -2024,6 +2030,16 @@ def __init__( raise ValueError( f"unknown exec_dedup_backend {exec_dedup_backend!r}; expected 'bloom' or 'cuckoo'" ) + # Pruned seed filter: when --cuckoo-seed-filter is set, a + # CuckooFilter tracks all pruned seeds so their mutations + # are skipped in _dedup_mutate(). None when disabled. + self.cuckoo_seed_filter: CuckooFilter | None = None + if cuckoo_seed_filter: + from fuzzer_tool.core.cuckoo import CuckooFilter + + # Sized at 10x the corpus; minimum 100_000 to match + # the exec bloom default. + self.cuckoo_seed_filter = CuckooFilter(capacity=max(10 * len(self.corpus), 100_000)) self._dedup_hits = 0 self._dedup_gaveup = 0 # Performance novelty (per-edge max hit count). Separate from the @@ -2338,6 +2354,11 @@ def __init__( self._load_corpus() loaded = self.corpus + + # Load pruned seeds into cuckoo filter if enabled + if self.cuckoo_seed_filter is not None and self.corpus_dir is not None: + self._load_pruned_seeds_into_cuckoo() + self._apply_seed_transforms() if self._corpus_boost > 0 and self.corpus: self._boost_corpus_sizes() @@ -2651,7 +2672,9 @@ def __init__( ) log.info("Position cmplog scheduling enabled") if pos_cmplog and not cmplog: - log.warning("--pos-cmplog needs cmplog, which is off: the arm stays out of the pool") + log.warning( + "--pos-cmplog needs cmplog, which is off: the arm stays out of the pool" + ) # Position-arena lineage: the mutation sites that produced a seed, as # its own landing prior (see core/schedulers/pos_lineage.py). # Tracker-style: the arena gates it on --lineage, the only mode that @@ -4121,6 +4144,25 @@ def _setup_ptrace(self, target, deep_coverage, max_bps, fallback_hint=False): def _load_corpus(self): return self._corpus_manager.load_corpus() + def _load_pruned_seeds_into_cuckoo(self): + """Add all pruned seeds to the cuckoo filter at startup.""" + pruned_dir = self.corpus_dir / "seeds" / "pruned" + if not pruned_dir.exists(): + return + count = 0 + for fh in pruned_dir.rglob("id_*"): + if not fh.is_file(): + continue + try: + data = fh.read_bytes() + h = self._seed_key(data) + self.cuckoo_seed_filter.add(h) + count += 1 + except OSError: + continue + if count: + log.info("Loaded %d pruned seeds into cuckoo seed filter", count) + def _init_seed_metadata(self): return self._corpus_manager.init_seed_metadata() @@ -4495,7 +4537,20 @@ def _dedup_mutate(self, data: bytes) -> bytes: That is harmless — it stays reachable on later iterations — and the filter never yields a false negative, so nothing already executed slips through as new. + + When --cuckoo-seed-filter is enabled, the pruned-seed cuckoo filter + is checked before the exec bloom. If the parent seed's hash matches + a previously-pruned seed, the mutation is skipped (original data + returned) — the seed was already deemed not valuable, so + re-discovering it is pure waste. """ + # Pruned-seed filter check: if the parent seed was previously + # pruned, skip the mutation entirely and return the original data. + if self.cuckoo_seed_filter is not None: + h = self._seed_key(data) + if self.cuckoo_seed_filter.contains(h): + self._dedup_hits += 1 + return data mutated = self.mutate(data) if not self._dedup_execs: return mutated diff --git a/tests/test_bloom_exec_dedup.py b/tests/test_bloom_exec_dedup.py index 4bff18d6..c4b0425d 100644 --- a/tests/test_bloom_exec_dedup.py +++ b/tests/test_bloom_exec_dedup.py @@ -149,6 +149,8 @@ def __init__(self, mutants, dedup_execs=True, capacity=1000, backend="bloom"): self._dedup_hits = 0 self._dedup_gaveup = 0 self.mutate_calls = 0 + # Cuckoo seed filter: None when feature is disabled + self.cuckoo_seed_filter = None def mutate(self, data): self.mutate_calls += 1 diff --git a/tests/test_regression_cuckoo_seed_filter.py b/tests/test_regression_cuckoo_seed_filter.py new file mode 100644 index 00000000..4b85abd1 --- /dev/null +++ b/tests/test_regression_cuckoo_seed_filter.py @@ -0,0 +1,191 @@ +"""Tests for --cuckoo-seed-filter: pruned seeds tracked in a CuckooFilter. + +At startup, all seeds under corpus/seeds/pruned/ are added to the filter. +When a seed is pruned during minimization, its hash is added. +In _dedup_mutate(), if a mutation's hash matches a pruned seed, the +mutation is skipped (original data returned). +""" + +from pathlib import Path + +import pytest + +from fuzzer_tool.services.fuzzer import Fuzzer + + +def _write_seed(corpus_dir: Path, seed: bytes) -> str: + """Write a seed to corpus/seeds/ and return its 16-char hash.""" + from fuzzer_tool.adapters.filesystem import hash_data + + h = hash_data(seed) + sub = corpus_dir / "seeds" / h[:2] + sub.mkdir(parents=True, exist_ok=True) + (sub / f"id_{h}").write_bytes(seed) + return h + + +def _make_corpus_dir(tmp_path: Path) -> Path: + corpus_dir = tmp_path / "corpus" + corpus_dir.mkdir(parents=True) + (corpus_dir / "seeds").mkdir() + return corpus_dir + + +@pytest.fixture +def corpus_dir(tmp_path: Path) -> Path: + return _make_corpus_dir(tmp_path) + + +def test_pruned_seed_added_to_cuckoo_at_startup(corpus_dir: Path) -> None: + """Pruned seeds under corpus/seeds/pruned/ are loaded into the cuckoo filter at startup.""" + seed = b"pruned_seed_data" + pruned_dir = corpus_dir / "seeds" / "pruned" + pruned_dir.mkdir(parents=True) + _write_seed(corpus_dir, seed) + # Move seed to pruned + from fuzzer_tool.adapters.filesystem import hash_data + + h = hash_data(seed) + (corpus_dir / "seeds" / h[:2] / f"id_{h}").unlink() + pruned_sub = pruned_dir / h[:2] + pruned_sub.mkdir(parents=True) + (pruned_sub / f"id_{h}").write_bytes(seed) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + assert f.cuckoo_seed_filter is not None + assert f.cuckoo_seed_filter.contains(h) + + +def test_non_pruned_seed_not_in_cuckoo_filter(corpus_dir: Path) -> None: + """Seeds not pruned should not be in the cuckoo filter.""" + seed = b"active_seed_data" + _write_seed(corpus_dir, seed) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + h = f._seed_key(seed) + # Active seed should NOT be in the cuckoo filter + assert not f.cuckoo_seed_filter.contains(h) + + +def test_pruned_seed_mutation_is_skipped(corpus_dir: Path) -> None: + """When a seed is pruned, its mutation is skipped in _dedup_mutate.""" + seed = b"seed_to_prune" + _write_seed(corpus_dir, seed) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + # Add the seed's hash to the cuckoo filter (simulating pruning) + h = f._seed_key(seed) + f.cuckoo_seed_filter.add(h) + + # Mutate the seed - the mutation should be skipped + result = f._dedup_mutate(seed) + # Since the seed's hash is in the cuckoo filter, the mutation + # should return the original data (pruned seed skipped) + assert result == seed + + +def test_non_pruned_seed_mutates_normally(corpus_dir: Path) -> None: + """Seeds not in the cuckoo filter should mutate normally.""" + seed = b"active_seed" + _write_seed(corpus_dir, seed) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + h = f._seed_key(seed) + # Ensure seed is NOT in the filter + assert not f.cuckoo_seed_filter.contains(h) + + # Mutation should proceed normally (return mutated data) + result = f._dedup_mutate(seed) + # Result should be different from original (mutation happened) + assert result != seed + + +def test_cuckoo_filter_disabled_by_default(corpus_dir: Path) -> None: + """When --cuckoo-seed-filter is not set, the filter is None.""" + seed = b"test_seed" + _write_seed(corpus_dir, seed) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + ) + assert f.cuckoo_seed_filter is None + + +def test_filter_capacity_scaled_to_corpus_size(corpus_dir: Path) -> None: + """Filter capacity is max(10 * len(corpus), 100_000).""" + # Add 5 seeds + for i in range(5): + _write_seed(corpus_dir, f"seed_{i}".encode()) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + assert f.cuckoo_seed_filter is not None + # Capacity should be at least 100_000 (min) since 5 * 10 = 50 < 100_000 + assert f.cuckoo_seed_filter.capacity == 100_000 + + +def test_empty_corpus_filter_has_minimum_capacity(corpus_dir: Path) -> None: + """Filter capacity is at least 100_000 even with empty corpus.""" + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + assert f.cuckoo_seed_filter is not None + assert f.cuckoo_seed_filter.capacity == 100_000 + + +def test_pruned_seeds_added_to_filter_at_startup(corpus_dir: Path) -> None: + """All seeds under corpus/seeds/pruned/ are added to the filter at startup.""" + pruned_dir = corpus_dir / "seeds" / "pruned" + pruned_dir.mkdir(parents=True) + + # Create multiple pruned seeds + seeds = [b"pruned_1", b"pruned_2", b"pruned_3"] + hashes = [] + for seed in seeds: + h = _write_seed(corpus_dir, seed) + # Move to pruned + from fuzzer_tool.adapters.filesystem import hash_data + + hash_data(seed) + sub = pruned_dir / h[:2] + sub.mkdir(parents=True, exist_ok=True) + (sub / f"id_{h}").write_bytes(seed) + hashes.append(h) + + f = Fuzzer( + target="nonexistent", + corpus_dir=str(corpus_dir), + crashes_dir=str(corpus_dir / "crashes"), + cuckoo_seed_filter=True, + ) + for h in hashes: + assert f.cuckoo_seed_filter.contains(h), f"Pruned seed {h} should be in cuckoo filter"