Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions src/fuzzer_tool/cli/commands.py
Original file line number Diff line number Diff line change
Expand Up @@ -830,6 +830,7 @@ def cmd_fuzz(args):
dedup_execs=not getattr(args, "no_dedup_execs", False),
seed_calibration=not getattr(args, "no_calibration", False),
exec_dedup_backend=getattr(args, "exec_dedup_backend", "bloom"),
cuckoo_seed_filter=getattr(args, "cuckoo_seed_filter", False),
perf_novelty=not getattr(args, "no_perf_novelty", False),
reject_code=getattr(args, "reject_code", None),
op_span_reverse=getattr(args, "op_span_reverse", False),
Expand Down Expand Up @@ -4041,6 +4042,19 @@ def main() -> int:
"contract, so the choice is opt-in and the default is unchanged."
),
)
fuzz_parser.add_argument(
"--cuckoo-seed-filter",
action="store_true",
default=False,
help=(
"Track pruned seeds in a CuckooFilter so their mutations are "
Comment on lines +4045 to +4050
"skipped at dedup time. At startup, all seeds under corpus/seeds/pruned/ "
"are added to the filter. When a seed is pruned during minimization, "
"its hash is added. In _dedup_mutate(), if a mutation's hash matches "
"a pruned seed, the mutation is skipped (original data returned). "
"Disabled by default."
),
)
fuzz_parser.add_argument(
"--reject-code",
type=int,
Expand Down
11 changes: 11 additions & 0 deletions src/fuzzer_tool/services/corpus_manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -1755,6 +1755,10 @@ def _recover_uncovered(self, unique: list[bytes], mandatory: set[int]) -> None:
mandatory.add(id(seed))
if f.corpus_dir:
self._promote_seed(seed)
# Remove recovered seed from cuckoo filter if present
if f.cuckoo_seed_filter is not None:
h = self.seed_key(seed)
f.cuckoo_seed_filter.remove(h)
Comment on lines +1759 to +1761
recovered_count += 1
if recovered_count:
log.warning(
Expand All @@ -1772,6 +1776,13 @@ def _commit_minimize(self, unique: list[bytes], removed: int, stale_ratio: float
self._prune_files(kept_set)
del kept_set # free kept hashes after file pruning

# Add pruned seeds to the cuckoo seed filter if enabled
if f.cuckoo_seed_filter is not None:
for seed in f.corpus:
if seed not in unique:
h = _hash(seed)
f.cuckoo_seed_filter.add(h)

f.corpus = unique
self.rebuild_entropy()
new_meta = {}
Expand Down
57 changes: 56 additions & 1 deletion src/fuzzer_tool/services/fuzzer.py
Original file line number Diff line number Diff line change
Expand Up @@ -1352,6 +1352,12 @@ def __init__(
# reset_on_full) contract that _dedup_mutate drives, so the
# branch lives in one place: here, at construction.
exec_dedup_backend="bloom",
# Cuckoo filter for pruned seed dedup. When enabled, seeds under
# corpus/seeds/pruned/ are loaded at startup and their hashes added
# to the filter. When a seed is pruned during minimization, its hash
# is added. In _dedup_mutate(), if a mutation's hash is in the
# filter, the mutation is skipped (original data returned).
cuckoo_seed_filter=False,
fluctuation=False,
fluctuation_beta=1.0,
fluctuation_window=1000,
Expand Down Expand Up @@ -2024,6 +2030,16 @@ def __init__(
raise ValueError(
f"unknown exec_dedup_backend {exec_dedup_backend!r}; expected 'bloom' or 'cuckoo'"
)
# Pruned seed filter: when --cuckoo-seed-filter is set, a
# CuckooFilter tracks all pruned seeds so their mutations
# are skipped in _dedup_mutate(). None when disabled.
self.cuckoo_seed_filter: CuckooFilter | None = None
if cuckoo_seed_filter:
from fuzzer_tool.core.cuckoo import CuckooFilter

# Sized at 10x the corpus; minimum 100_000 to match
# the exec bloom default.
self.cuckoo_seed_filter = CuckooFilter(capacity=max(10 * len(self.corpus), 100_000))
Comment on lines +2040 to +2042
self._dedup_hits = 0
self._dedup_gaveup = 0
# Performance novelty (per-edge max hit count). Separate from the
Expand Down Expand Up @@ -2338,6 +2354,11 @@ def __init__(

self._load_corpus()
loaded = self.corpus

# Load pruned seeds into cuckoo filter if enabled
if self.cuckoo_seed_filter is not None and self.corpus_dir is not None:
self._load_pruned_seeds_into_cuckoo()

self._apply_seed_transforms()
if self._corpus_boost > 0 and self.corpus:
self._boost_corpus_sizes()
Expand Down Expand Up @@ -2651,7 +2672,9 @@ def __init__(
)
log.info("Position cmplog scheduling enabled")
if pos_cmplog and not cmplog:
log.warning("--pos-cmplog needs cmplog, which is off: the arm stays out of the pool")
log.warning(
"--pos-cmplog needs cmplog, which is off: the arm stays out of the pool"
)
# Position-arena lineage: the mutation sites that produced a seed, as
# its own landing prior (see core/schedulers/pos_lineage.py).
# Tracker-style: the arena gates it on --lineage, the only mode that
Expand Down Expand Up @@ -4121,6 +4144,25 @@ def _setup_ptrace(self, target, deep_coverage, max_bps, fallback_hint=False):
def _load_corpus(self):
return self._corpus_manager.load_corpus()

def _load_pruned_seeds_into_cuckoo(self):
"""Add all pruned seeds to the cuckoo filter at startup."""
pruned_dir = self.corpus_dir / "seeds" / "pruned"
if not pruned_dir.exists():
return
count = 0
for fh in pruned_dir.rglob("id_*"):
if not fh.is_file():
continue
try:
data = fh.read_bytes()
Comment on lines +4154 to +4157
h = self._seed_key(data)
self.cuckoo_seed_filter.add(h)
count += 1
except OSError:
continue
if count:
log.info("Loaded %d pruned seeds into cuckoo seed filter", count)

def _init_seed_metadata(self):
return self._corpus_manager.init_seed_metadata()

Expand Down Expand Up @@ -4495,7 +4537,20 @@ def _dedup_mutate(self, data: bytes) -> bytes:
That is harmless — it stays reachable on later iterations — and the
filter never yields a false negative, so nothing already executed
slips through as new.

When --cuckoo-seed-filter is enabled, the pruned-seed cuckoo filter
is checked before the exec bloom. If the parent seed's hash matches
a previously-pruned seed, the mutation is skipped (original data
returned) — the seed was already deemed not valuable, so
re-discovering it is pure waste.
"""
# Pruned-seed filter check: if the parent seed was previously
# pruned, skip the mutation entirely and return the original data.
if self.cuckoo_seed_filter is not None:
h = self._seed_key(data)
if self.cuckoo_seed_filter.contains(h):
self._dedup_hits += 1
return data
mutated = self.mutate(data)
if not self._dedup_execs:
return mutated
Expand Down
2 changes: 2 additions & 0 deletions tests/test_bloom_exec_dedup.py
Original file line number Diff line number Diff line change
Expand Up @@ -149,6 +149,8 @@ def __init__(self, mutants, dedup_execs=True, capacity=1000, backend="bloom"):
self._dedup_hits = 0
self._dedup_gaveup = 0
self.mutate_calls = 0
# Cuckoo seed filter: None when feature is disabled
self.cuckoo_seed_filter = None

def mutate(self, data):
self.mutate_calls += 1
Expand Down
191 changes: 191 additions & 0 deletions tests/test_regression_cuckoo_seed_filter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,191 @@
"""Tests for --cuckoo-seed-filter: pruned seeds tracked in a CuckooFilter.

At startup, all seeds under corpus/seeds/pruned/ are added to the filter.
When a seed is pruned during minimization, its hash is added.
In _dedup_mutate(), if a mutation's hash matches a pruned seed, the
mutation is skipped (original data returned).
"""

from pathlib import Path

import pytest

from fuzzer_tool.services.fuzzer import Fuzzer


def _write_seed(corpus_dir: Path, seed: bytes) -> str:
"""Write a seed to corpus/seeds/ and return its 16-char hash."""
from fuzzer_tool.adapters.filesystem import hash_data

h = hash_data(seed)
sub = corpus_dir / "seeds" / h[:2]
sub.mkdir(parents=True, exist_ok=True)
(sub / f"id_{h}").write_bytes(seed)
return h


def _make_corpus_dir(tmp_path: Path) -> Path:
corpus_dir = tmp_path / "corpus"
corpus_dir.mkdir(parents=True)
(corpus_dir / "seeds").mkdir()
return corpus_dir


@pytest.fixture
def corpus_dir(tmp_path: Path) -> Path:
return _make_corpus_dir(tmp_path)


def test_pruned_seed_added_to_cuckoo_at_startup(corpus_dir: Path) -> None:
"""Pruned seeds under corpus/seeds/pruned/ are loaded into the cuckoo filter at startup."""
seed = b"pruned_seed_data"
pruned_dir = corpus_dir / "seeds" / "pruned"
pruned_dir.mkdir(parents=True)
_write_seed(corpus_dir, seed)
# Move seed to pruned
from fuzzer_tool.adapters.filesystem import hash_data

h = hash_data(seed)
(corpus_dir / "seeds" / h[:2] / f"id_{h}").unlink()
pruned_sub = pruned_dir / h[:2]
pruned_sub.mkdir(parents=True)
(pruned_sub / f"id_{h}").write_bytes(seed)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
assert f.cuckoo_seed_filter is not None
assert f.cuckoo_seed_filter.contains(h)


def test_non_pruned_seed_not_in_cuckoo_filter(corpus_dir: Path) -> None:
"""Seeds not pruned should not be in the cuckoo filter."""
seed = b"active_seed_data"
_write_seed(corpus_dir, seed)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
h = f._seed_key(seed)
# Active seed should NOT be in the cuckoo filter
assert not f.cuckoo_seed_filter.contains(h)


def test_pruned_seed_mutation_is_skipped(corpus_dir: Path) -> None:
"""When a seed is pruned, its mutation is skipped in _dedup_mutate."""
seed = b"seed_to_prune"
_write_seed(corpus_dir, seed)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
# Add the seed's hash to the cuckoo filter (simulating pruning)
h = f._seed_key(seed)
f.cuckoo_seed_filter.add(h)

# Mutate the seed - the mutation should be skipped
result = f._dedup_mutate(seed)
# Since the seed's hash is in the cuckoo filter, the mutation
# should return the original data (pruned seed skipped)
assert result == seed


def test_non_pruned_seed_mutates_normally(corpus_dir: Path) -> None:
"""Seeds not in the cuckoo filter should mutate normally."""
seed = b"active_seed"
_write_seed(corpus_dir, seed)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
h = f._seed_key(seed)
# Ensure seed is NOT in the filter
assert not f.cuckoo_seed_filter.contains(h)

# Mutation should proceed normally (return mutated data)
result = f._dedup_mutate(seed)
# Result should be different from original (mutation happened)
assert result != seed


def test_cuckoo_filter_disabled_by_default(corpus_dir: Path) -> None:
"""When --cuckoo-seed-filter is not set, the filter is None."""
seed = b"test_seed"
_write_seed(corpus_dir, seed)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
)
assert f.cuckoo_seed_filter is None


def test_filter_capacity_scaled_to_corpus_size(corpus_dir: Path) -> None:
"""Filter capacity is max(10 * len(corpus), 100_000)."""
# Add 5 seeds
for i in range(5):
_write_seed(corpus_dir, f"seed_{i}".encode())

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
assert f.cuckoo_seed_filter is not None
# Capacity should be at least 100_000 (min) since 5 * 10 = 50 < 100_000
assert f.cuckoo_seed_filter.capacity == 100_000


def test_empty_corpus_filter_has_minimum_capacity(corpus_dir: Path) -> None:
"""Filter capacity is at least 100_000 even with empty corpus."""
f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
assert f.cuckoo_seed_filter is not None
assert f.cuckoo_seed_filter.capacity == 100_000


def test_pruned_seeds_added_to_filter_at_startup(corpus_dir: Path) -> None:
"""All seeds under corpus/seeds/pruned/ are added to the filter at startup."""
pruned_dir = corpus_dir / "seeds" / "pruned"
pruned_dir.mkdir(parents=True)

# Create multiple pruned seeds
seeds = [b"pruned_1", b"pruned_2", b"pruned_3"]
hashes = []
for seed in seeds:
h = _write_seed(corpus_dir, seed)
# Move to pruned
from fuzzer_tool.adapters.filesystem import hash_data

hash_data(seed)
sub = pruned_dir / h[:2]
sub.mkdir(parents=True, exist_ok=True)
(sub / f"id_{h}").write_bytes(seed)
hashes.append(h)

f = Fuzzer(
target="nonexistent",
corpus_dir=str(corpus_dir),
crashes_dir=str(corpus_dir / "crashes"),
cuckoo_seed_filter=True,
)
for h in hashes:
assert f.cuckoo_seed_filter.contains(h), f"Pruned seed {h} should be in cuckoo filter"
Loading