diff --git a/.gitignore b/.gitignore index 0452083..55e6d70 100644 --- a/.gitignore +++ b/.gitignore @@ -5,3 +5,4 @@ __pycache__/ downloads/ server.log .test_*/ +.claude/ diff --git a/README.md b/README.md index 73ed51f..0f81c0e 100644 --- a/README.md +++ b/README.md @@ -198,6 +198,29 @@ disruptive to a training set. `sks_creature, animal ears, solo, ...` for every image, matching the common LoRA/Dreambooth training convention. +## Remove duplicate images + +A standalone "Remove duplicate images" card (below captioning) scans a folder +for images with **byte-identical content** (exact SHA256 match -- the same +rule already used to skip an already-downloaded duplicate during a search) +and removes every copy but one, keeping the alphabetically-first filename in +each group. + +- **Recoverable, not a permanent delete**: removed files go to the OS Recycle + Bin via [`send2trash`](https://pypi.org/project/Send2Trash/), confirmed live + by checking `Shell.Application`'s Recycle Bin namespace after a run -- so a + bad run can still be undone from there. +- Any orphaned `.txt` caption for a removed duplicate is removed alongside it + (same basename convention as captioning); the keeper's own caption is left + untouched. +- **Include subfolders** toggles a recursive scan; off by default (folder-only). +- The UI shows a confirmation dialog before running, since this is a bulk + action across a whole folder. + +This is exact-content dedup only -- it won't catch near-duplicates (resizes, +re-encodes, crops). See **Where to next** below for perceptual-hash dedup as a +possible future addition. + ## Optional LLM providers Both blocks (query expansion and captioning) are configured independently right diff --git a/app/dedup.py b/app/dedup.py new file mode 100644 index 0000000..260063a --- /dev/null +++ b/app/dedup.py @@ -0,0 +1,41 @@ +from __future__ import annotations + +import hashlib +from pathlib import Path + +from .download import IMAGE_EXTENSIONS + +_CHUNK_SIZE = 1 << 20 # 1 MiB -- read large files in chunks instead of all at once + + +def hash_file(path: Path) -> str: + """sha256 of a file's raw bytes -- exact-content match, same notion of + "duplicate" used during downloads (download.py's seen_hashes).""" + h = hashlib.sha256() + with path.open("rb") as f: + while chunk := f.read(_CHUNK_SIZE): + h.update(chunk) + return h.hexdigest() + + +def find_duplicate_groups(root: Path, recursive: bool = False) -> list[list[Path]]: + """Group images under `root` by exact content hash, returning only the + groups that actually have more than one member (i.e. the real + duplicates) -- each group sorted so the caller can treat index 0 as "the + one to keep" deterministically (alphabetically first) and the rest as + redundant copies to remove. + + Groups themselves are also sorted (by their keeper's name) so repeated + runs against an unchanged folder report results in the same order. + """ + pattern = root.rglob("*") if recursive else root.glob("*") + by_hash: dict[str, list[Path]] = {} + for p in pattern: + if not p.is_file() or p.suffix.lower() not in IMAGE_EXTENSIONS: + continue + digest = hash_file(p) + by_hash.setdefault(digest, []).append(p) + + groups = [sorted(paths, key=lambda p: p.name) for paths in by_hash.values() if len(paths) > 1] + groups.sort(key=lambda g: g[0].name) + return groups diff --git a/app/download.py b/app/download.py index ce059d2..05abd5c 100644 --- a/app/download.py +++ b/app/download.py @@ -21,6 +21,11 @@ "bmp": "bmp", } +# File suffixes treated as images elsewhere (caption-folder scanning, dedup +# scanning) -- kept alongside EXT_MAP since both describe "what counts as an +# image this app handles", just for different directions (write vs. scan). +IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"} + _INVALID_CHARS = re.compile(r'[<>:"/\\|?*\n\r\t]') diff --git a/app/jobs.py b/app/jobs.py index 7aecdc9..2278cdc 100644 --- a/app/jobs.py +++ b/app/jobs.py @@ -8,11 +8,13 @@ from pathlib import Path import httpx +from send2trash import send2trash from .captioning import build_captioner -from .download import download_image, sanitize_folder_name +from .dedup import find_duplicate_groups +from .download import IMAGE_EXTENSIONS, download_image, sanitize_folder_name from .llm_client import LLMClient -from .models import CaptionFolderRequest, JobCreateRequest +from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest from .search import build_search_provider logger = logging.getLogger(__name__) @@ -24,8 +26,6 @@ OVERFETCH_MIN_EXTRA = 10 MAX_SEARCH_FETCH = 150 -IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"} - @dataclass class JobEvent: @@ -36,8 +36,8 @@ class JobEvent: @dataclass class JobState: id: str - request: JobCreateRequest | CaptionFolderRequest - kind: str = "download" # "download" | "caption_folder" + request: JobCreateRequest | CaptionFolderRequest | DedupFolderRequest + kind: str = "download" # "download" | "caption_folder" | "dedup" status: str = "pending" # pending, running, done, error, cancelled events: list[JobEvent] = field(default_factory=list) subscribers: list[asyncio.Queue] = field(default_factory=list) @@ -69,6 +69,12 @@ def create_caption_folder_job(self, request: CaptionFolderRequest) -> JobState: self.jobs[job_id] = state return state + def create_dedup_job(self, request: DedupFolderRequest) -> JobState: + job_id = uuid.uuid4().hex[:12] + state = JobState(id=job_id, kind="dedup", request=request) + self.jobs[job_id] = state + return state + def get(self, job_id: str) -> JobState | None: return self.jobs.get(job_id) @@ -292,5 +298,71 @@ async def run_caption_folder_job(self, state: JobState) -> None: await captioner.aclose() await self.emit(state, "status", {"status": state.status, "stats": state.stats}) + async def run_dedup_job(self, state: JobState) -> None: + """Find images with byte-identical content in a folder and move every + copy but one to the Recycle Bin -- independent of any download job.""" + req: DedupFolderRequest = state.request + state.status = "running" + await self.emit(state, "status", {"status": "running"}) + + try: + root = Path(req.folder) + if not root.is_dir(): + raise ValueError(f"Folder not found: {req.folder}") + + groups = await asyncio.to_thread(find_duplicate_groups, root, req.recursive) + if not groups: + await self.emit(state, "warning", {"message": "No duplicate images found."}) + + for keeper, *dupes in groups: + if state.cancel_requested: + break + + file_index = len(state.downloaded_files) + state.downloaded_files.append(keeper) + state.stats["downloaded"] += 1 + await self.emit( + state, "downloaded", + { + "query": f"{len(dupes)} duplicate{'s' if len(dupes) != 1 else ''} found", + "path": str(keeper), "url": "", "index": file_index, + }, + ) + + for dupe in dupes: + if state.cancel_requested: + break + try: + await asyncio.to_thread(send2trash, str(dupe)) + # A duplicate image's own caption file (if any) is now + # orphaned -- send it along rather than leave it behind + # pointing at nothing. + txt_path = dupe.with_suffix(".txt") + if txt_path.exists(): + await asyncio.to_thread(send2trash, str(txt_path)) + state.stats["duplicates"] += 1 + await self.emit( + state, "skip", + { + "query": keeper.name, "url": str(dupe), + "error": f"duplicate of {keeper.name} -- moved to Recycle Bin", + "duplicate": True, "filtered": False, + }, + ) + except Exception as e: + state.stats["errors"] += 1 + await self.emit( + state, "warning", + {"message": f"Could not remove duplicate {dupe.name}: {e}"}, + ) + + state.status = "cancelled" if state.cancel_requested else "done" + except Exception as e: + logger.exception("Dedup job %s failed", state.id) + state.status = "error" + await self.emit(state, "warning", {"message": f"Job failed: {e}"}) + finally: + await self.emit(state, "status", {"status": state.status, "stats": state.stats}) + job_manager = JobManager() diff --git a/app/main.py b/app/main.py index 9c6106e..bbda0fd 100644 --- a/app/main.py +++ b/app/main.py @@ -14,7 +14,7 @@ from . import local_models, model_control, model_registry from .jobs import job_manager from .llm_client import LLMClient -from .models import CaptionFolderRequest, JobCreateRequest, LLMConfig +from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest, LLMConfig logger = logging.getLogger(__name__) @@ -182,6 +182,17 @@ async def caption_folder(req: CaptionFolderRequest): return {"job_id": state.id} +@app.post("/api/dedup-folder") +async def dedup_folder(req: DedupFolderRequest): + """Find images with byte-identical content in a folder and move every + copy but one to the Recycle Bin -- independent of any download job.""" + if not req.folder.strip(): + raise HTTPException(400, "folder is required") + state = job_manager.create_dedup_job(req) + asyncio.create_task(job_manager.run_dedup_job(state)) + return {"job_id": state.id} + + @app.get("/api/jobs/{job_id}") async def get_job(job_id: str): state = job_manager.get(job_id) diff --git a/app/models.py b/app/models.py index 0538650..bbf1933 100644 --- a/app/models.py +++ b/app/models.py @@ -129,3 +129,14 @@ class CaptionFolderRequest(BaseModel): recursive: bool = False overwrite: bool = False trigger: TriggerWordConfig = Field(default_factory=TriggerWordConfig) + + +class DedupFolderRequest(BaseModel): + """Find images in a folder with byte-identical content (exact sha256 + match) and remove every copy but one -- e.g. after downloading the same + query from multiple search providers. Removed files go to the OS Recycle + Bin (via send2trash), not a permanent delete -- this is a bulk action on + a user's dataset, so it stays recoverable.""" + + folder: str + recursive: bool = False diff --git a/docs/landscape-and-roadmap-notes.md b/docs/landscape-and-roadmap-notes.md index f694376..7f4e015 100644 --- a/docs/landscape-and-roadmap-notes.md +++ b/docs/landscape-and-roadmap-notes.md @@ -71,6 +71,12 @@ The closer competition, especially for the trigger-word feature: - ✅ **A second (third, fourth...) search backend** -- Yandex, Google, and booru boards (e621/gelbooru/rule34/danbooru) alongside DuckDuckGo, via `build_search_provider()` in `app/search/__init__.py`. +- ✅ **Standalone "Remove duplicates" button** -- `app/dedup.py`, scans an + existing folder for exact content-hash matches and removes every copy but + one (to the Recycle Bin via `send2trash`, not a permanent delete). This is + the same exact-hash rule the "no similarity-based de-dup" weak spot below + already referred to, just exposed as its own on-demand action instead of + only running implicitly during a download. ## Candidate next steps (not decided -- discuss before building any of these) diff --git a/requirements.txt b/requirements.txt index e5335ca..dc8ee85 100644 --- a/requirements.txt +++ b/requirements.txt @@ -7,3 +7,4 @@ ddgs>=9.0 numpy>=1.26 onnxruntime>=1.18 huggingface_hub>=0.24 +send2trash>=1.8 diff --git a/static/app.js b/static/app.js index 9e25bce..6ac2b24 100644 --- a/static/app.js +++ b/static/app.js @@ -216,6 +216,7 @@ const FIELD_IDS = [ "cap_enabled", "cap_method", "cap_provider", "cap_model", "cap_base_url", "cap_api_key", "cap_timeout", "cap_models_folder", "cap_disable_reasoning", "cap_wd14_model", "cap_wd14_general_threshold", "cap_wd14_character_threshold", "capfolder_path", "capfolder_recursive", "capfolder_overwrite", + "dedup_path", "dedup_recursive", "trigger_enabled", "trigger_word", "trigger_role", "trigger_custom", ]; @@ -402,11 +403,12 @@ function buildWD14Config() { // standalone "caption a folder" action -- only one can run at a time) ---- let ws = null; let currentJobId = null; -let currentJobKind = "download"; // "download" | "caption" +let currentJobKind = "download"; // "download" | "caption" | "dedup" function setActionButtonsDisabled(disabled) { $("start-btn").disabled = disabled; $("capfolder-btn").disabled = disabled; + $("dedup-btn").disabled = disabled; $("cancel-btn").disabled = !disabled; } @@ -416,8 +418,15 @@ function applyJobKindLabels(kind) { $("stat-duplicates-row").hidden = true; $("stat-filtered-row").hidden = true; $("thumbs-title").textContent = "Captioned images"; + } else if (kind === "dedup") { + $("stat-downloaded-label").textContent = "Duplicate groups"; + $("stat-duplicates-label").textContent = "Removed"; + $("stat-duplicates-row").hidden = false; + $("stat-filtered-row").hidden = true; + $("thumbs-title").textContent = "Kept files"; } else { $("stat-downloaded-label").textContent = "Downloaded"; + $("stat-duplicates-label").textContent = "Duplicates"; $("stat-duplicates-row").hidden = false; $("stat-filtered-row").hidden = false; $("thumbs-title").textContent = "Downloaded images"; @@ -523,7 +532,9 @@ function handleEvent(ev) { // anything else is a genuine error (network failure, bad data, ...). const tag = ev.data.duplicate ? "DUP" : ev.data.filtered ? "FLT" : "ERR"; const key = ev.data.duplicate ? "duplicates" : ev.data.filtered ? "filtered" : "errors"; - logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.filtered ? "info" : "err"); + // Duplicates and filtered-out results are expected/working-as-intended + // outcomes, not failures -- only a real error gets the alarming red. + logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.duplicate || ev.data.filtered ? "info" : "err"); updateStats(bumpStats({ [key]: 1 })); break; } @@ -658,6 +669,38 @@ $("capfolder-btn").addEventListener("click", async () => { await startJob("/api/caption-folder", buildCaptionFolderRequest(), "caption"); }); +// ---- standalone duplicate removal ---- +function buildDedupFolderRequest() { + return { + folder: $("dedup_path").value.trim(), + recursive: $("dedup_recursive").checked, + }; +} + +$("dedup-btn").addEventListener("click", async () => { + saveForm(); + const hint = $("dedup-hint"); + hint.textContent = ""; + hint.className = "hint"; + + const folder = $("dedup_path").value.trim(); + if (!folder) { + alert("Please set a folder to deduplicate"); + return; + } + // A bulk-delete action deserves an explicit confirmation even though it's + // recoverable (Recycle Bin, not a permanent delete) -- the exact count + // isn't known until the scan runs, so this confirms the action itself, + // not a specific number. + const proceed = confirm( + `This will scan "${folder}" for images with identical content and move every copy but one to the Recycle Bin ` + + "(along with any orphaned .txt caption). This can be undone from the Recycle Bin, but not from here. Continue?" + ); + if (!proceed) return; + + await startJob("/api/dedup-folder", buildDedupFolderRequest(), "dedup"); +}); + document.getElementById("cancel-btn").addEventListener("click", async () => { if (!currentJobId) return; await fetch(`/api/jobs/${currentJobId}/cancel`, { method: "POST" }); diff --git a/static/index.html b/static/index.html index 4a0606d..8846dcb 100644 --- a/static/index.html +++ b/static/index.html @@ -332,6 +332,29 @@

Caption an existing folder

+
+

Remove duplicate images

+

Scans a folder for images with byte-identical content (exact hash match, same rule used to skip duplicates during a download) and removes every copy but one. Removed files go to the Recycle Bin, not a permanent delete -- and any orphaned .txt caption for a removed duplicate goes with it.

+ +
+ +
+ + +
+
+ + + +
+ + +
+
+
@@ -349,7 +372,7 @@

Progress

Downloaded: 0 Filtered: 0 Errors: 0 - Duplicates: 0 + Duplicates: 0 Captions: 0 running
diff --git a/tests/test_dedup.py b/tests/test_dedup.py new file mode 100644 index 0000000..a2e881a --- /dev/null +++ b/tests/test_dedup.py @@ -0,0 +1,100 @@ +from __future__ import annotations + +from app.dedup import find_duplicate_groups, hash_file + + +class TestHashFile: + def test_identical_content_hashes_the_same(self, tmp_path): + a = tmp_path / "a.jpg" + b = tmp_path / "b.jpg" + a.write_bytes(b"same content") + b.write_bytes(b"same content") + assert hash_file(a) == hash_file(b) + + def test_different_content_hashes_differently(self, tmp_path): + a = tmp_path / "a.jpg" + b = tmp_path / "b.jpg" + a.write_bytes(b"content one") + b.write_bytes(b"content two") + assert hash_file(a) != hash_file(b) + + def test_large_file_is_hashed_in_chunks_correctly(self, tmp_path): + # bigger than _CHUNK_SIZE (1 MiB) so the chunked read loop actually + # runs more than once -- content is still just repeated bytes, but + # this exercises the loop instead of a single read() call. + a = tmp_path / "a.bin" + b = tmp_path / "b.bin" + content = b"x" * (2 * 1024 * 1024 + 137) + a.write_bytes(content) + b.write_bytes(content) + assert hash_file(a) == hash_file(b) + + +class TestFindDuplicateGroups: + def test_no_duplicates_returns_empty_list(self, tmp_path): + (tmp_path / "a.jpg").write_bytes(b"one") + (tmp_path / "b.jpg").write_bytes(b"two") + assert find_duplicate_groups(tmp_path) == [] + + def test_finds_a_simple_duplicate_pair(self, tmp_path): + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "b.jpg").write_bytes(b"same") + groups = find_duplicate_groups(tmp_path) + assert len(groups) == 1 + assert {p.name for p in groups[0]} == {"a.jpg", "b.jpg"} + + def test_keeper_is_alphabetically_first_in_the_group(self, tmp_path): + (tmp_path / "z.jpg").write_bytes(b"same") + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "m.jpg").write_bytes(b"same") + groups = find_duplicate_groups(tmp_path) + keeper, *dupes = groups[0] + assert keeper.name == "a.jpg" + assert {p.name for p in dupes} == {"m.jpg", "z.jpg"} + + def test_groups_of_three_or_more_are_handled(self, tmp_path): + for name in ("a.jpg", "b.jpg", "c.jpg"): + (tmp_path / name).write_bytes(b"same") + groups = find_duplicate_groups(tmp_path) + assert len(groups) == 1 + assert len(groups[0]) == 3 + + def test_multiple_independent_duplicate_groups(self, tmp_path): + (tmp_path / "a1.jpg").write_bytes(b"group a") + (tmp_path / "a2.jpg").write_bytes(b"group a") + (tmp_path / "b1.jpg").write_bytes(b"group b") + (tmp_path / "b2.jpg").write_bytes(b"group b") + (tmp_path / "unique.jpg").write_bytes(b"only one") + + groups = find_duplicate_groups(tmp_path) + assert len(groups) == 2 + names = [{p.name for p in g} for g in groups] + assert {"a1.jpg", "a2.jpg"} in names + assert {"b1.jpg", "b2.jpg"} in names + + def test_non_image_files_are_ignored_even_if_content_matches(self, tmp_path): + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "a.txt").write_bytes(b"same") # caption file, not an image + (tmp_path / "readme.md").write_bytes(b"same") + assert find_duplicate_groups(tmp_path) == [] + + def test_non_recursive_by_default_ignores_subfolders(self, tmp_path): + (tmp_path / "a.jpg").write_bytes(b"same") + sub = tmp_path / "sub" + sub.mkdir() + (sub / "b.jpg").write_bytes(b"same") + + assert find_duplicate_groups(tmp_path, recursive=False) == [] + + groups = find_duplicate_groups(tmp_path, recursive=True) + assert len(groups) == 1 + assert {p.name for p in groups[0]} == {"a.jpg", "b.jpg"} + + def test_case_insensitive_extension_matching(self, tmp_path): + (tmp_path / "a.JPG").write_bytes(b"same") + (tmp_path / "b.jpg").write_bytes(b"same") + groups = find_duplicate_groups(tmp_path) + assert len(groups) == 1 + + def test_empty_folder_returns_empty_list(self, tmp_path): + assert find_duplicate_groups(tmp_path) == [] diff --git a/tests/test_jobs.py b/tests/test_jobs.py index bdaf1ea..d95e7c6 100644 --- a/tests/test_jobs.py +++ b/tests/test_jobs.py @@ -11,6 +11,7 @@ from app.models import ( CaptionFolderRequest, CaptioningConfig, + DedupFolderRequest, FilterConfig, JobCreateRequest, LLMConfig, @@ -480,3 +481,147 @@ async def caption_image(self, image_bytes, mime="image/jpeg", extra_instruction= assert received["instruction"] is not None assert "zog" in received["instruction"] + + +class TestRunDedupJob: + def _fake_send2trash(self, monkeypatch, removed: list[str] | None = None, fail_for: set[str] | None = None): + """Stand-in for send2trash that actually deletes the file (so tests + can assert on real filesystem state afterwards) instead of hitting + the real OS Recycle Bin -- optionally raising for specific paths to + exercise the failure path.""" + removed = removed if removed is not None else [] + fail_for = fail_for or set() + + def fake(path): + if path in fail_for: + raise OSError(f"simulated failure removing {path}") + removed.append(path) + Path(path).unlink() + + monkeypatch.setattr(jobs_module, "send2trash", fake) + return removed + + async def test_removes_all_but_one_of_a_duplicate_pair(self, manager, monkeypatch, tmp_path): + self._fake_send2trash(monkeypatch) + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "b.jpg").write_bytes(b"same") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert state.status == "done" + remaining = sorted(p.name for p in tmp_path.glob("*.jpg")) + assert remaining == ["a.jpg"] # alphabetically-first kept, "b.jpg" removed + + async def test_stats_downloaded_is_groups_and_duplicates_is_removed_count(self, manager, monkeypatch, tmp_path): + self._fake_send2trash(monkeypatch) + for name in ("a1.jpg", "a2.jpg", "a3.jpg"): + (tmp_path / name).write_bytes(b"group a") + (tmp_path / "b1.jpg").write_bytes(b"group b") + (tmp_path / "b2.jpg").write_bytes(b"group b") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert state.stats["downloaded"] == 2 # 2 groups + assert state.stats["duplicates"] == 3 # 2 removed from group a + 1 from group b + + async def test_orphaned_caption_file_is_removed_alongside_its_image(self, manager, monkeypatch, tmp_path): + removed = self._fake_send2trash(monkeypatch) + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "b.jpg").write_bytes(b"same") + (tmp_path / "b.txt").write_text("stale caption for a duplicate", encoding="utf-8") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert str(tmp_path / "b.txt") in removed + assert not (tmp_path / "b.txt").exists() + + async def test_keepers_caption_file_is_left_alone(self, manager, monkeypatch, tmp_path): + removed = self._fake_send2trash(monkeypatch) + (tmp_path / "a.jpg").write_bytes(b"same") + (tmp_path / "a.txt").write_text("this caption belongs to the keeper", encoding="utf-8") + (tmp_path / "b.jpg").write_bytes(b"same") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert str(tmp_path / "a.txt") not in removed + assert (tmp_path / "a.txt").exists() + + async def test_no_duplicates_finishes_done_with_a_warning_and_zero_stats(self, manager, monkeypatch, tmp_path): + self._fake_send2trash(monkeypatch) + (tmp_path / "a.jpg").write_bytes(b"one") + (tmp_path / "b.jpg").write_bytes(b"two") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert state.status == "done" + assert state.stats["downloaded"] == 0 + assert state.stats["duplicates"] == 0 + warnings = [e for e in state.events if e.type == "warning"] + assert any("No duplicate images found" in e.data["message"] for e in warnings) + + async def test_missing_folder_reports_error_status(self, manager, tmp_path): + req = DedupFolderRequest(folder=str(tmp_path / "nope")) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + assert state.status == "error" + + async def test_cancellation_stops_before_removing_more_groups(self, manager, monkeypatch, tmp_path): + self._fake_send2trash(monkeypatch) + (tmp_path / "a1.jpg").write_bytes(b"group a") + (tmp_path / "a2.jpg").write_bytes(b"group a") + (tmp_path / "b1.jpg").write_bytes(b"group b") + (tmp_path / "b2.jpg").write_bytes(b"group b") + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + state.cancel_requested = True + await manager.run_dedup_job(state) + + assert state.status == "cancelled" + assert state.stats["duplicates"] == 0 # nothing removed once already cancelled + + async def test_removal_failure_is_counted_as_an_error_and_does_not_stop_the_job( + self, manager, monkeypatch, tmp_path + ): + (tmp_path / "a.jpg").write_bytes(b"group a") + (tmp_path / "a2.jpg").write_bytes(b"group a") + (tmp_path / "b1.jpg").write_bytes(b"group b") + (tmp_path / "b2.jpg").write_bytes(b"group b") + self._fake_send2trash(monkeypatch, fail_for={str(tmp_path / "a2.jpg")}) + + req = DedupFolderRequest(folder=str(tmp_path)) + state = manager.create_dedup_job(req) + await manager.run_dedup_job(state) + + assert state.status == "done" + assert state.stats["errors"] == 1 + assert state.stats["duplicates"] == 1 # the other group's duplicate still got removed + assert (tmp_path / "a2.jpg").exists() # the failed removal is still there + assert not (tmp_path / "b2.jpg").exists() + + async def test_recursive_flag_controls_whether_subfolders_are_scanned(self, manager, monkeypatch, tmp_path): + self._fake_send2trash(monkeypatch) + (tmp_path / "a.jpg").write_bytes(b"same") + sub = tmp_path / "sub" + sub.mkdir() + (sub / "b.jpg").write_bytes(b"same") + + non_recursive = DedupFolderRequest(folder=str(tmp_path), recursive=False) + state = manager.create_dedup_job(non_recursive) + await manager.run_dedup_job(state) + assert state.stats["duplicates"] == 0 + + recursive = DedupFolderRequest(folder=str(tmp_path), recursive=True) + state2 = manager.create_dedup_job(recursive) + await manager.run_dedup_job(state2) + assert state2.stats["duplicates"] == 1 diff --git a/tests/test_main.py b/tests/test_main.py index c0ae043..1bd081e 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -239,6 +239,31 @@ async def fake_run(state): assert "job_id" in resp.json() +class TestDedupFolderValidation: + def test_rejects_empty_folder(self, client): + resp = client.post("/api/dedup-folder", json={"folder": " "}) + assert resp.status_code == 400 + + def test_accepts_a_valid_request(self, client, monkeypatch): + async def fake_run(state): + pass + + monkeypatch.setattr(job_manager, "run_dedup_job", fake_run) + resp = client.post("/api/dedup-folder", json={"folder": "C:/data"}) + assert resp.status_code == 200 + assert "job_id" in resp.json() + + def test_recursive_flag_is_passed_through(self, client, monkeypatch): + captured = {} + + async def fake_run(state): + captured["recursive"] = state.request.recursive + + monkeypatch.setattr(job_manager, "run_dedup_job", fake_run) + resp = client.post("/api/dedup-folder", json={"folder": "C:/data", "recursive": True}) + assert resp.status_code == 200 + + class TestJobStatusAndImages: def test_get_unknown_job_is_404(self, client): assert client.get("/api/jobs/doesnotexist").status_code == 404