diff --git a/.gitignore b/.gitignore index 0452083..55e6d70 100644 --- a/.gitignore +++ b/.gitignore @@ -5,3 +5,4 @@ __pycache__/ downloads/ server.log .test_*/ +.claude/ diff --git a/README.md b/README.md index 73ed51f..0f81c0e 100644 --- a/README.md +++ b/README.md @@ -198,6 +198,29 @@ disruptive to a training set. `sks_creature, animal ears, solo, ...` for every image, matching the common LoRA/Dreambooth training convention. +## Remove duplicate images + +A standalone "Remove duplicate images" card (below captioning) scans a folder +for images with **byte-identical content** (exact SHA256 match -- the same +rule already used to skip an already-downloaded duplicate during a search) +and removes every copy but one, keeping the alphabetically-first filename in +each group. + +- **Recoverable, not a permanent delete**: removed files go to the OS Recycle + Bin via [`send2trash`](https://pypi.org/project/Send2Trash/), confirmed live + by checking `Shell.Application`'s Recycle Bin namespace after a run -- so a + bad run can still be undone from there. +- Any orphaned `.txt` caption for a removed duplicate is removed alongside it + (same basename convention as captioning); the keeper's own caption is left + untouched. +- **Include subfolders** toggles a recursive scan; off by default (folder-only). +- The UI shows a confirmation dialog before running, since this is a bulk + action across a whole folder. + +This is exact-content dedup only -- it won't catch near-duplicates (resizes, +re-encodes, crops). See **Where to next** below for perceptual-hash dedup as a +possible future addition. + ## Optional LLM providers Both blocks (query expansion and captioning) are configured independently right diff --git a/app/dedup.py b/app/dedup.py new file mode 100644 index 0000000..260063a --- /dev/null +++ b/app/dedup.py @@ -0,0 +1,41 @@ +from __future__ import annotations + +import hashlib +from pathlib import Path + +from .download import IMAGE_EXTENSIONS + +_CHUNK_SIZE = 1 << 20 # 1 MiB -- read large files in chunks instead of all at once + + +def hash_file(path: Path) -> str: + """sha256 of a file's raw bytes -- exact-content match, same notion of + "duplicate" used during downloads (download.py's seen_hashes).""" + h = hashlib.sha256() + with path.open("rb") as f: + while chunk := f.read(_CHUNK_SIZE): + h.update(chunk) + return h.hexdigest() + + +def find_duplicate_groups(root: Path, recursive: bool = False) -> list[list[Path]]: + """Group images under `root` by exact content hash, returning only the + groups that actually have more than one member (i.e. the real + duplicates) -- each group sorted so the caller can treat index 0 as "the + one to keep" deterministically (alphabetically first) and the rest as + redundant copies to remove. + + Groups themselves are also sorted (by their keeper's name) so repeated + runs against an unchanged folder report results in the same order. + """ + pattern = root.rglob("*") if recursive else root.glob("*") + by_hash: dict[str, list[Path]] = {} + for p in pattern: + if not p.is_file() or p.suffix.lower() not in IMAGE_EXTENSIONS: + continue + digest = hash_file(p) + by_hash.setdefault(digest, []).append(p) + + groups = [sorted(paths, key=lambda p: p.name) for paths in by_hash.values() if len(paths) > 1] + groups.sort(key=lambda g: g[0].name) + return groups diff --git a/app/download.py b/app/download.py index ce059d2..05abd5c 100644 --- a/app/download.py +++ b/app/download.py @@ -21,6 +21,11 @@ "bmp": "bmp", } +# File suffixes treated as images elsewhere (caption-folder scanning, dedup +# scanning) -- kept alongside EXT_MAP since both describe "what counts as an +# image this app handles", just for different directions (write vs. scan). +IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"} + _INVALID_CHARS = re.compile(r'[<>:"/\\|?*\n\r\t]') diff --git a/app/jobs.py b/app/jobs.py index 7aecdc9..2278cdc 100644 --- a/app/jobs.py +++ b/app/jobs.py @@ -8,11 +8,13 @@ from pathlib import Path import httpx +from send2trash import send2trash from .captioning import build_captioner -from .download import download_image, sanitize_folder_name +from .dedup import find_duplicate_groups +from .download import IMAGE_EXTENSIONS, download_image, sanitize_folder_name from .llm_client import LLMClient -from .models import CaptionFolderRequest, JobCreateRequest +from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest from .search import build_search_provider logger = logging.getLogger(__name__) @@ -24,8 +26,6 @@ OVERFETCH_MIN_EXTRA = 10 MAX_SEARCH_FETCH = 150 -IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"} - @dataclass class JobEvent: @@ -36,8 +36,8 @@ class JobEvent: @dataclass class JobState: id: str - request: JobCreateRequest | CaptionFolderRequest - kind: str = "download" # "download" | "caption_folder" + request: JobCreateRequest | CaptionFolderRequest | DedupFolderRequest + kind: str = "download" # "download" | "caption_folder" | "dedup" status: str = "pending" # pending, running, done, error, cancelled events: list[JobEvent] = field(default_factory=list) subscribers: list[asyncio.Queue] = field(default_factory=list) @@ -69,6 +69,12 @@ def create_caption_folder_job(self, request: CaptionFolderRequest) -> JobState: self.jobs[job_id] = state return state + def create_dedup_job(self, request: DedupFolderRequest) -> JobState: + job_id = uuid.uuid4().hex[:12] + state = JobState(id=job_id, kind="dedup", request=request) + self.jobs[job_id] = state + return state + def get(self, job_id: str) -> JobState | None: return self.jobs.get(job_id) @@ -292,5 +298,71 @@ async def run_caption_folder_job(self, state: JobState) -> None: await captioner.aclose() await self.emit(state, "status", {"status": state.status, "stats": state.stats}) + async def run_dedup_job(self, state: JobState) -> None: + """Find images with byte-identical content in a folder and move every + copy but one to the Recycle Bin -- independent of any download job.""" + req: DedupFolderRequest = state.request + state.status = "running" + await self.emit(state, "status", {"status": "running"}) + + try: + root = Path(req.folder) + if not root.is_dir(): + raise ValueError(f"Folder not found: {req.folder}") + + groups = await asyncio.to_thread(find_duplicate_groups, root, req.recursive) + if not groups: + await self.emit(state, "warning", {"message": "No duplicate images found."}) + + for keeper, *dupes in groups: + if state.cancel_requested: + break + + file_index = len(state.downloaded_files) + state.downloaded_files.append(keeper) + state.stats["downloaded"] += 1 + await self.emit( + state, "downloaded", + { + "query": f"{len(dupes)} duplicate{'s' if len(dupes) != 1 else ''} found", + "path": str(keeper), "url": "", "index": file_index, + }, + ) + + for dupe in dupes: + if state.cancel_requested: + break + try: + await asyncio.to_thread(send2trash, str(dupe)) + # A duplicate image's own caption file (if any) is now + # orphaned -- send it along rather than leave it behind + # pointing at nothing. + txt_path = dupe.with_suffix(".txt") + if txt_path.exists(): + await asyncio.to_thread(send2trash, str(txt_path)) + state.stats["duplicates"] += 1 + await self.emit( + state, "skip", + { + "query": keeper.name, "url": str(dupe), + "error": f"duplicate of {keeper.name} -- moved to Recycle Bin", + "duplicate": True, "filtered": False, + }, + ) + except Exception as e: + state.stats["errors"] += 1 + await self.emit( + state, "warning", + {"message": f"Could not remove duplicate {dupe.name}: {e}"}, + ) + + state.status = "cancelled" if state.cancel_requested else "done" + except Exception as e: + logger.exception("Dedup job %s failed", state.id) + state.status = "error" + await self.emit(state, "warning", {"message": f"Job failed: {e}"}) + finally: + await self.emit(state, "status", {"status": state.status, "stats": state.stats}) + job_manager = JobManager() diff --git a/app/main.py b/app/main.py index 9c6106e..bbda0fd 100644 --- a/app/main.py +++ b/app/main.py @@ -14,7 +14,7 @@ from . import local_models, model_control, model_registry from .jobs import job_manager from .llm_client import LLMClient -from .models import CaptionFolderRequest, JobCreateRequest, LLMConfig +from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest, LLMConfig logger = logging.getLogger(__name__) @@ -182,6 +182,17 @@ async def caption_folder(req: CaptionFolderRequest): return {"job_id": state.id} +@app.post("/api/dedup-folder") +async def dedup_folder(req: DedupFolderRequest): + """Find images with byte-identical content in a folder and move every + copy but one to the Recycle Bin -- independent of any download job.""" + if not req.folder.strip(): + raise HTTPException(400, "folder is required") + state = job_manager.create_dedup_job(req) + asyncio.create_task(job_manager.run_dedup_job(state)) + return {"job_id": state.id} + + @app.get("/api/jobs/{job_id}") async def get_job(job_id: str): state = job_manager.get(job_id) diff --git a/app/models.py b/app/models.py index 0538650..bbf1933 100644 --- a/app/models.py +++ b/app/models.py @@ -129,3 +129,14 @@ class CaptionFolderRequest(BaseModel): recursive: bool = False overwrite: bool = False trigger: TriggerWordConfig = Field(default_factory=TriggerWordConfig) + + +class DedupFolderRequest(BaseModel): + """Find images in a folder with byte-identical content (exact sha256 + match) and remove every copy but one -- e.g. after downloading the same + query from multiple search providers. Removed files go to the OS Recycle + Bin (via send2trash), not a permanent delete -- this is a bulk action on + a user's dataset, so it stays recoverable.""" + + folder: str + recursive: bool = False diff --git a/docs/landscape-and-roadmap-notes.md b/docs/landscape-and-roadmap-notes.md index f694376..7f4e015 100644 --- a/docs/landscape-and-roadmap-notes.md +++ b/docs/landscape-and-roadmap-notes.md @@ -71,6 +71,12 @@ The closer competition, especially for the trigger-word feature: - ✅ **A second (third, fourth...) search backend** -- Yandex, Google, and booru boards (e621/gelbooru/rule34/danbooru) alongside DuckDuckGo, via `build_search_provider()` in `app/search/__init__.py`. +- ✅ **Standalone "Remove duplicates" button** -- `app/dedup.py`, scans an + existing folder for exact content-hash matches and removes every copy but + one (to the Recycle Bin via `send2trash`, not a permanent delete). This is + the same exact-hash rule the "no similarity-based de-dup" weak spot below + already referred to, just exposed as its own on-demand action instead of + only running implicitly during a download. ## Candidate next steps (not decided -- discuss before building any of these) diff --git a/requirements.txt b/requirements.txt index e5335ca..dc8ee85 100644 --- a/requirements.txt +++ b/requirements.txt @@ -7,3 +7,4 @@ ddgs>=9.0 numpy>=1.26 onnxruntime>=1.18 huggingface_hub>=0.24 +send2trash>=1.8 diff --git a/static/app.js b/static/app.js index 9e25bce..6ac2b24 100644 --- a/static/app.js +++ b/static/app.js @@ -216,6 +216,7 @@ const FIELD_IDS = [ "cap_enabled", "cap_method", "cap_provider", "cap_model", "cap_base_url", "cap_api_key", "cap_timeout", "cap_models_folder", "cap_disable_reasoning", "cap_wd14_model", "cap_wd14_general_threshold", "cap_wd14_character_threshold", "capfolder_path", "capfolder_recursive", "capfolder_overwrite", + "dedup_path", "dedup_recursive", "trigger_enabled", "trigger_word", "trigger_role", "trigger_custom", ]; @@ -402,11 +403,12 @@ function buildWD14Config() { // standalone "caption a folder" action -- only one can run at a time) ---- let ws = null; let currentJobId = null; -let currentJobKind = "download"; // "download" | "caption" +let currentJobKind = "download"; // "download" | "caption" | "dedup" function setActionButtonsDisabled(disabled) { $("start-btn").disabled = disabled; $("capfolder-btn").disabled = disabled; + $("dedup-btn").disabled = disabled; $("cancel-btn").disabled = !disabled; } @@ -416,8 +418,15 @@ function applyJobKindLabels(kind) { $("stat-duplicates-row").hidden = true; $("stat-filtered-row").hidden = true; $("thumbs-title").textContent = "Captioned images"; + } else if (kind === "dedup") { + $("stat-downloaded-label").textContent = "Duplicate groups"; + $("stat-duplicates-label").textContent = "Removed"; + $("stat-duplicates-row").hidden = false; + $("stat-filtered-row").hidden = true; + $("thumbs-title").textContent = "Kept files"; } else { $("stat-downloaded-label").textContent = "Downloaded"; + $("stat-duplicates-label").textContent = "Duplicates"; $("stat-duplicates-row").hidden = false; $("stat-filtered-row").hidden = false; $("thumbs-title").textContent = "Downloaded images"; @@ -523,7 +532,9 @@ function handleEvent(ev) { // anything else is a genuine error (network failure, bad data, ...). const tag = ev.data.duplicate ? "DUP" : ev.data.filtered ? "FLT" : "ERR"; const key = ev.data.duplicate ? "duplicates" : ev.data.filtered ? "filtered" : "errors"; - logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.filtered ? "info" : "err"); + // Duplicates and filtered-out results are expected/working-as-intended + // outcomes, not failures -- only a real error gets the alarming red. + logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.duplicate || ev.data.filtered ? "info" : "err"); updateStats(bumpStats({ [key]: 1 })); break; } @@ -658,6 +669,38 @@ $("capfolder-btn").addEventListener("click", async () => { await startJob("/api/caption-folder", buildCaptionFolderRequest(), "caption"); }); +// ---- standalone duplicate removal ---- +function buildDedupFolderRequest() { + return { + folder: $("dedup_path").value.trim(), + recursive: $("dedup_recursive").checked, + }; +} + +$("dedup-btn").addEventListener("click", async () => { + saveForm(); + const hint = $("dedup-hint"); + hint.textContent = ""; + hint.className = "hint"; + + const folder = $("dedup_path").value.trim(); + if (!folder) { + alert("Please set a folder to deduplicate"); + return; + } + // A bulk-delete action deserves an explicit confirmation even though it's + // recoverable (Recycle Bin, not a permanent delete) -- the exact count + // isn't known until the scan runs, so this confirms the action itself, + // not a specific number. + const proceed = confirm( + `This will scan "${folder}" for images with identical content and move every copy but one to the Recycle Bin ` + + "(along with any orphaned .txt caption). This can be undone from the Recycle Bin, but not from here. Continue?" + ); + if (!proceed) return; + + await startJob("/api/dedup-folder", buildDedupFolderRequest(), "dedup"); +}); + document.getElementById("cancel-btn").addEventListener("click", async () => { if (!currentJobId) return; await fetch(`/api/jobs/${currentJobId}/cancel`, { method: "POST" }); diff --git a/static/index.html b/static/index.html index 4a0606d..8846dcb 100644 --- a/static/index.html +++ b/static/index.html @@ -332,6 +332,29 @@
Scans a folder for images with byte-identical content (exact hash match, same rule used to skip duplicates during a download) and removes every copy but one. Removed files go to the Recycle Bin, not a permanent delete -- and any orphaned .txt caption for a removed duplicate goes with it.