Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -5,3 +5,4 @@ __pycache__/
downloads/
server.log
.test_*/
.claude/
23 changes: 23 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -198,6 +198,29 @@ disruptive to a training set.
`sks_creature, animal ears, solo, ...` for every image, matching the common
LoRA/Dreambooth training convention.

## Remove duplicate images

A standalone "Remove duplicate images" card (below captioning) scans a folder
for images with **byte-identical content** (exact SHA256 match -- the same
rule already used to skip an already-downloaded duplicate during a search)
and removes every copy but one, keeping the alphabetically-first filename in
each group.

- **Recoverable, not a permanent delete**: removed files go to the OS Recycle
Bin via [`send2trash`](https://pypi.org/project/Send2Trash/), confirmed live
by checking `Shell.Application`'s Recycle Bin namespace after a run -- so a
bad run can still be undone from there.
- Any orphaned `.txt` caption for a removed duplicate is removed alongside it
(same basename convention as captioning); the keeper's own caption is left
untouched.
- **Include subfolders** toggles a recursive scan; off by default (folder-only).
- The UI shows a confirmation dialog before running, since this is a bulk
action across a whole folder.

This is exact-content dedup only -- it won't catch near-duplicates (resizes,
re-encodes, crops). See **Where to next** below for perceptual-hash dedup as a
possible future addition.

## Optional LLM providers

Both blocks (query expansion and captioning) are configured independently right
Expand Down
41 changes: 41 additions & 0 deletions app/dedup.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
from __future__ import annotations

import hashlib
from pathlib import Path

from .download import IMAGE_EXTENSIONS

_CHUNK_SIZE = 1 << 20 # 1 MiB -- read large files in chunks instead of all at once


def hash_file(path: Path) -> str:
"""sha256 of a file's raw bytes -- exact-content match, same notion of
"duplicate" used during downloads (download.py's seen_hashes)."""
h = hashlib.sha256()
with path.open("rb") as f:
while chunk := f.read(_CHUNK_SIZE):
h.update(chunk)
return h.hexdigest()


def find_duplicate_groups(root: Path, recursive: bool = False) -> list[list[Path]]:
"""Group images under `root` by exact content hash, returning only the
groups that actually have more than one member (i.e. the real
duplicates) -- each group sorted so the caller can treat index 0 as "the
one to keep" deterministically (alphabetically first) and the rest as
redundant copies to remove.

Groups themselves are also sorted (by their keeper's name) so repeated
runs against an unchanged folder report results in the same order.
"""
pattern = root.rglob("*") if recursive else root.glob("*")
by_hash: dict[str, list[Path]] = {}
for p in pattern:
if not p.is_file() or p.suffix.lower() not in IMAGE_EXTENSIONS:
continue
digest = hash_file(p)
by_hash.setdefault(digest, []).append(p)

groups = [sorted(paths, key=lambda p: p.name) for paths in by_hash.values() if len(paths) > 1]
groups.sort(key=lambda g: g[0].name)
return groups
5 changes: 5 additions & 0 deletions app/download.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,11 @@
"bmp": "bmp",
}

# File suffixes treated as images elsewhere (caption-folder scanning, dedup
# scanning) -- kept alongside EXT_MAP since both describe "what counts as an
# image this app handles", just for different directions (write vs. scan).
IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"}

_INVALID_CHARS = re.compile(r'[<>:"/\\|?*\n\r\t]')


Expand Down
84 changes: 78 additions & 6 deletions app/jobs.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,11 +8,13 @@
from pathlib import Path

import httpx
from send2trash import send2trash

from .captioning import build_captioner
from .download import download_image, sanitize_folder_name
from .dedup import find_duplicate_groups
from .download import IMAGE_EXTENSIONS, download_image, sanitize_folder_name
from .llm_client import LLMClient
from .models import CaptionFolderRequest, JobCreateRequest
from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest
from .search import build_search_provider

logger = logging.getLogger(__name__)
Expand All @@ -24,8 +26,6 @@
OVERFETCH_MIN_EXTRA = 10
MAX_SEARCH_FETCH = 150

IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".webp", ".bmp", ".gif"}


@dataclass
class JobEvent:
Expand All @@ -36,8 +36,8 @@ class JobEvent:
@dataclass
class JobState:
id: str
request: JobCreateRequest | CaptionFolderRequest
kind: str = "download" # "download" | "caption_folder"
request: JobCreateRequest | CaptionFolderRequest | DedupFolderRequest
kind: str = "download" # "download" | "caption_folder" | "dedup"
status: str = "pending" # pending, running, done, error, cancelled
events: list[JobEvent] = field(default_factory=list)
subscribers: list[asyncio.Queue] = field(default_factory=list)
Expand Down Expand Up @@ -69,6 +69,12 @@ def create_caption_folder_job(self, request: CaptionFolderRequest) -> JobState:
self.jobs[job_id] = state
return state

def create_dedup_job(self, request: DedupFolderRequest) -> JobState:
job_id = uuid.uuid4().hex[:12]
state = JobState(id=job_id, kind="dedup", request=request)
self.jobs[job_id] = state
return state

def get(self, job_id: str) -> JobState | None:
return self.jobs.get(job_id)

Expand Down Expand Up @@ -292,5 +298,71 @@ async def run_caption_folder_job(self, state: JobState) -> None:
await captioner.aclose()
await self.emit(state, "status", {"status": state.status, "stats": state.stats})

async def run_dedup_job(self, state: JobState) -> None:
"""Find images with byte-identical content in a folder and move every
copy but one to the Recycle Bin -- independent of any download job."""
req: DedupFolderRequest = state.request
state.status = "running"
await self.emit(state, "status", {"status": "running"})

try:
root = Path(req.folder)
if not root.is_dir():
raise ValueError(f"Folder not found: {req.folder}")

groups = await asyncio.to_thread(find_duplicate_groups, root, req.recursive)
if not groups:
await self.emit(state, "warning", {"message": "No duplicate images found."})

for keeper, *dupes in groups:
if state.cancel_requested:
break

file_index = len(state.downloaded_files)
state.downloaded_files.append(keeper)
state.stats["downloaded"] += 1
await self.emit(
state, "downloaded",
{
"query": f"{len(dupes)} duplicate{'s' if len(dupes) != 1 else ''} found",
"path": str(keeper), "url": "", "index": file_index,
},
)

for dupe in dupes:
if state.cancel_requested:
break
try:
await asyncio.to_thread(send2trash, str(dupe))
# A duplicate image's own caption file (if any) is now
# orphaned -- send it along rather than leave it behind
# pointing at nothing.
txt_path = dupe.with_suffix(".txt")
if txt_path.exists():
await asyncio.to_thread(send2trash, str(txt_path))
state.stats["duplicates"] += 1
await self.emit(
state, "skip",
{
"query": keeper.name, "url": str(dupe),
"error": f"duplicate of {keeper.name} -- moved to Recycle Bin",
"duplicate": True, "filtered": False,
},
)
except Exception as e:
state.stats["errors"] += 1
await self.emit(
state, "warning",
{"message": f"Could not remove duplicate {dupe.name}: {e}"},
)

state.status = "cancelled" if state.cancel_requested else "done"
except Exception as e:
logger.exception("Dedup job %s failed", state.id)
state.status = "error"
await self.emit(state, "warning", {"message": f"Job failed: {e}"})
finally:
await self.emit(state, "status", {"status": state.status, "stats": state.stats})


job_manager = JobManager()
13 changes: 12 additions & 1 deletion app/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
from . import local_models, model_control, model_registry
from .jobs import job_manager
from .llm_client import LLMClient
from .models import CaptionFolderRequest, JobCreateRequest, LLMConfig
from .models import CaptionFolderRequest, DedupFolderRequest, JobCreateRequest, LLMConfig

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -182,6 +182,17 @@ async def caption_folder(req: CaptionFolderRequest):
return {"job_id": state.id}


@app.post("/api/dedup-folder")
async def dedup_folder(req: DedupFolderRequest):
"""Find images with byte-identical content in a folder and move every
copy but one to the Recycle Bin -- independent of any download job."""
if not req.folder.strip():
raise HTTPException(400, "folder is required")
state = job_manager.create_dedup_job(req)
asyncio.create_task(job_manager.run_dedup_job(state))
return {"job_id": state.id}


@app.get("/api/jobs/{job_id}")
async def get_job(job_id: str):
state = job_manager.get(job_id)
Expand Down
11 changes: 11 additions & 0 deletions app/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -129,3 +129,14 @@ class CaptionFolderRequest(BaseModel):
recursive: bool = False
overwrite: bool = False
trigger: TriggerWordConfig = Field(default_factory=TriggerWordConfig)


class DedupFolderRequest(BaseModel):
"""Find images in a folder with byte-identical content (exact sha256
match) and remove every copy but one -- e.g. after downloading the same
query from multiple search providers. Removed files go to the OS Recycle
Bin (via send2trash), not a permanent delete -- this is a bulk action on
a user's dataset, so it stays recoverable."""

folder: str
recursive: bool = False
6 changes: 6 additions & 0 deletions docs/landscape-and-roadmap-notes.md
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,12 @@ The closer competition, especially for the trigger-word feature:
- ✅ **A second (third, fourth...) search backend** -- Yandex, Google, and
booru boards (e621/gelbooru/rule34/danbooru) alongside DuckDuckGo, via
`build_search_provider()` in `app/search/__init__.py`.
- ✅ **Standalone "Remove duplicates" button** -- `app/dedup.py`, scans an
existing folder for exact content-hash matches and removes every copy but
one (to the Recycle Bin via `send2trash`, not a permanent delete). This is
the same exact-hash rule the "no similarity-based de-dup" weak spot below
already referred to, just exposed as its own on-demand action instead of
only running implicitly during a download.

## Candidate next steps (not decided -- discuss before building any of these)

Expand Down
1 change: 1 addition & 0 deletions requirements.txt
Original file line number Diff line number Diff line change
Expand Up @@ -7,3 +7,4 @@ ddgs>=9.0
numpy>=1.26
onnxruntime>=1.18
huggingface_hub>=0.24
send2trash>=1.8
47 changes: 45 additions & 2 deletions static/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -216,6 +216,7 @@ const FIELD_IDS = [
"cap_enabled", "cap_method", "cap_provider", "cap_model", "cap_base_url", "cap_api_key", "cap_timeout", "cap_models_folder", "cap_disable_reasoning",
"cap_wd14_model", "cap_wd14_general_threshold", "cap_wd14_character_threshold",
"capfolder_path", "capfolder_recursive", "capfolder_overwrite",
"dedup_path", "dedup_recursive",
"trigger_enabled", "trigger_word", "trigger_role", "trigger_custom",
];

Expand Down Expand Up @@ -402,11 +403,12 @@ function buildWD14Config() {
// standalone "caption a folder" action -- only one can run at a time) ----
let ws = null;
let currentJobId = null;
let currentJobKind = "download"; // "download" | "caption"
let currentJobKind = "download"; // "download" | "caption" | "dedup"

function setActionButtonsDisabled(disabled) {
$("start-btn").disabled = disabled;
$("capfolder-btn").disabled = disabled;
$("dedup-btn").disabled = disabled;
$("cancel-btn").disabled = !disabled;
}

Expand All @@ -416,8 +418,15 @@ function applyJobKindLabels(kind) {
$("stat-duplicates-row").hidden = true;
$("stat-filtered-row").hidden = true;
$("thumbs-title").textContent = "Captioned images";
} else if (kind === "dedup") {
$("stat-downloaded-label").textContent = "Duplicate groups";
$("stat-duplicates-label").textContent = "Removed";
$("stat-duplicates-row").hidden = false;
$("stat-filtered-row").hidden = true;
$("thumbs-title").textContent = "Kept files";
} else {
$("stat-downloaded-label").textContent = "Downloaded";
$("stat-duplicates-label").textContent = "Duplicates";
$("stat-duplicates-row").hidden = false;
$("stat-filtered-row").hidden = false;
$("thumbs-title").textContent = "Downloaded images";
Expand Down Expand Up @@ -523,7 +532,9 @@ function handleEvent(ev) {
// anything else is a genuine error (network failure, bad data, ...).
const tag = ev.data.duplicate ? "DUP" : ev.data.filtered ? "FLT" : "ERR";
const key = ev.data.duplicate ? "duplicates" : ev.data.filtered ? "filtered" : "errors";
logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.filtered ? "info" : "err");
// Duplicates and filtered-out results are expected/working-as-intended
// outcomes, not failures -- only a real error gets the alarming red.
logLine(`${tag} [${ev.data.query}] ${ev.data.url} - ${ev.data.error}`, ev.data.duplicate || ev.data.filtered ? "info" : "err");
updateStats(bumpStats({ [key]: 1 }));
break;
}
Expand Down Expand Up @@ -658,6 +669,38 @@ $("capfolder-btn").addEventListener("click", async () => {
await startJob("/api/caption-folder", buildCaptionFolderRequest(), "caption");
});

// ---- standalone duplicate removal ----
function buildDedupFolderRequest() {
return {
folder: $("dedup_path").value.trim(),
recursive: $("dedup_recursive").checked,
};
}

$("dedup-btn").addEventListener("click", async () => {
saveForm();
const hint = $("dedup-hint");
hint.textContent = "";
hint.className = "hint";

const folder = $("dedup_path").value.trim();
if (!folder) {
alert("Please set a folder to deduplicate");
return;
}
// A bulk-delete action deserves an explicit confirmation even though it's
// recoverable (Recycle Bin, not a permanent delete) -- the exact count
// isn't known until the scan runs, so this confirms the action itself,
// not a specific number.
const proceed = confirm(
`This will scan "${folder}" for images with identical content and move every copy but one to the Recycle Bin ` +
"(along with any orphaned .txt caption). This can be undone from the Recycle Bin, but not from here. Continue?"
);
if (!proceed) return;

await startJob("/api/dedup-folder", buildDedupFolderRequest(), "dedup");
});

document.getElementById("cancel-btn").addEventListener("click", async () => {
if (!currentJobId) return;
await fetch(`/api/jobs/${currentJobId}/cancel`, { method: "POST" });
Expand Down
25 changes: 24 additions & 1 deletion static/index.html
Original file line number Diff line number Diff line change
Expand Up @@ -332,6 +332,29 @@ <h3>Caption an existing folder</h3>
</div>
</section>

<section class="card">
<h2>Remove duplicate images</h2>
<p class="card-note">Scans a folder for images with byte-identical content (exact hash match, same rule used to skip duplicates during a download) and removes every copy but one. Removed files go to the Recycle Bin, not a permanent delete -- and any orphaned <code>.txt</code> caption for a removed duplicate goes with it.</p>

<div class="field">
<label>Folder to deduplicate</label>
<div class="input-with-button">
<input type="text" id="dedup_path" placeholder="C:\path\to\images" />
<button type="button" class="browse-btn" data-target="dedup_path" data-title="Select folder to deduplicate">Browse&hellip;</button>
</div>
</div>

<label class="checkbox-row">
<input type="checkbox" id="dedup_recursive" />
Include subfolders
</label>

<div class="query-actions">
<button type="button" id="dedup-btn">&#128465; Remove duplicates</button>
<span class="hint" id="dedup-hint"></span>
</div>
</section>

<div class="actions">
<button type="submit" id="start-btn">Start download</button>
<button type="button" id="cancel-btn" disabled>Cancel</button>
Expand All @@ -349,7 +372,7 @@ <h2>Progress</h2>
<span><span id="stat-downloaded-label">Downloaded</span>: <b id="stat-downloaded">0</b></span>
<span id="stat-filtered-row" title="Excluded by your format/size filters -- not an error, working as configured">Filtered: <b id="stat-filtered">0</b></span>
<span>Errors: <b id="stat-errors">0</b></span>
<span id="stat-duplicates-row">Duplicates: <b id="stat-duplicates">0</b></span>
<span id="stat-duplicates-row"><span id="stat-duplicates-label">Duplicates</span>: <b id="stat-duplicates">0</b></span>
<span>Captions: <b id="stat-captioned">0</b></span>
<span id="job-status" class="status-badge">running</span>
</div>
Expand Down
Loading
Loading