From b77b5e8cadc71437d80b968c2e1d4c56c90a942b Mon Sep 17 00:00:00 2001 From: EnragedAntelope Date: Sat, 3 Oct 2026 17:59:09 -0400 Subject: [PATCH] =?UTF-8?q?Local=20=E2=91=A1=20engine:=20Qwen-Image=202.1?= =?UTF-8?q?=20replaces=20Qwen-Image-Edit=202511=20+=20angles=20LoRA=20(0.1?= =?UTF-8?q?7.0)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - add qwen21_edit.json (core nodes only, 25 steps, CFG 1, ~2.4 MP); old graph moved to comfy_workflows/legacy/ with a README - settings LDS_QWEN21_MODEL / _TEXT_ENCODER / _VAE replace LDS_QWEN_EDIT_MODEL / _TEXT_ENCODER / _VAE / LDS_ANGLES_LORA - local prompt is now the same plain-English instruction as the cloud prompt, close-ups keep their setting - three-quarter fronts use camera-orbit wording; left shot names the image edge so it mirrors - local prompt never carries a negation (2.1 draws what it is told not to); prop exclusion is cloud-only - remove ShotStyle.local and the dead config.ENGINES - fix apply_wardrobe appending the outfit after the style sentence - ComfyUI engine: httpx errors fail one shot, not the run; run_prompt tolerates stalled /history polls - loading a pre-0.17 plan with prompts shows a warning - tests: engine, poll tolerance, wardrobe order, negation ban, stale-plan warning - docs, AGENTS.md and version updated --- .env.example | 7 +- AGENTS.md | 8 +- README.md | 5 +- app.py | 27 ++- docs/ARCHITECTURE.md | 82 ++++--- docs/comfyui-setup.md | 26 +-- studio/__init__.py | 2 +- studio/comfy_api.py | 32 ++- studio/comfy_workflows/legacy/README.md | 24 +++ .../qwen_edit_2511.json} | 0 studio/comfy_workflows/qwen21_edit.json | 10 + studio/config.py | 16 +- studio/engines/comfyui.py | 25 ++- studio/isolate.py | 6 +- studio/shot_style.py | 18 +- studio/shotplan.py | 203 +++++++++--------- tests/test_cli_preflight.py | 2 +- tests/test_comfy_api.py | 2 +- tests/test_comfy_diagnostics.py | 42 ++++ tests/test_comfy_engine.py | 87 ++++++++ tests/test_concept_plan.py | 49 ++--- tests/test_plan_io.py | 16 ++ tests/test_prop_exclusion.py | 17 +- tests/test_shot_style.py | 30 ++- tests/test_shotplan.py | 57 +++-- 25 files changed, 509 insertions(+), 284 deletions(-) create mode 100644 studio/comfy_workflows/legacy/README.md rename studio/comfy_workflows/{qwen_edit.json => legacy/qwen_edit_2511.json} (100%) create mode 100644 studio/comfy_workflows/qwen21_edit.json create mode 100644 tests/test_comfy_engine.py diff --git a/.env.example b/.env.example index 2a1285e..2169405 100644 --- a/.env.example +++ b/.env.example @@ -37,10 +37,9 @@ HF_TOKEN= # LDS_UPDATE_CHECK_ENABLED=true # checks GitHub releases on launch (no data sent); set false to disable # ComfyUI model filenames, if yours are named differently (see docs/comfyui-setup.md) -# LDS_QWEN_EDIT_MODEL=qwen_image_edit_2511_int8_convrot.safetensors -# LDS_ANGLES_LORA=qwen/Qwen-Image-Edit-2511-Multiple-Angles-LoRA.safetensors -# LDS_QWEN_TEXT_ENCODER=qwen_2.5_vl_7b_fp8_scaled.safetensors -# LDS_QWEN_VAE=qwen_image_vae.safetensors +# LDS_QWEN21_MODEL=qwen_image_2.1_int8_convrot.safetensors +# LDS_QWEN21_TEXT_ENCODER=qwen3vl_8b_int8_convrot.safetensors +# LDS_QWEN21_VAE=qwen_image_2.1_vae_bf16.safetensors # LDS_UPSCALE_MODEL=4xNomosWebPhoto_RealPLKSR.safetensors # LDS_DEJPG_MODEL=1xDeJPG_OmniSR.pth # LDS_SAM3_CHECKPOINT=sam3.1_multiplex_fp16.safetensors diff --git a/AGENTS.md b/AGENTS.md index 82eb542..1a3a60d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -6,10 +6,10 @@ Turn a character, style, or concept into a ready-to-train LoRA dataset. One refe ## Current state -_Last verified: 2026-08-23_ +_Last verified: 2026-10-03_ -- **Status:** in active development, released at v0.16.0 (git tag `v0.16.0`). CI green. **Every version bump must ship a GitHub Release** or the in-app update check never fires. The project was renamed twice (lora-dataset-studio → lora-distillery → dataset-deviser); older references under the previous names are stale. -- **Works:** all five stages end to end (preprocess → generate & curate → caption → export → train config); the three dataset types (character, style, concept) with their own shot plans and caption framing; local and cloud paths for every stage; gallery curation with selection carried forward between stages, including shift-click range select; advisory dedupe, quality flags and caption lint; trainer configs for ai-toolkit (incl. SDXL), kohya sd-scripts and musubi-tuner; opt-in private Hugging Face publish. Since 0.15.0: per-image fault isolation in ① (one bad source never ends the batch), a cooperative ⏹ Stop across ①/②/③ with documented resume paths, translated ComfyUI failures (missing node, bad input, unreachable server) and an in-app 🩺 setup check. Since 0.15.1: a ② **shot style** (default = keep the reference's own medium; eight presets + custom text) threaded into both prompt builders, the CLI, ④ metadata and ⑤'s sample prompt, a final-prompt preview, and a prev/next/save-and-next caption editor with the image on screen. Since 0.16.0: **one name box and one trigger box in the header** that ②/③/④/⑤ all read (they used to be per-tab and went stale), ④ stating the pair it stamped, an opt-in ④ **hand-off to [Idiot LoRa Builder](https://github.com/Fablestarexpanse/Idiot-Lora-Builder)** (writes its `.lora-studio/ratings.json` so its grid opens pre-triaged — nothing is launched), a ⑤ advisory when the dataset can't fill the training resolution, every bundled ComfyUI model filename overridable from `.env`, and a `cli build` preflight for the local engine. +- **Status:** in active development, released at v0.16.0 (git tag `v0.16.0`); 0.17.0 (local ② generation moved to Qwen-Image 2.1) is committed but not yet tagged or released. CI green at 0.16.0. **Every version bump must ship a GitHub Release** or the in-app update check never fires. The project was renamed twice (lora-dataset-studio → lora-distillery → dataset-deviser); older references under the previous names are stale. +- **Works:** all five stages end to end (preprocess → generate & curate → caption → export → train config); the three dataset types (character, style, concept) with their own shot plans and caption framing; local and cloud paths for every stage; gallery curation with selection carried forward between stages, including shift-click range select; advisory dedupe, quality flags and caption lint; trainer configs for ai-toolkit (incl. SDXL), kohya sd-scripts and musubi-tuner; opt-in private Hugging Face publish. Since 0.15.0: per-image fault isolation in ① (one bad source never ends the batch), a cooperative ⏹ Stop across ①/②/③ with documented resume paths, translated ComfyUI failures (missing node, bad input, unreachable server) and an in-app 🩺 setup check. Since 0.15.1: a ② **shot style** (default = keep the reference's own medium; eight presets + custom text) threaded into both prompt builders, the CLI, ④ metadata and ⑤'s sample prompt, a final-prompt preview, and a prev/next/save-and-next caption editor with the image on screen. Since 0.16.0: **one name box and one trigger box in the header** that ②/③/④/⑤ all read (they used to be per-tab and went stale), ④ stating the pair it stamped, an opt-in ④ **hand-off to [Idiot LoRa Builder](https://github.com/Fablestarexpanse/Idiot-Lora-Builder)** (writes its `.lora-studio/ratings.json` so its grid opens pre-triaged — nothing is launched), a ⑤ advisory when the dataset can't fill the training resolution, every bundled ComfyUI model filename overridable from `.env`, and a `cli build` preflight for the local engine. Since 0.17.0: the local ② engine is **Qwen-Image 2.1** (plain-English prompts shared with Gemini, ~26 s/shot, `LDS_QWEN21_*` filenames) instead of Qwen-Image-Edit 2511 + the Multiple-Angles LoRA, whose graph stays in `studio/comfy_workflows/legacy/`; a dead ComfyUI mid-batch now fails one shot, not the run. - **In progress:** nothing half-built — 0.16.0 closed out the hand-off, cross-tab-identity and config-hardening work. `docs/ARCHITECTURE.md` carries the "Future ideas" and "Deferred" sections that hold the real backlog. - **Known gaps / next steps:** pick up from `docs/ARCHITECTURE.md` → "Future ideas" / "Deferred", and read its *Maintainer principles* before changing anything; `gradio` is pinned `<6` because of a real stuck-loading-overlay regression — unpinning needs that verified fixed upstream; training is never launched for you and is not planned to be; the default export `output_root` writes into the repo root, so a smoke-test export leaves an untracked dataset folder to delete before committing. - **Deep docs:** `docs/ARCHITECTURE.md` (module map, stage and backend detail, gotchas, backlog — the deep reference), `docs/comfyui-setup.md`. Worklogs under `docs/` are gitignored and local-only by design. @@ -73,7 +73,7 @@ python cli.py keys --setup - ⏹ Stop is cooperative and checked *between* items; every stage calls `JOB.start()` first, and the Stop button must stay `queue=False`. - Bundled ComfyUI workflows are core-nodes-only, so an unknown node means an out-of-date ComfyUI — `comfy_api` preflights node classes and translates rejection bodies. - `tests/conftest.py` autouse fixture repoints the output roots at tmp; without it the suite writes real folders into `runs/`. -- ② prompts never hard-code a medium — it comes from `studio/shot_style.py` (default: keep the reference's). Never write "photorealistic"/"hyperrealistic"; a test bans them from every generated prompt. Angle shots stay pure `` grammar. +- ② prompts never hard-code a medium — it comes from `studio/shot_style.py` (default: keep the reference's). Never write "photorealistic"/"hyperrealistic"; a test bans them from every generated prompt. The local prompt never carries a negation ("without", "do not", a named prop): Qwen-Image 2.1 draws what it is told not to; a test bans it. - Re-value a Gradio dropdown with `gr.update(value=…)`, not `gr.Dropdown(value=…)` — the constructor form re-validates against the ORIGINAL `choices` and can silently drop the value. - The dataset **name and trigger are single header components** (`project_name`/`project_trigger`); every stage is wired straight to them. Never reintroduce a per-tab copy, and never mirror components per keystroke — Gradio's unqueued `.input` replies land out of order and settle on a prefix of what was typed. - Every model filename in a bundled ComfyUI template must be reachable from `comfy_api._MODEL_INPUTS` (plus a setting in `config.py` and a line in `.env.example`) — a test enforces it. Un-mapped means the user can't fix it in `.env` and `doctor` can't check it. diff --git a/README.md b/README.md index cdcd52a..afd92d1 100644 --- a/README.md +++ b/README.md @@ -45,7 +45,7 @@ Run them in order (each step auto-fills the next) or jump straight to the one yo |---|---|---| | ① Restore / upscale | ComfyUI models, or basic Lanczos | — | | ① Subject isolation | **Built-in SAM3** (no ComfyUI) or ComfyUI SAM3 | — | -| ② Generate shots *(character + concept)* | ComfyUI: Qwen Image Edit 2511 + Multiple-Angles LoRA | Gemini (Nano Banana) | +| ② Generate shots *(character + concept)* | ComfyUI: Qwen-Image 2.1 *(Qwen Research License)* | Gemini (Nano Banana) | | ③ Caption | Qwen3-VL-8B, JoyCaption, NSFW finetune, **WD + e621 taggers**, LM Studio / Ollama / any OpenAI endpoint | Gemini Flash, Groq free tier | | ④ Export | always local (+ optional **.zip** and **Hugging Face** publish) | — | | ⑤ Train config | ai-toolkit (incl. SDXL) / **kohya sd-scripts** / musubi-tuner | — | @@ -213,8 +213,7 @@ comic, digital illustration, traditional painting, 3D render, ink line art, or * where you describe the medium yourself (*"a 1970s screen-printed poster, halftone dots"*). Changing it rebuilds the shot plan, so hand-edited prompt cells are replaced — edit prompts -after you've settled on a style. Camera-angle shots deliberately keep their bare -turnaround grammar, which is what the angles LoRA was trained on. Use **👁 Preview final +after you've settled on a style. Use **👁 Preview final prompt** to see exactly what a row will send, including outfit and prop-exclusion. The style is recorded in the exported `metadata.json` and picked up by ⑤'s sample prompt. diff --git a/app.py b/app.py index fbb93e6..d3354a8 100644 --- a/app.py +++ b/app.py @@ -54,7 +54,7 @@ ENGINE_CHOICES = [ ("Cloud — Gemini image model (best identity fidelity, SFW only)", "gemini"), - ("Local — ComfyUI Qwen Image Edit 2511 (free, private, uncensored)", "comfyui"), + ("Local — ComfyUI Qwen Image 2.1 (free, private, uncensored)", "comfyui"), ] CLOUD_MODEL_CHOICES = [(f"{m} (~${p:.3f}/img est.)", m) for m, p in CLOUD_IMAGE_PRICES.items()] CAPTIONER_CHOICES = [(c.label, c.key) for c in CAPTIONERS] @@ -303,7 +303,7 @@ def preview_final_prompt(plan_df: pd.DataFrame, engine: str, exclude_props: bool if exclude_props: shot = apply_prop_exclusion(shot) field = "local_prompt" if engine == "comfyui" else "cloud_prompt" - which = "Local (ComfyUI / Qwen-Edit)" if engine == "comfyui" else "Cloud (Gemini)" + which = "Local (ComfyUI / Qwen-Image 2.1)" if engine == "comfyui" else "Cloud (Gemini)" return (f"**{which} prompt for `{shot.id}`** — exactly what the engine receives, " f"after outfit and prop-exclusion are folded in:\n\n```\n" f"{getattr(shot, field)}\n```") @@ -1343,7 +1343,15 @@ def do_load_plan(plan_name: str): raise gr.Error(f"Couldn't read the plan at {path}: {e}. Check the YAML — " f"each shot needs at least an `id`, `kind`, `local_prompt` " f"and `cloud_prompt`.") from e - return _shots_to_df(shots), f"✅ Loaded {len(shots)} shots from {path}" + note = f"✅ Loaded {len(shots)} shots from {path}" + # Plans saved before 0.17.0 carry the old engine's `` LoRA grammar, which + # Qwen-Image 2.1 reads as literal text and renders badly. + stale = sum("" in s.local_prompt for s in shots) + if stale: + note += (f" ⚠️ {stale} shot(s) still use the old Qwen-Image-Edit `` prompts, " + f"which the local Qwen-Image 2.1 engine doesn't understand — rebuild the " + f"plan from the dataset type / shot style, or rewrite those local prompts.") + return _shots_to_df(shots), note def estimate_cost(engine: str, cloud_model: str, df: pd.DataFrame) -> str: @@ -1795,15 +1803,18 @@ def _check_for_update(): gen_exclude_props = gr.Checkbox( value=True, label="Exclude props/accessories from the reference", - info="Asks the generator to drop bags, held objects and " - "accessories carried in your reference, so they don't get " - "baked into every dataset image. Isolating the source in ① " - "is the more reliable fix. Character-oriented wording — off " + info="Cloud engine only: asks Gemini to drop bags, held objects " + "and accessories carried in your reference, so they don't " + "get baked into every dataset image. The local engine " + "ignores it (naming a prop, even to forbid it, makes " + "Qwen draw it) — isolate the source in ① instead, the more " + "reliable fix either way. Character-oriented wording — off " "by default for Concept datasets.") gen_isolate = gr.Checkbox(value=False, label="Isolate generated angle shots (white background)", info="Cut generated angle shots onto white too " - "(helps the angles LoRA on back views).") + "(replaces each angle shot's setting " + "with a plain white background).") gen_iso_backend = gr.Dropdown(ISOLATION_CHOICES, value=settings.isolation_backend, label="Isolation backend", diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 549cb1e..b14c47c 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -1,6 +1,6 @@ # Architecture -Version: 0.15.1 +Version: 0.17.0 ``` app.py Gradio UI — thin wiring over the stage functions (5 tabs). @@ -102,8 +102,8 @@ studio/ shot_style.py The medium ②'s prompts ask for: SHOT_STYLES (match | photographic | anime | comic | illustration | painting | render3d | lineart | custom) + resolve(). Pure data + - one resolver; each style carries a terse `local` clause - (Qwen-Edit), a full `cloud` sentence (Gemini) and a + one resolver; each style carries a full `cloud` sentence + (both engines' prompts end with it) and a `sample_lead` for ⑤. Default `match` PRESERVES the reference's own medium — see the three gotchas shotplan.py Shot plans + Shot model. default_plan() = Character (curated 24: @@ -148,8 +148,10 @@ studio/ (list_image_models / list_caption_models). Retries transient statuses with backoff (GENERATE_RETRIES / GENERATE_BACKOFF_S) and reports failures through friendly_api_error() - comfyui.py Local engine (Qwen Image Edit 2511 + Multiple-Angles LoRA) - comfy_workflows/*.json API-format ComfyUI graphs (restore, isolate ×2, qwen edit) + comfyui.py Local engine (Qwen-Image 2.1, one reference per shot) + comfy_workflows/*.json API-format ComfyUI graphs (restore, isolate ×2, qwen21 edit) + comfy_workflows/legacy/ The pre-0.17 Qwen-Image-Edit-2511 + Multiple-Angles graph, kept + for anyone who wants it; nothing loads it (see its README) ``` ## Testing & CI @@ -409,11 +411,11 @@ quietly gets someone else's value. `photographic` preset is built from camera vocabulary (lens, depth of field, sensor, texture) instead. A test sweeps every style × both plans × both prompt fields to keep the banned wording out. -- **Angle shots get no style clause.** `kind="angle"` local prompts stay pure `` - grammar for the same measured reason `apply_prop_exclusion` skips them (the - Multiple-Angles LoRA is trained on clean splat renders and degrades when prose is - appended). They carried no medium claim before either, so nothing is lost — Qwen-Edit - follows the reference image's medium on its own. +- **Local and cloud prompts are the same sentence.** Qwen-Image 2.1 follows the plain-English + instruction Gemini gets (`shotplan._instruction`), and kept identity *and medium* (photo, + anime, 3D creature) far better with it than 2511 did with terse tags. The one difference: + close-ups keep their setting locally, because without it every close-up came back on the + reference's own backdrop. Every kind, angles included, ends with the style sentence. - **A stage that loops over user files must isolate each item.** ① was the last stage without this: `preprocess_sources` let one `IsolationError` (SAM3 finding no subject in image 3 of 20) unwind the whole batch, and `app.do_preprocess` turned it into a @@ -582,10 +584,26 @@ quietly gets someone else's value. - **Occlusion holes cannot be masked away.** Where a prop physically covers the body, the subject mask has a real hole and the composite renders it white. Only inpainting fixes that; no amount of threshold/dilation tuning will. -- **The Multiple-Angles LoRA is trained on clean splat renders** — isolating the subject - onto white dramatically improves its output, especially direct back views. The LoRA is - trigger-based (`` grammar), so its strength is 0.9 for `kind="angle"` shots and - zeroed for pose/emotion shots. +- **Qwen-Image 2.1 draws what a prompt names, even negated.** "Do not include any backpacks…" + put a backpack on the character in 2 of 8 creature shots (and cost one its tail), so the local + prompt never carries a negation — `apply_prop_exclusion` edits the cloud prompt only, and a + test bans "without"/"do not"/"backpack" from every local prompt. Isolate the source in ① to + keep a prop out locally. (One reference set, one seed — a measurement, not a law.) +- **Three-quarter fronts need camera-orbit wording.** "front-right quarter view" came back as a + plain front view; "with the camera moved 45 degrees around the character to its right, so the + character is seen in a three-quarter view showing one side of the face and body" rotates it. + "turned 45 degrees toward the right side of the image" overshoots to a full profile. The + *left* shot cannot just say "to its left" — that came back facing image-right too (found in the + first full live run); "in a three-quarter view, with the character's body and face turned toward + the left edge of the image" faces image-left. Side, back, low and high views work as plain + descriptions. The back three-quarter rows (and the concept plan's wording) are the least-tested + phrases — check them first if a set looks off. +- **Qwen-Image 2.1 numbers (RTX 5090, 0.17.0 A/B).** ~26 s per shot at ~2.4 MP / 25 steps / CFG 1, + against ~65 s for 2511 at ~1 MP / 40 steps / CFG 4, and ~7 GB of weights against ~20 GB. The + graph is core nodes only (`TextEncodeQwenImage21`, whose `latent` output sizes the render from + the reference's aspect ratio). Hard side-light once darkened light hair; soften the setting + phrase if a set shows it. The model is under the Qwen Research License (non-commercial); + what users do with their outputs is theirs to decide, the README just names the license. - **VRAM choreography:** before loading a ~17 GB local captioner, the pipeline asks ComfyUI to `/free` its models (best effort) and unloads the in-process SAM3. - **ComfyUI queue guard:** if ComfyUI already has >10 pending jobs, the client fails @@ -607,12 +625,10 @@ quietly gets someone else's value. - **The Concept plan reuses the character machinery, not a parallel one.** `concept_plan()` emits the same `Shot` model with `emotion`/`outfit` left empty, so the ② dataframe, `plan_io` YAML round-trip, `apply_wardrobe` (a no-op on an empty outfit) and `generate_shots` need no branch. - Its angle shots reuse the **attested** `` grammar from the character plan (view + camera - height + shot size) — the one addition, `front view high-angle`, mirrors the existing low-angle - shot. Do not invent grammar the Multiple-Angles LoRA was never trained on (a "top view" would - silently render a plain front view); a test pins the vocabulary. The non-angle kinds are + Its angle shots use the same plain-English phrases as the character plan (the three-quarter + fronts with camera-orbit wording, see above) plus `high-angle`. The non-angle kinds are `framing` (scale: detail → wide) and `context` (where it sits / how it's used, including a hand - for scale), which keep LoRA strength at 0 like character pose shots. + for scale). - **Character-only ② controls are hidden, not just ignored, for Concept.** Wardrobe (`OUTFIT_SHOT_KINDS` includes `angle`, so the randomizer *would* dress an object's turnaround) and prop exclusion (its clause literally says "show only the character and the clothing worn on @@ -801,14 +817,19 @@ quietly gets someone else's value. - **`zip_dataset` arcnames are derived from the folder, never absolute.** Entries are stored under `/…` and the archive is non-clobbering (`-2.zip`, …) — no path-traversal surface because we only ever add files found *inside* the packaged dataset dir. +- **ComfyUI stops answering HTTP while it swaps models, and the job is fine.** The first full live + run lost 2 of 24 shots to `httpx.ReadTimeout` on a `/history` poll while ComfyUI was loading + weights; the retry then queued the same job twice. `comfy_api.run_prompt` now tolerates + `MAX_STALLED_POLLS` consecutive failed polls before giving up, and `ComfyUIEngine.generate` turns + any `httpx` error (upload, poll, fetch) into a per-shot `GenerationError`, never a crashed run. - **ComfyUI caches model combo lists**; a freshly downloaded model file may need a ComfyUI restart before the bundled workflows validate. - **Model filenames are configuration — all of them.** The bundled workflow JSONs are patched at - load time from settings (`LDS_QWEN_EDIT_MODEL` etc.), so users don't edit JSON to match their + load time from settings (`LDS_QWEN21_MODEL` etc.), so users don't edit JSON to match their filenames. `comfy_api._MODEL_INPUTS` is keyed by the loader's *input name*, which is unique per class across every bundled template; `UpscaleModelLoader` is the one exception and is mapped positionally through `_UPSCALE_SETTINGS`. Anything **not** in that map is a filename the user - cannot fix from `.env` and that `doctor` cannot check — `qwen_edit.json`'s `clip_name` and + cannot fix from `.env` and that `doctor` cannot check — the old `qwen_edit.json`'s `clip_name` and `vae_name` were exactly that for four releases, and the symptom was ComfyUI's own opaque *"Value not in list: … (list of length 10574)"*. `tests/test_comfy_api.py` now fails if any model-file-looking input in any template is unreachable from the map, or if a mapping names a @@ -945,8 +966,8 @@ grouped by stage. Milestone versions are noted only where they explain a design crop, resize to target long-side. Optional alpha (RGBA) cutout output (builtin backend only) for external compositing workflows. - **② Generate & curate** — curated 24-shot Character plan (angles/poses/emotions/settings) or - 18-shot Concept plan (turnaround/framing/context), Qwen-Image-Edit 2511 + Multiple-Angles LoRA - (local) or Gemini (cloud), chained rear views, prop exclusion, wardrobe randomizer, per-shot + 18-shot Concept plan (turnaround/framing/context), Qwen-Image 2.1 (local, 0.17.0 — was Qwen-Image-Edit + 2511 + the Multiple-Angles LoRA, graph kept in `comfy_workflows/legacy/`) or Gemini (cloud), chained rear views, prop exclusion, wardrobe randomizer, per-shot outfit column, **shot style** (match the reference's medium by default, or convert the set to one of eight presets / your own description), final-prompt preview, save/load YAML plans, sharpness + exposure/contrast advisories, per-model cost estimate. Character + Concept; @@ -1072,6 +1093,10 @@ ordered by benefit-to-cost. the rendered prompts, not by a generation run — the shot mix (10 angles / 4 framing / 4 context) and the `front view high-angle` addition are reasoned, not measured. If object turnarounds come back weak, retune the mix or drop the shots the LoRA can't do, and record the finding here. +- **Multi-reference generation (②, local).** Qwen-Image 2.1 accepts up to 10 references + (`TextEncodeQwenImage21`'s `images.image_N`); the engine still sends one per shot. Several + references of the same character could hold identity better on hard angles. Untested — needs + its own A/B before it earns a control. - **Face-similarity identity guard (② curate).** Flag generated shots that drift from the reference's identity. Genuinely useful for character LoRAs but needs a face-recognition dependency (InsightFace + onnxruntime); could be an optional extra like the gated SAM3 download. @@ -1095,7 +1120,8 @@ Both items below shipped in 0.14.1. cap does lift**, two things move with it: pass `head=` to `launch()` instead of the `Blocks` constructor (and drop the warning filter around it), and re-verify `_PICKER_SCRIPT` still finds `.thumbnail-item` — a Gallery DOM change would break click-to-select silently, which - is what `tests/test_selection_flow.py`'s id-contract tests guard against on our side. + is what `tests/test_selection_flow.py`'s id-contract tests guard against on our side. A third: + the suite already warns that a tuple `row_count` is removed in Gradio 6. ## Repo rename: lora-dataset-studio → lora-distillery (0.12.2) @@ -1209,9 +1235,9 @@ Considered and deliberately **not** pursued, with the reason each stays out. sheet prompt returned a grotesque mashup. Klein 9B img2img with core nodes reached "recognisable but not dataset-quality"; closing the gap needs third-party identity packs (IdentityFeatureTransferFinal / Multi ReferenceLatent), which the core-nodes-only - rule for bundled workflows rules out. **Qwen Image Edit 2511 + the Multiple-Angles LoRA - remains the only proven approach**; don't re-evaluate these two without a new capability - to test. (Full test log kept locally, not in the repo — it is machine-specific.) + rule for bundled workflows rules out. **Qwen-Image 2.1 replaced 2511 + the Multiple-Angles LoRA + in 0.17.0** after a live A/B (faster, lighter, better identity and medium on poses/emotions); + don't re-evaluate Krea2/Klein without a new capability to test. (Full test log kept locally, not in the repo — it is machine-specific.) - **In-app / cloud training launch, Test Studio / checkpoint ranking, Merge Lab.** These cross the "never launch training" line this tool holds on purpose — launching is fragile across trainer @@ -1221,7 +1247,7 @@ Considered and deliberately **not** pursued, with the reason each stays out. away. (For sourcing, the README points to the author's separate video-frame extractor tool.) - **Inpainting occluded regions.** Where a prop physically covers the body, isolation leaves a real hole; the honest fix needs a generative edit pass per source. The app has the machinery - (Qwen-Image-Edit) — revisit only if the white notch bites users. + (Qwen-Image 2.1 edit) — revisit only if the white notch bites users. - **Live cloud pricing.** Google publishes no pricing API; only scraping the pricing page could beat the build-time table, and it would break silently. Estimates are labelled as such. - **Automated aesthetic scoring beyond sharpness.** Needs a heavy scoring model; the cheap Laplacian diff --git a/docs/comfyui-setup.md b/docs/comfyui-setup.md index 100a3a0..2cce813 100644 --- a/docs/comfyui-setup.md +++ b/docs/comfyui-setup.md @@ -3,21 +3,20 @@ ComfyUI is **not required**: cloud generation, built-in SAM3 isolation, local captioning, and export all work without it. Install it only if you want: -- **Fully-local image generation** (free, private, uncensored) — Qwen Image Edit 2511 - with the fal Multiple-Angles LoRA, or +- **Fully-local image generation** (free, private, uncensored) — Qwen-Image 2.1, or - **Model-based photo restoration** (DeJPG + photo upscaler) for degraded sources, or - To run **SAM3 isolation inside ComfyUI** instead of in-process. ## 1. Install ComfyUI Follow (or use the desktop installer / -ComfyUI portable). A recent build is required — the SAM3 nodes (`SAM3_Detect`) are -part of ComfyUI core in current releases. Start it on the default port; if yours +ComfyUI portable). A recent build is required — the SAM3 nodes (`SAM3_Detect`) and the +Qwen-Image 2.1 nodes (`TextEncodeQwenImage21`) are part of ComfyUI core in current releases. Start it on the default port; if yours differs, set `LDS_COMFY_URL` in `.env`. **No custom nodes are needed.** Every bundled workflow uses only core ComfyUI nodes. -If a workflow reports a missing node, you're on a build too old for core SAM3 — update -ComfyUI rather than installing a node pack. +If a workflow reports a missing node, your build is too old — update ComfyUI rather +than installing a node pack. ## 2. Download models @@ -26,17 +25,13 @@ if yours differ — see `.env.example`): | Purpose | File (default name) | Goes in | Source | |---|---|---|---| -| Edit model | `qwen_image_edit_2511_int8_convrot.safetensors` | `models/unet/` (a.k.a. `diffusion_models/`) | any Qwen-Image-Edit-2511 checkpoint packaged for ComfyUI — e.g. the [Comfy-Org repackages](https://huggingface.co/Comfy-Org) or a quantized variant that fits your VRAM; set `LDS_QWEN_EDIT_MODEL` to its filename | -| Qwen text encoder | `qwen_2.5_vl_7b_fp8_scaled.safetensors` | `models/text_encoders/` | same source as the edit model (follow its README); set `LDS_QWEN_TEXT_ENCODER` if yours is named differently | -| Qwen VAE | `qwen_image_vae.safetensors` | `models/vae/` | same source as the edit model; set `LDS_QWEN_VAE` if yours is named differently | -| Multi-angle LoRA | `qwen/Qwen-Image-Edit-2511-Multiple-Angles-LoRA.safetensors` | `models/loras/qwen/` | [fal/Qwen-Image-Edit-2511-Multiple-Angles-LoRA](https://huggingface.co/fal/Qwen-Image-Edit-2511-Multiple-Angles-LoRA) (Apache-2.0) | +| Qwen-Image 2.1 model | `qwen_image_2.1_int8_convrot.safetensors` | `models/diffusion_models/` (a.k.a. `unet/`) | [Comfy-Org/Qwen-Image-2.1](https://huggingface.co/Comfy-Org/Qwen-Image-2.1) (`diffusion_models/`) — a bf16 file also exists there; set `LDS_QWEN21_MODEL` to whichever filename you have. **Qwen Research License (non-commercial).** | +| Qwen3-VL text encoder | `qwen3vl_8b_int8_convrot.safetensors` | `models/text_encoders/` | same repo (`text_encoders/`); set `LDS_QWEN21_TEXT_ENCODER` if yours is named differently | +| Qwen-Image 2.1 VAE | `qwen_image_2.1_vae_bf16.safetensors` | `models/vae/` | same repo (`vae/`); set `LDS_QWEN21_VAE` if yours is named differently | | SAM3 (ComfyUI backend only) | `sam3.1_multiplex_fp16.safetensors` | `models/checkpoints/` | [Comfy-Org/sam3.1](https://huggingface.co/Comfy-Org/sam3.1) | | Restoration: JPEG cleanup | `1xDeJPG_OmniSR.pth` | `models/upscale_models/` | [OpenModelDB](https://openmodeldb.info/models/1x-DeJPG-OmniSR) | | Restoration: photo upscale | `4xNomosWebPhoto_RealPLKSR.safetensors` | `models/upscale_models/` | [OpenModelDB](https://openmodeldb.info/models/4x-NomosWebPhoto-RealPLKSR) | -> The angle LoRA uses fal's `` camera grammar (` [azimuth] [elevation] -> [distance]`, 96 poses); the default shot plan already speaks it. - ## 3. How the app talks to ComfyUI `studio/comfy_workflows/*.json` are API-format graphs submitted over ComfyUI's HTTP @@ -46,8 +41,9 @@ never a JSON edit. Workflows used: -- `qwen_edit.json` — generation (angles via the LoRA at 0.9 strength, pose/scene edits - with the LoRA zeroed) +- `qwen21_edit.json` — generation: one reference image plus one plain-English instruction + per shot (~26 s a shot on an RTX 5090). The old Qwen-Image-Edit-2511 + Multiple-Angles + graph is kept, unwired, in `studio/comfy_workflows/legacy/` (see its README) - `restore_upscale.json` — DeJPG → 4× photo upscale - `isolate_subject.json` / `isolate_exclude.json` — SAM3 cutout onto white, optionally removing held props via a second segmentation diff --git a/studio/__init__.py b/studio/__init__.py index 35f2a37..4c4b1aa 100644 --- a/studio/__init__.py +++ b/studio/__init__.py @@ -1,3 +1,3 @@ """Dataset Deviser: turn a character, style, or concept into a ready-to-train LoRA dataset.""" -__version__ = "0.16.0" +__version__ = "0.17.0" diff --git a/studio/comfy_api.py b/studio/comfy_api.py index fff6e80..6257199 100644 --- a/studio/comfy_api.py +++ b/studio/comfy_api.py @@ -18,16 +18,20 @@ class ComfyError(Exception): pass +# Consecutive failed /history polls before a running job is given up on (each poll +# waits up to 30 s, so a stalled server gets minutes, not seconds). +MAX_STALLED_POLLS = 5 + + # Model filenames inside the bundled templates, remapped to whatever the user -# configured in .env (LDS_QWEN_EDIT_MODEL etc.) so renamed files just work. +# configured in .env (LDS_QWEN21_MODEL etc.) so renamed files just work. # Keyed by the *input name*, which is unique per loader class across every # bundled template — adding an entry here also extends `doctor`'s ComfyUI-models # check, which validates each configured name against the server's own list. _MODEL_INPUTS = { - "unet_name": "qwen_edit_model", - "lora_name": "angles_lora", - "clip_name": "qwen_text_encoder", - "vae_name": "qwen_vae", + "unet_name": "qwen21_model", + "clip_name": "qwen21_text_encoder", + "vae_name": "qwen21_vae", "ckpt_name": "sam3_checkpoint", } _UPSCALE_SETTINGS = ("dejpg_model", "upscale_model") # in template node order @@ -261,8 +265,24 @@ def run_prompt(graph: dict, timeout: float = 600.0, front: bool = False) -> list prompt_id = r.json()["prompt_id"] deadline = time.monotonic() + timeout + stalled = 0 while time.monotonic() < deadline: - h = httpx.get(f"{settings.comfy_url}/history/{prompt_id}", timeout=30).json() + # ComfyUI stops answering HTTP for tens of seconds while it swaps models in + # and out of VRAM, yet the job is fine and finishes. One slow poll must not + # fail the shot (and the retry would queue the same job a second time) — but + # a server that stays silent for several polls in a row is genuinely gone. + try: + h = httpx.get(f"{settings.comfy_url}/history/{prompt_id}", timeout=30).json() + except httpx.HTTPError as e: + stalled += 1 + if stalled >= MAX_STALLED_POLLS: + raise ComfyError( + f"ComfyUI stopped answering while running prompt {prompt_id} " + f"({type(e).__name__}) — it may have crashed or run out of memory; " + f"check the ComfyUI window.") from e + time.sleep(1.5) + continue + stalled = 0 if prompt_id in h: entry = h[prompt_id] status = entry.get("status", {}) diff --git a/studio/comfy_workflows/legacy/README.md b/studio/comfy_workflows/legacy/README.md new file mode 100644 index 0000000..303960d --- /dev/null +++ b/studio/comfy_workflows/legacy/README.md @@ -0,0 +1,24 @@ +# Legacy workflows + +Kept for anyone who prefers them. **Nothing in the app loads this folder**, and the +template tests and `doctor` only scan `studio/comfy_workflows/*.json`. + +## `qwen_edit_2511.json` — Qwen-Image-Edit-2511 + Multiple-Angles LoRA + +The local ② generator up to v0.16.0 (Apache-2.0 models; the current engine, +Qwen-Image 2.1, is under the Qwen Research License). API-format graph with two +placeholders and hard-coded model filenames: + +- `__SOURCE__` — the uploaded reference's filename, in node 1 (`LoadImage`) +- `__PROMPT__` — the prompt, in node 9 (`TextEncodeQwenImageEditPlus`) + +To use it, edit the model filenames (UNET, text encoder, VAE, LoRA) to match your +ComfyUI, fill in the placeholders and POST `{"prompt": }` to ComfyUI's +`/prompt` — or rebuild it in the ComfyUI editor from the node list. + +Prompts for it were written in the LoRA's `` grammar, e.g. +` front-right quarter view eye-level shot medium shot` (LoRA strength 0.9 for +angle shots, 0 for pose/emotion), and plain English for everything else. Those +prompt builders live in the `v0.16.0` git tag (`studio/shotplan.py`, +`studio/engines/comfyui.py`); the LoRA is +[fal/Qwen-Image-Edit-2511-Multiple-Angles-LoRA](https://huggingface.co/fal/Qwen-Image-Edit-2511-Multiple-Angles-LoRA). diff --git a/studio/comfy_workflows/qwen_edit.json b/studio/comfy_workflows/legacy/qwen_edit_2511.json similarity index 100% rename from studio/comfy_workflows/qwen_edit.json rename to studio/comfy_workflows/legacy/qwen_edit_2511.json diff --git a/studio/comfy_workflows/qwen21_edit.json b/studio/comfy_workflows/qwen21_edit.json new file mode 100644 index 0000000..4531221 --- /dev/null +++ b/studio/comfy_workflows/qwen21_edit.json @@ -0,0 +1,10 @@ +{ + "1": {"class_type": "LoadImage", "inputs": {"image": "__SOURCE__"}}, + "2": {"class_type": "UNETLoader", "inputs": {"unet_name": "qwen_image_2.1_int8_convrot.safetensors", "weight_dtype": "default"}}, + "3": {"class_type": "CLIPLoader", "inputs": {"clip_name": "qwen3vl_8b_int8_convrot.safetensors", "type": "qwen_image", "device": "default"}}, + "4": {"class_type": "VAELoader", "inputs": {"vae_name": "qwen_image_2.1_vae_bf16.safetensors"}}, + "5": {"class_type": "TextEncodeQwenImage21", "inputs": {"prompt": "__PROMPT__", "negative_prompt": "", "resolution": 1536, "clip": ["3", 0], "vae": ["4", 0], "images.image_1": ["1", 0]}}, + "6": {"class_type": "KSampler", "inputs": {"seed": 0, "steps": 25, "cfg": 1, "sampler_name": "euler", "scheduler": "simple", "denoise": 1, "model": ["2", 0], "positive": ["5", 0], "negative": ["5", 1], "latent_image": ["5", 2]}}, + "7": {"class_type": "VAEDecode", "inputs": {"samples": ["6", 0], "vae": ["4", 0]}}, + "8": {"class_type": "SaveImage", "inputs": {"filename_prefix": "LDS", "images": ["7", 0]}} +} diff --git a/studio/config.py b/studio/config.py index 94356f0..1a4bf1b 100644 --- a/studio/config.py +++ b/studio/config.py @@ -351,11 +351,6 @@ def _compose_prompt(self, style: str, dataset_type: str, sparse: bool) -> str: CAPTIONERS_BY_KEY = {c.key: c for c in CAPTIONERS} -ENGINES = { - "gemini": "Cloud - Gemini image model (best identity fidelity, SFW only)", - "comfyui": "Local - ComfyUI Qwen Image Edit 2511 (free, private, uncensored)", -} - # Known Gemini image models -> USD per standard-resolution (1K) image. # ESTIMATES captured at build time — actual costs are billed by Google to the # user's own API key; the UI can live-pull the current model list. @@ -493,14 +488,15 @@ class Settings(BaseSettings): # ComfyUI model filenames used by the optional workflow templates # (relative to your ComfyUI models folders — see docs/comfyui-setup.md) - qwen_edit_model: str = "qwen_image_edit_2511_int8_convrot.safetensors" - angles_lora: str = "qwen/Qwen-Image-Edit-2511-Multiple-Angles-LoRA.safetensors" - # The text encoder and VAE the Qwen Edit graph loads. Overridable for the + qwen21_model: str = "qwen_image_2.1_int8_convrot.safetensors" + # The text encoder and VAE the Qwen-Image 2.1 graph loads. Overridable for the # same reason as the model above: everyone downloads these from a different # mirror and they arrive under different names. Left hard-coded, a rename # surfaced as ComfyUI's own "Value not in list: ... (list of length 10574)". - qwen_text_encoder: str = "qwen_2.5_vl_7b_fp8_scaled.safetensors" - qwen_vae: str = "qwen_image_vae.safetensors" + # Named qwen21_* (not the old qwen_*) on purpose: a stale LDS_QWEN_TEXT_ENCODER + # pointing at the Qwen2.5-VL file would otherwise be fed to the 2.1 graph. + qwen21_text_encoder: str = "qwen3vl_8b_int8_convrot.safetensors" + qwen21_vae: str = "qwen_image_2.1_vae_bf16.safetensors" upscale_model: str = "4xNomosWebPhoto_RealPLKSR.safetensors" dejpg_model: str = "1xDeJPG_OmniSR.pth" sam3_checkpoint: str = "sam3.1_multiplex_fp16.safetensors" diff --git a/studio/engines/comfyui.py b/studio/engines/comfyui.py index 2d466d4..b933669 100644 --- a/studio/engines/comfyui.py +++ b/studio/engines/comfyui.py @@ -1,9 +1,11 @@ -"""Fully local engine: Qwen Image Edit 2511 (+ Multiple-Angles LoRA) via ComfyUI.""" +"""Fully local engine: Qwen-Image 2.1 (unified generate + edit) via ComfyUI.""" from __future__ import annotations from pathlib import Path +import httpx + from studio import comfy_api from studio.engines.base import GenerationError from studio.shotplan import Shot @@ -27,21 +29,22 @@ def _source_name(self, source: Path) -> str: return self._uploaded[source] def generate(self, sources: list[Path], shot: Shot, out_path: Path, seed: int) -> Path: - # Qwen Edit works from one reference; use the primary (first) source. - graph = comfy_api.load_template("qwen_edit") - graph["1"]["inputs"]["image"] = self._source_name(sources[0]) - graph["9"]["inputs"]["prompt"] = shot.local_prompt - # The Multiple-Angles LoRA is trigger-based (); zero it out for - # plain pose/scene edits so it cannot bias them. 0.9 per fal's tips. - graph["8"]["inputs"]["strength_model"] = 0.9 if shot.kind == "angle" else 0.0 - graph["14"]["inputs"]["seed"] = seed + # One reference per shot; use the primary (first) source. Qwen-Image 2.1 + # accepts up to 10, but that is untested here (see ARCHITECTURE.md). + graph = comfy_api.load_template("qwen21_edit") + graph["5"]["inputs"]["prompt"] = shot.local_prompt + graph["6"]["inputs"]["seed"] = seed last_err: Exception | None = None for _ in range(2): try: + # Upload inside the try: ComfyUI dying mid-batch (OOM restart) + # surfaces as a raw httpx error from here, from /history polling + # or from /view — all of which must fail ONE shot, not the run. + graph["1"]["inputs"]["image"] = self._source_name(sources[0]) refs = comfy_api.run_prompt(graph, timeout=600) return comfy_api.fetch_image(refs[0], out_path) - except comfy_api.ComfyError as e: + except (comfy_api.ComfyError, httpx.HTTPError) as e: last_err = e - graph["14"]["inputs"]["seed"] = seed + 1 + graph["6"]["inputs"]["seed"] = seed + 1 raise GenerationError(f"shot {shot.id}: {last_err}") diff --git a/studio/isolate.py b/studio/isolate.py index f54cd65..41bf92a 100644 --- a/studio/isolate.py +++ b/studio/isolate.py @@ -6,9 +6,9 @@ page and authenticate (`hf auth login` or HF_TOKEN) before first use. - "comfyui" — the bundled SAM3 workflow templates, run on your ComfyUI. -Isolation matters twice: backgrounds/props can't leak into generations or the -final dataset, and edit models trained on clean renders (the Multiple-Angles -LoRA) behave far better on isolated subjects. +Isolation matters because backgrounds and carried props otherwise leak into +generations and the final dataset. Locally it is the only prop control — the +local engine cannot be told to omit one (naming it makes Qwen draw it). """ from __future__ import annotations diff --git a/studio/shot_style.py b/studio/shot_style.py index 83d9040..94000de 100644 --- a/studio/shot_style.py +++ b/studio/shot_style.py @@ -25,9 +25,9 @@ and imperfections of a real capture. The `photographic` preset is therefore built from camera vocabulary (lens, depth of field, sensor, texture). -`local` is kept short — it goes to Qwen-Image-Edit, which follows terse prompts -and is not a prose model. `cloud` is a full sentence for Gemini. `sample_lead` -is the noun phrase ⑤ uses to open a training sample prompt. +`cloud` is a full sentence, and goes into BOTH engines' prompts (Qwen-Image 2.1 +follows prose as well as Gemini does). `sample_lead` is the noun phrase ⑤ uses +to open a training sample prompt. """ from __future__ import annotations @@ -42,7 +42,6 @@ class ShotStyle: key: str label: str - local: str # terse clause for the ComfyUI / Qwen-Image-Edit prompt cloud: str # full sentence for the Gemini prompt sample_lead: str # how ⑤'s sample prompt names the medium @@ -54,7 +53,6 @@ def is_match(self) -> bool: _STYLES: tuple[ShotStyle, ...] = ( ShotStyle( MATCH, "Match the reference image (default)", - "in the same art style and medium as the reference", "Match the reference image's medium, art style and rendering exactly — if " "the reference is a drawing, painting, or 3D render, the result must be the " "same medium, not a photograph.", @@ -62,8 +60,6 @@ def is_match(self) -> bool: ), ShotStyle( "photographic", "Photographic (real camera capture)", - "photographic, real camera capture, natural lens depth of field, " - "true skin and material texture", "Rendered as a photograph: a real camera capture, with natural lens depth of " "field and focus falloff, true-to-life skin and material texture, and light " "behaving as it does on a sensor.", @@ -71,49 +67,42 @@ def is_match(self) -> bool: ), ShotStyle( "anime", "Anime / manga illustration", - "anime illustration, clean linework, cel shading", "Rendered as an anime illustration: clean linework, cel shading and flat " "colour fills, in the drawing style of the reference.", "an anime illustration of", ), ShotStyle( "comic", "Comic / cartoon (Western)", - "western comic illustration, bold ink outlines, flat colour", "Rendered as a Western comic illustration: bold ink outlines, flat colour " "fills and graphic shading.", "a comic illustration of", ), ShotStyle( "illustration", "Digital illustration / concept art", - "digital illustration, painterly brushwork, concept art", "Rendered as a digital illustration: painterly brushwork and concept-art " "finish, with visible artistic rendering rather than a photographic capture.", "a digital illustration of", ), ShotStyle( "painting", "Traditional painting (oil / watercolour)", - "traditional painting, visible brush strokes, canvas texture", "Rendered as a traditional painting: visible brush strokes, pigment and " "canvas or paper texture.", "a painting of", ), ShotStyle( "render3d", "3D render / CGI", - "3D render, CGI, physically based shading", "Rendered as a 3D CGI render: physically based shading and materials, clean " "computer-generated geometry.", "a 3D render of", ), ShotStyle( "lineart", "Ink line art / sketch", - "ink line art, black and white linework, minimal shading", "Rendered as ink line art: black-and-white linework with minimal shading and " "no photographic detail.", "a line art drawing of", ), ShotStyle( CUSTOM, "Custom — describe it below", - "", # filled in from the user's text by resolve() "", "", ), @@ -153,7 +142,6 @@ def resolve(key: str, custom_text: str = "") -> ShotStyle: return SHOT_STYLES[MATCH] return ShotStyle( CUSTOM, style.label, - _lower_first(text), # The connective is the whole point — see rule 2 in the module docstring. f"Rendered as {_lower_first(text)}.", f"{text}, ", diff --git a/studio/shotplan.py b/studio/shotplan.py index 34aa5c1..cee51bb 100644 --- a/studio/shotplan.py +++ b/studio/shotplan.py @@ -31,10 +31,9 @@ class Shot(BaseModel): id: str # Character plan: "angle" | "pose" | "emotion". # Concept plan: "angle" | "framing" | "context". - # Only "angle" is special downstream (it drives the Multiple-Angles LoRA - # strength in the ComfyUI engine and the "isolate angle shots" option). + # Only "angle" is special downstream (the "isolate angle shots" option). kind: str - local_prompt: str # Qwen-Image-Edit-2511 prompt (angles use the LoRA grammar) + local_prompt: str # Qwen-Image 2.1 instruction (same shape as cloud_prompt) cloud_prompt: str # plain-English instruction for Nano Banana # Rear views hallucinate when generated straight from a front reference; # chain them off a generated side view instead (stepwise rotation). @@ -49,8 +48,8 @@ class Shot(BaseModel): outfit: str = "" -# Each tuple is: (id_suffix, kind, grammar or pose stub, plain-English -# description, chain_from, emotion, setting). +# Each tuple is: (id_suffix, kind, local phrase for Qwen-Image 2.1, plain-English +# description for Gemini, chain_from, emotion, setting). # # Design goals: # - 9 angles: the core turnaround, each with a different setting/lighting so no @@ -67,7 +66,7 @@ class Shot(BaseModel): ( "front", "angle", - "front view eye-level shot medium shot", + "seen directly from the front at eye level, full body visible", "seen directly from the front at eye level, full body visible", "", "neutral", @@ -76,7 +75,7 @@ class Shot(BaseModel): ( "front-right", "angle", - "front-right quarter view eye-level shot medium shot", + "with the camera moved 45 degrees around the character to its right, so the character is seen in a three-quarter view showing one side of the face and body, full body visible", "seen from a front-right three-quarter angle at eye level, full body visible", "", "neutral", @@ -85,7 +84,7 @@ class Shot(BaseModel): ( "right", "angle", - "right side view eye-level shot medium shot", + "seen directly from the right side in full profile, full body visible", "seen directly from the right side in full profile, full body visible", "", "neutral", @@ -94,7 +93,7 @@ class Shot(BaseModel): ( "back-right", "angle", - "back-right quarter view eye-level shot medium shot", + "seen from a back-right three-quarter angle, full body visible", "seen from a back-right three-quarter angle, full body visible", "angle-right", "neutral", @@ -103,7 +102,7 @@ class Shot(BaseModel): ( "back", "angle", - "back view eye-level shot medium shot", + "seen directly from behind, full body visible", "seen directly from behind, full body visible", "angle-right", "neutral", @@ -112,7 +111,7 @@ class Shot(BaseModel): ( "back-left", "angle", - "back-left quarter view eye-level shot medium shot", + "seen from a back-left three-quarter angle, full body visible", "seen from a back-left three-quarter angle, full body visible", "angle-left", "neutral", @@ -121,7 +120,7 @@ class Shot(BaseModel): ( "left", "angle", - "left side view eye-level shot medium shot", + "seen directly from the left side in full profile, full body visible", "seen directly from the left side in full profile, full body visible", "", "neutral", @@ -130,7 +129,7 @@ class Shot(BaseModel): ( "front-left", "angle", - "front-left quarter view eye-level shot medium shot", + "in a three-quarter view, with the character's body and face turned toward the left edge of the image, full body visible", "seen from a front-left three-quarter angle at eye level, full body visible", "", "neutral", @@ -139,7 +138,7 @@ class Shot(BaseModel): ( "low", "angle", - "front view low-angle shot medium shot", + "seen from a low camera angle looking up", "seen from a low camera angle looking up", "", "confident", @@ -222,7 +221,7 @@ class Shot(BaseModel): ( "smiling", "emotion", - "close-up of the face, smiling expression", + "a close-up of the face and upper shoulders, smiling warmly", "a close-up of the face and upper shoulders, smiling warmly", "", "smiling", @@ -231,7 +230,7 @@ class Shot(BaseModel): ( "serious", "emotion", - "close-up of the face, serious expression", + "a close-up of the face and upper shoulders, with a stern, serious expression and a furrowed brow", "a close-up of the face and upper shoulders, serious expression", "", "serious", @@ -240,7 +239,7 @@ class Shot(BaseModel): ( "surprised", "emotion", - "close-up of the face, surprised expression", + "a close-up of the face and upper shoulders, with wide eyes and raised eyebrows, mouth open in surprise", "a close-up of the face and upper shoulders, surprised expression", "", "surprised", @@ -249,7 +248,7 @@ class Shot(BaseModel): ( "laughing", "emotion", - "close-up of the face, laughing expression", + "a close-up of the face and upper shoulders, laughing heartily with the eyes squeezed half shut", "a close-up of the face and upper shoulders, laughing openly", "", "laughing", @@ -258,7 +257,7 @@ class Shot(BaseModel): ( "contemplative", "emotion", - "close-up of the face, contemplative expression", + "a close-up of the face and upper shoulders, with a thoughtful, distant gaze", "a close-up of the face and upper shoulders, contemplative gaze", "", "contemplative", @@ -267,7 +266,7 @@ class Shot(BaseModel): ( "confident", "emotion", - "close-up of the face, confident expression", + "a close-up of the face and upper shoulders, with a confident look and a slight smirk", "a close-up of the face and upper shoulders, confident expression", "", "confident", @@ -276,7 +275,7 @@ class Shot(BaseModel): ( "sad", "emotion", - "close-up of the face, sad expression", + "a close-up of the face and upper shoulders, with a sad, downcast expression and glistening eyes", "a close-up of the face and upper shoulders, sad expression", "", "sad", @@ -285,14 +284,13 @@ class Shot(BaseModel): ] -# Concept plan. Each tuple is: (id_suffix, kind, grammar or phrase, -# plain-English description, chain_from, setting). +# Concept plan. Each tuple is: (id_suffix, kind, local phrase, plain-English +# description, chain_from, setting). # # Design goals (mirroring the character plan's, minus identity): -# - 10 angles: the turnaround the Multiple-Angles LoRA is actually good at, -# reusing the SAME grammar (view + camera height + shot size). No -# invented grammar terms — a "top view" the LoRA was never trained on would -# silently produce a normal front view. +# - 10 angles: the turnaround. Three-quarter fronts use camera-orbit wording +# ("moved 45 degrees around it") because "front-right quarter view" came back +# as a plain front view on Qwen-Image 2.1; side/back/low/high plain views work. # - 4 framing shots: scale variation (extreme detail -> tiny in a wide shot) so # the LoRA isn't locked to one distance. # - 4 context shots: where the thing sits and how it is used, including a hand @@ -306,7 +304,7 @@ class Shot(BaseModel): ( "front", "angle", - "front view eye-level shot medium shot", + "seen directly from the front at eye level, the whole subject in frame", "seen directly from the front at eye level, the whole subject in frame", "", "on a plain neutral gray studio backdrop with soft even lighting", @@ -314,7 +312,7 @@ class Shot(BaseModel): ( "front-right", "angle", - "front-right quarter view eye-level shot medium shot", + "with the camera moved 45 degrees around it to its right, so it is seen in a three-quarter view showing its front and one side", "seen from a front-right three-quarter angle at eye level", "", "on a wooden tabletop in a warmly lit room", @@ -322,7 +320,7 @@ class Shot(BaseModel): ( "right", "angle", - "right side view eye-level shot medium shot", + "seen directly from the right side in full profile", "seen directly from the right side in full profile", "", "outdoors in daylight on flat open ground", @@ -330,7 +328,7 @@ class Shot(BaseModel): ( "back-right", "angle", - "back-right quarter view eye-level shot medium shot", + "seen from a back-right three-quarter angle", "seen from a back-right three-quarter angle", "angle-right", "on a concrete surface under overcast daylight", @@ -338,7 +336,7 @@ class Shot(BaseModel): ( "back", "angle", - "back view eye-level shot medium shot", + "seen directly from behind", "seen directly from behind", "angle-right", "on a plain neutral gray studio backdrop with soft even lighting", @@ -346,7 +344,7 @@ class Shot(BaseModel): ( "back-left", "angle", - "back-left quarter view eye-level shot medium shot", + "seen from a back-left three-quarter angle", "seen from a back-left three-quarter angle", "angle-left", "against a dark background with dramatic hard side lighting", @@ -354,7 +352,7 @@ class Shot(BaseModel): ( "left", "angle", - "left side view eye-level shot medium shot", + "seen directly from the left side in full profile", "seen directly from the left side in full profile", "", "outdoors at golden hour with warm backlighting", @@ -362,7 +360,7 @@ class Shot(BaseModel): ( "front-left", "angle", - "front-left quarter view eye-level shot medium shot", + "in a three-quarter view, with it turned toward the left edge of the image, showing its front and one side", "seen from a front-left three-quarter angle at eye level", "", "on a wooden tabletop in a warmly lit room", @@ -370,7 +368,7 @@ class Shot(BaseModel): ( "low", "angle", - "front view low-angle shot medium shot", + "seen from a low camera angle looking up at it", "seen from a low camera angle looking up at it", "", "outdoors in daylight on flat open ground", @@ -378,7 +376,7 @@ class Shot(BaseModel): ( "high", "angle", - "front view high-angle shot medium shot", + "seen from a high camera angle looking down on it", "seen from a high camera angle looking down on it", "", "on a concrete surface under overcast daylight", @@ -472,43 +470,47 @@ def _indefinite_article(word: str) -> str: return "an" if word[:1].lower() in "aeiou" else "a" +def _instruction(subject: str, phrase: str, setting: str, emotion: str, + outfit: str, style: ShotStyle, with_mood: bool) -> str: + """The one-sentence edit instruction both engines are given. + + Qwen-Image 2.1 follows the same plain-English instruction Gemini does, and it + keeps identity and medium far better with it than 2511 did with terse tags + (measured). The medium is stated once, at the END, as its own sentence — never + as an adjective on "image" (that used to read "Generate a photorealistic image + of …", which turned every illustrated reference into a photograph). + """ + parts = [ + f"Generate an image of exactly the same {_subject_phrase(subject)} " + "from the reference image(s), identical in every physical detail", + f", {phrase}", + ] + if setting: + parts.append(f", {setting}") + if with_mood and emotion and emotion != "neutral": + parts.append(f", with {_indefinite_article(emotion)} {emotion} expression") + if outfit: + parts.append(f", wearing {outfit}") + parts.append(f". {style.cloud}") + return "".join(parts) + + def _build_local_prompt( - kind: str, grammar_or_pose: str, setting: str, emotion: str, + kind: str, phrase: str, setting: str, emotion: str, outfit: str = "", subject: str = "subject", style: ShotStyle | None = None ) -> str: - """Build the ComfyUI/Qwen-Edit prompt. + """Build the ComfyUI / Qwen-Image 2.1 prompt. - Angle shots keep the tight Multiple-Angles LoRA grammar so the LoRA - can do its job; every other kind is plain English with setting/lighting - folded in. The emotion is appended so it influences expression without - breaking the LoRA grammar for angles. An explicit outfit, when given, is - appended after the grammar/pose so clothing can vary. + Same sentence as the cloud prompt, with one difference: close-up shots keep + their setting. Without it every close-up came back on the reference's own + backdrop, and the dataset ends up with seven identical backgrounds. `subject` is interpolated here, not left as a `{subject}` placeholder: - nothing downstream formats the local prompt, so a placeholder went to - ComfyUI verbatim. Shots with no emotion (the Concept plan) drop the mood - clause instead of emitting a dangling " mood". - - **Angle shots get no style clause.** Their prompt is the grammar the - Multiple-Angles LoRA was trained on (clean splat renders); appending prose - degrades it, which is the same reason `apply_prop_exclusion` skips them. - They carry no medium claim today either, so nothing is lost — Qwen-Edit - follows the reference image's medium on its own. + nothing downstream formats the local prompt. """ - wardrobe = f", wearing {outfit}" if outfit else "" - if kind == "angle": - prompt = f" {grammar_or_pose}" - if emotion and emotion != "neutral": - prompt += f", {emotion} expression" - return prompt + wardrobe - # Settings are complete phrases ("in a warmly lit interior room", "outdoors - # at golden hour") — they carry their own preposition. - medium = (style or shot_style.SHOT_STYLES[shot_style.MATCH]).local - tail = f"{emotion} mood, " if emotion else "" - tail += f"{medium}, " if medium else "" - tail += "consistent identity" if emotion else "unchanged form" - return (f"the same {_subject_phrase(subject)}, {grammar_or_pose}{wardrobe}, " - f"{setting}, {tail}") + style = style or shot_style.SHOT_STYLES[shot_style.MATCH] + return _instruction(subject, phrase, setting, emotion, outfit, style, + with_mood=kind != "emotion") def _build_cloud_prompt( @@ -517,28 +519,25 @@ def _build_cloud_prompt( ) -> str: """Build the plain-English Nano Banana instruction. - The medium is stated once, at the END, as its own sentence — never as an - adjective on "image" (that used to read "Generate a photorealistic image - of …", which turned every illustrated reference into a photograph). + Close-up expression shots carry their own framing AND their expression in + the description, so they get neither a setting nor a mood clause here. """ style = style or shot_style.SHOT_STYLES[shot_style.MATCH] - parts = [ - f"Generate an image of exactly the same {_subject_phrase(subject)} " - "from the reference image(s), identical in every physical detail", - f", {description}", - ] - # Close-up expression shots carry their own framing AND their expression in - # the description; every other kind names the setting (already a complete - # phrase) and, for characters, the mood. - if kind != "emotion": - if setting: - parts.append(f", {setting}") - if emotion and emotion != "neutral": - parts.append(f", with {_indefinite_article(emotion)} {emotion} expression") - if outfit: - parts.append(f", wearing {outfit}") - parts.append(f". {style.cloud}") - return "".join(parts) + is_close_up = kind == "emotion" + return _instruction(subject, description, "" if is_close_up else setting, emotion, + outfit, style, with_mood=not is_close_up) + + +def _insert_outfit(prompt: str, phrase: str) -> str: + """Put ", wearing …" at the end of the instruction clause, before the style + sentence. Appends instead when the prompt has no sentence break (a cell the + user rewrote by hand). + + ponytail: splits at the first ". " — a subject name containing one ("Dr. X") + would take the outfit early; rename the subject or edit the prompt cell. + """ + head, sep, tail = prompt.partition(". ") + return f"{head}, {phrase}{sep}{tail}" def apply_wardrobe(shot: Shot) -> Shot: @@ -555,14 +554,9 @@ def apply_wardrobe(shot: Shot) -> Shot: local = shot.local_prompt cloud = shot.cloud_prompt if phrase.lower() not in local.lower(): - local = f"{local}, {phrase}" + local = _insert_outfit(local, phrase) if phrase.lower() not in cloud.lower(): - # Insert before the trailing "Keep the same..." sentence when present. - if ". Keep the same" in cloud: - head, _, tail = cloud.partition(". Keep the same") - cloud = f"{head}, {phrase}. Keep the same{tail}" - else: - cloud = f"{cloud}, {phrase}" + cloud = _insert_outfit(cloud, phrase) return shot.model_copy(update={"local_prompt": local, "cloud_prompt": cloud}) @@ -570,34 +564,29 @@ def apply_wardrobe(shot: Shot) -> Shot: # dataset where 20/24 images show the same backpack teaches the LoRA that the # backpack IS the character. These clauses ask the generator to drop them. # -# Deliberately NOT applied to `kind="angle"` local prompts: those use the -# Multiple-Angles LoRA grammar, which is trained on clean splat renders and -# degrades when prose is appended (see ARCHITECTURE.md). Diffusion models also -# handle negation poorly in a positive prompt — naming "backpack" can summon one. -# Angle shots rely on isolation instead, which removes props from the reference -# itself and is the mechanism that actually works. +# Deliberately NOT applied to local prompts: Qwen-Image 2.1 draws what a prompt +# names, even negated — "do not include any backpacks" put a backpack on the +# character in 2 of 8 test shots. Gemini follows the negation. Locally, isolating +# the source in ① removes props from the reference itself, which is the mechanism +# that actually works. _CLOUD_NO_PROPS = ( " Show only the character and the clothing worn on their body — do not " "include any backpacks, bags, straps, held objects, tools, props, or " "accessories that appear in the reference image." ) -_LOCAL_NO_PROPS = ", without any bags or carried accessories" def apply_prop_exclusion(shot: Shot) -> Shot: - """Return a copy of `shot` asking the generator to omit reference props. + """Return a copy of `shot` whose CLOUD prompt asks to omit reference props. Applied at generation time (like `apply_wardrobe`) rather than baked into the plan, so the column stays honest and hand-edited prompt cells still get the - clause. Idempotent. + clause. The local prompt is never touched (see above). Idempotent. """ cloud = shot.cloud_prompt if _CLOUD_NO_PROPS.strip() not in cloud: cloud = f"{cloud}{_CLOUD_NO_PROPS}" - local = shot.local_prompt - if shot.kind != "angle" and _LOCAL_NO_PROPS not in local: - local = f"{local}{_LOCAL_NO_PROPS}" - return shot.model_copy(update={"local_prompt": local, "cloud_prompt": cloud}) + return shot.model_copy(update={"cloud_prompt": cloud}) def default_plan(subject: str = "the character", diff --git a/tests/test_cli_preflight.py b/tests/test_cli_preflight.py index 1e156ed..cade4ef 100644 --- a/tests/test_cli_preflight.py +++ b/tests/test_cli_preflight.py @@ -108,7 +108,7 @@ def test_a_model_name_mismatch_warns_but_does_not_abort( monkeypatch.setattr(comfy_api, "server_status", _up) monkeypatch.setattr(doctor, "check_comfyui_models", lambda: doctor.Check("ComfyUI models", True, - "qwen_edit: no such vae_name", warn=True)) + "qwen21_edit: no such vae_name", warn=True)) cli._preflight_comfyui("comfyui") # must not raise diff --git a/tests/test_comfy_api.py b/tests/test_comfy_api.py index 741037b..a05d87d 100644 --- a/tests/test_comfy_api.py +++ b/tests/test_comfy_api.py @@ -145,7 +145,7 @@ def test_every_bundled_model_filename_is_overridable() -> None: Everyone downloads these weights from a different mirror under a different name, so an un-remapped input fails as ComfyUI's own opaque "Value not in list: ... (list of length 10574)" with no setting to change. This caught - `clip_name` and `vae_name` in qwen_edit.json, which were invisible to both + `clip_name` and `vae_name` in the old qwen_edit.json, which were invisible to both `.env` and `doctor`'s ComfyUI-models check for four releases. """ covered = set(comfy_api._MODEL_INPUTS) diff --git a/tests/test_comfy_diagnostics.py b/tests/test_comfy_diagnostics.py index 4b4dfed..f0db2cf 100644 --- a/tests/test_comfy_diagnostics.py +++ b/tests/test_comfy_diagnostics.py @@ -146,3 +146,45 @@ def must_not_post(*a, **kw): comfy_api.run_prompt({"1": {"class_type": "Krea2EditModelPatch"}}) assert "Krea2EditModelPatch" in str(e.value) assert "out of date" in str(e.value) + + +# ---------- polling a running job ---------- + +def _queue_stubs(monkeypatch: pytest.MonkeyPatch, history) -> None: + """`run_prompt` with the network replaced: `history` answers each /history poll.""" + monkeypatch.setattr(comfy_api, "queue_backlog", lambda: 0) + monkeypatch.setattr(comfy_api, "missing_node_types", lambda graph: []) + monkeypatch.setattr(comfy_api.time, "sleep", lambda s: None) + monkeypatch.setattr(comfy_api.httpx, "post", lambda *a, **k: httpx.Response( + 200, json={"prompt_id": "p1"}, request=httpx.Request("POST", "http://x"))) + monkeypatch.setattr(comfy_api.httpx, "get", history) + + +def test_one_slow_poll_does_not_fail_a_running_job(monkeypatch: pytest.MonkeyPatch) -> None: + """ComfyUI stops answering HTTP while it swaps models, but the job finishes. + A single ReadTimeout used to fail the shot (and the retry queued it twice).""" + polls = iter([httpx.ReadTimeout("timed out"), httpx.ReadTimeout("timed out"), "ok"]) + + def history(*a, **k): + item = next(polls) + if isinstance(item, Exception): + raise item + done = {"p1": {"status": {"completed": True}, + "outputs": {"8": {"images": [{"filename": "o.png"}]}}}} + return httpx.Response(200, json=done, request=httpx.Request("GET", "http://x")) + + _queue_stubs(monkeypatch, history) + assert comfy_api.run_prompt({}) == [{"filename": "o.png"}] + + +def test_a_server_that_stays_silent_is_given_up_on(monkeypatch: pytest.MonkeyPatch) -> None: + calls = {"n": 0} + + def history(*a, **k): + calls["n"] += 1 + raise httpx.ConnectError("refused") + + _queue_stubs(monkeypatch, history) + with pytest.raises(comfy_api.ComfyError, match="stopped answering"): + comfy_api.run_prompt({}) + assert calls["n"] == comfy_api.MAX_STALLED_POLLS diff --git a/tests/test_comfy_engine.py b/tests/test_comfy_engine.py new file mode 100644 index 0000000..beec759 --- /dev/null +++ b/tests/test_comfy_engine.py @@ -0,0 +1,87 @@ +"""The local engine: what it puts in the Qwen-Image 2.1 graph, and how it fails. + +No ComfyUI needed — every `comfy_api` call that would touch the network is stubbed. +""" + +from __future__ import annotations + +from pathlib import Path + +import httpx +import pytest + +from studio import comfy_api, pipeline +from studio.engines.base import GenerationError +from studio.engines.comfyui import ComfyUIEngine +from studio.shotplan import default_plan + + +@pytest.fixture +def offline(monkeypatch: pytest.MonkeyPatch) -> None: + """A 'reachable' server that has no model lists — and never a real one: a + developer's own ComfyUI on :8188 would otherwise answer `_combo_options`.""" + monkeypatch.setattr(comfy_api, "server_status", lambda timeout=3.0: (True, "stub")) + monkeypatch.setattr(comfy_api, "_combo_options", lambda *_: []) + + +def _src(tmp_path: Path) -> list[Path]: + src = tmp_path / "ref.png" + src.write_bytes(b"x") + return [src] + + +def test_the_graph_carries_the_shots_prompt_seed_and_reference( + offline: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + seen: dict = {} + monkeypatch.setattr(comfy_api, "upload_image", lambda path: "uploaded.png") + + def run_prompt(graph: dict, timeout: float = 0, **_: object) -> list[dict]: + seen.update(graph) + return [{"filename": "o.png"}] + + monkeypatch.setattr(comfy_api, "run_prompt", run_prompt) + monkeypatch.setattr(comfy_api, "fetch_image", lambda ref, out: out) + shot = default_plan()[0] + + ComfyUIEngine().generate(_src(tmp_path), shot, tmp_path / "o.png", seed=1234) + + assert seen["1"]["inputs"]["image"] == "uploaded.png" + assert seen["5"]["inputs"]["prompt"] == shot.local_prompt + assert seen["6"]["inputs"]["seed"] == 1234 + # Model names come from settings, not the template's hard-coded defaults. + assert seen["2"]["inputs"]["unet_name"].endswith(".safetensors") + + +@pytest.mark.parametrize("stage", ["upload_image", "run_prompt", "fetch_image"]) +def test_a_dead_comfyui_fails_the_shot_not_the_run( + stage: str, offline: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + """ComfyUI dying mid-batch (an OOM restart) raises a raw httpx error from any + of three calls. It must become a per-shot failure: anything else reaches the + UI's gr.Error, which throws away the shots that already finished.""" + def boom(*_: object, **__: object) -> None: + raise httpx.ConnectError("connection refused") + + monkeypatch.setattr(comfy_api, "upload_image", lambda path: "uploaded.png") + monkeypatch.setattr(comfy_api, "run_prompt", lambda *a, **k: [{"filename": "o.png"}]) + monkeypatch.setattr(comfy_api, "fetch_image", lambda ref, out: out) + monkeypatch.setattr(comfy_api, stage, boom) + + with pytest.raises(GenerationError, match="connection refused"): + ComfyUIEngine().generate(_src(tmp_path), default_plan()[0], tmp_path / "o.png", seed=1) + + +def test_generate_shots_keeps_going_when_comfyui_dies( + offline: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + def boom(path: Path) -> str: + raise httpx.ConnectError("connection refused") + + monkeypatch.setattr(comfy_api, "upload_image", boom) + + results = pipeline.generate_shots(_src(tmp_path), default_plan()[:2], "comfyui", + tmp_path / "out", progress=lambda _: None) + + assert [r.path for r in results] == [None, None] + assert all("connection refused" in r.error for r in results) diff --git a/tests/test_concept_plan.py b/tests/test_concept_plan.py index 4726d93..3257064 100644 --- a/tests/test_concept_plan.py +++ b/tests/test_concept_plan.py @@ -77,32 +77,24 @@ def test_wardrobe_is_a_noop_on_concept_shots() -> None: assert apply_wardrobe(shot) is shot -# ---------- angle shots keep the LoRA grammar ---------- +# ---------- local prompts are plain-English instructions ---------- -def test_concept_angle_shots_use_sks_grammar() -> None: - angles = [s for s in concept_plan() if s.kind == "angle"] - assert len(angles) == 10 - for shot in angles: - assert shot.local_prompt.startswith(" ") +def test_concept_local_prompts_are_plain_english_instructions() -> None: + plan = concept_plan() + assert len([s for s in plan if s.kind == "angle"]) == 10 + for shot in plan: + assert "" not in shot.local_prompt + assert shot.local_prompt.startswith("Generate an image of exactly the same") + assert shot.setting in shot.local_prompt -def test_concept_non_angle_shots_do_not_use_sks() -> None: - for shot in concept_plan(): - if shot.kind != "angle": - assert "" not in shot.local_prompt - - -def test_concept_angle_grammar_reuses_the_character_plans_vocabulary() -> None: - """The Multiple-Angles LoRA only knows the grammar it was trained on, so the - concept turnaround reuses the character plan's phrases rather than inventing - new ones (an untrained "top view" would silently render a front view). The - one addition follows the same shape as the attested low-angle shot.""" - # Character angle prompts may carry a trailing ", expression". - known = {s.local_prompt.split(",")[0] for s in default_plan() if s.kind == "angle"} - known.add(" front view high-angle shot medium shot") - for shot in concept_plan(): - if shot.kind == "angle": - assert shot.local_prompt in known +def test_concept_three_quarter_fronts_turn_opposite_ways() -> None: + """Same wording split as the character plan (see test_shotplan): orbit wording + for the right, the image edge for the left, because "to its left" never mirrored.""" + by_id = {s.id: s for s in concept_plan()} + assert "camera moved 45 degrees around it to its right" in \ + by_id["angle-front-right"].local_prompt + assert "turned toward the left edge of the image" in by_id["angle-front-left"].local_prompt def test_rear_concept_views_chain_off_a_generated_side_view() -> None: @@ -160,14 +152,13 @@ def test_character_setting_is_not_prefixed_twice() -> None: assert "in outdoors" not in shot.cloud_prompt -def test_character_pose_shots_keep_their_mood_and_identity_clause() -> None: +def test_character_pose_shots_keep_their_mood_and_medium_clause() -> None: pose = next(s for s in default_plan() if s.kind == "pose") assert pose.emotion - assert f"{pose.emotion} mood" in pose.local_prompt + assert f"with a {pose.emotion} expression" in pose.local_prompt # The medium clause used to be a hard-coded "photorealistic"; it now comes # from the shot style, whose default preserves the reference's own medium. - assert "same art style and medium as the reference" in pose.local_prompt - assert pose.local_prompt.endswith("consistent identity") + assert "Match the reference image's medium" in pose.local_prompt def test_character_closeups_do_not_repeat_the_expression() -> None: @@ -176,6 +167,10 @@ def test_character_closeups_do_not_repeat_the_expression() -> None: for shot in default_plan(): if shot.kind == "emotion": assert "with a " not in shot.cloud_prompt + # The local phrases are vivid ("with a stern, serious expression"), so + # only the builder's own appended clause is forbidden here. + for article in ("a", "an"): + assert f"with {article} {shot.emotion} expression" not in shot.local_prompt def test_concept_prop_exclusion_still_composes() -> None: diff --git a/tests/test_plan_io.py b/tests/test_plan_io.py index 6b4dbfe..5c4f0a4 100644 --- a/tests/test_plan_io.py +++ b/tests/test_plan_io.py @@ -29,3 +29,19 @@ def test_load_rejects_non_list(tmp_path: Path) -> None: bad.write_text("just: a-mapping\n", encoding="utf-8") with pytest.raises(ValueError): load_plan(bad) + + +def test_loading_a_pre_0_17_plan_warns_about_sks_prompts() -> None: + """Plans saved by 0.16 hold the old engine's `` grammar; the 2.1 engine + would send it as literal text, so the load note has to say so.""" + import app + from studio.config import settings + + old = default_plan()[:2] + old[0].local_prompt = " front view eye-level shot medium shot" + save_plan(old, settings.shot_plans_dir / "old") + _, note = app.do_load_plan("old") + assert "1 shot(s) still use the old" in note + + save_plan(default_plan()[:2], settings.shot_plans_dir / "new") + assert "⚠️" not in app.do_load_plan("new")[1] diff --git a/tests/test_prop_exclusion.py b/tests/test_prop_exclusion.py index 1958e66..514c380 100644 --- a/tests/test_prop_exclusion.py +++ b/tests/test_prop_exclusion.py @@ -23,17 +23,12 @@ def test_cloud_prompt_gets_the_exclusion_clause() -> None: assert "do not include any backpacks" in shot.cloud_prompt -def test_angle_local_prompt_is_left_alone() -> None: - """Angle shots use the Multiple-Angles LoRA grammar, which is trained - on clean splat renders and degrades when prose is appended.""" - original = _shot("angle") - shot = apply_prop_exclusion(original) - assert shot.local_prompt == original.local_prompt - - -def test_pose_local_prompt_gets_a_short_clause() -> None: - shot = apply_prop_exclusion(_shot("pose")) - assert "without any bags or carried accessories" in shot.local_prompt +def test_local_prompt_is_never_touched() -> None: + """Qwen-Image 2.1 draws what a prompt names, even negated — the clause put a + backpack on the character in 2 of 8 test shots. Gemini follows the negation.""" + for kind in ("angle", "pose", "emotion"): + original = _shot(kind) + assert apply_prop_exclusion(original).local_prompt == original.local_prompt def test_is_idempotent() -> None: diff --git a/tests/test_shot_style.py b/tests/test_shot_style.py index ed354b2..d7f7f15 100644 --- a/tests/test_shot_style.py +++ b/tests/test_shot_style.py @@ -25,12 +25,12 @@ def test_match_is_the_default_everywhere() -> None: assert resolve("no-such-style").key == MATCH -def test_every_preset_has_all_three_renderings() -> None: +def test_every_preset_has_both_renderings() -> None: for key, style in SHOT_STYLES.items(): assert style.label if key == CUSTOM: continue # filled in from the user's text by resolve() - assert style.local and style.cloud and style.sample_lead + assert style.cloud and style.sample_lead # ---------- rule 1: `match` instructs, it does not merely stay silent ---------- @@ -79,8 +79,8 @@ def test_photographic_uses_camera_vocabulary_not_photorealistic() -> None: """They are different aesthetic targets: 'photorealistic' trends toward an idealized CG-adjacent render, a photograph has real optics.""" style = SHOT_STYLES["photographic"] - assert "photorealistic" not in (style.local + style.cloud).lower() - assert "hyperrealistic" not in (style.local + style.cloud).lower() + assert "photorealistic" not in style.cloud.lower() + assert "hyperrealistic" not in style.cloud.lower() for term in ("camera", "lens", "depth of field", "texture"): assert term in style.cloud.lower() @@ -110,27 +110,23 @@ def test_no_generated_prompt_anywhere_says_photorealistic() -> None: def test_style_reaches_both_plans_and_both_prompt_fields() -> None: shots = plan_for_type("character", "Sy Snootles", "anime") pose = next(s for s in shots if s.kind == "pose") - assert "anime illustration" in pose.local_prompt + assert "Rendered as an anime illustration" in pose.local_prompt assert "Rendered as an anime illustration" in pose.cloud_prompt concept = plan_for_type("concept", "brass compass", "render3d") framing = next(s for s in concept if s.kind == "framing") - assert "3D render" in framing.local_prompt + assert "Rendered as a 3D CGI render" in framing.local_prompt assert "Rendered as a 3D CGI render" in framing.cloud_prompt -def test_angle_shots_keep_the_sks_grammar_clean() -> None: - """The Multiple-Angles LoRA is trained on clean splat renders; appending - prose degrades it (same reason prop exclusion skips angle local prompts).""" +def test_every_local_prompt_ends_with_the_style_sentence() -> None: + """Angle shots included: they used to be exempt (the LoRA grammar broke + when prose was appended); Qwen-Image 2.1 follows the same instruction as Gemini.""" for style_key in SHOT_STYLES: - for shot in plan_for_type("character", "X", style_key, "a woodcut print"): - if shot.kind != "angle": - continue - assert shot.local_prompt.startswith(" ") - body = SHOT_STYLES[style_key].local - if body: - assert body not in shot.local_prompt - assert "Rendered as" not in shot.local_prompt + sentence = resolve(style_key, "a woodcut print").cloud + for dtype in ("character", "concept"): + for shot in plan_for_type(dtype, "X", style_key, "a woodcut print"): + assert shot.local_prompt.endswith(sentence), (style_key, shot.id) def test_style_never_generates_regardless_of_style() -> None: diff --git a/tests/test_shotplan.py b/tests/test_shotplan.py index a94aa81..9a598e9 100644 --- a/tests/test_shotplan.py +++ b/tests/test_shotplan.py @@ -58,20 +58,42 @@ def test_no_duplicate_angle_pose_setting_combinations() -> None: assert len(combos) == len(plan) -def test_angle_shots_use_sks_grammar() -> None: - plan = default_plan() - angles = [s for s in plan if s.kind == "angle"] - assert angles - for shot in angles: - assert "" in shot.local_prompt +def test_local_prompts_are_plain_english_instructions() -> None: + for shot in default_plan(): + assert "" not in shot.local_prompt + assert shot.local_prompt.startswith("Generate an image of exactly the same") -def test_pose_and_emotion_shots_do_not_use_sks() -> None: - plan = default_plan() - non_angles = [s for s in plan if s.kind != "angle"] - assert non_angles - for shot in non_angles: - assert "" not in shot.local_prompt +def test_every_local_shot_names_its_setting() -> None: + """Close-ups included: without it each one came back on the reference's own + backdrop and the dataset ended up with seven identical backgrounds.""" + for shot in default_plan(): + assert shot.setting in shot.local_prompt, shot.id + + +def test_three_quarter_fronts_turn_opposite_ways() -> None: + """"front-right quarter view" came back as a plain front view on Qwen-Image + 2.1. Camera-orbit wording rotates it toward image-right; "to its left" did NOT + mirror it (both shots faced image-right, seen live), so the left shot names the + image edge instead (verified to face image-left).""" + by_id = {x.id: x for x in default_plan()} + right = by_id["angle-front-right"].local_prompt + left = by_id["angle-front-left"].local_prompt + assert "camera moved 45 degrees around the character to its right" in right + assert "turned toward the left edge of the image" in left + + +def test_local_prompts_never_negate() -> None: + """Qwen-Image 2.1 draws what a prompt names, even negated — "do not include any + backpacks" put a backpack on the character in 2 of 8 test shots. So the local + prompt must stay free of negation however the options are set.""" + from studio.shotplan import apply_prop_exclusion, concept_plan + + for plan in (default_plan(), concept_plan()): + for shot in plan: + local = apply_prop_exclusion(apply_wardrobe(shot)).local_prompt.lower() + for phrase in ("without", "do not", "don't", "no bags", "backpack"): + assert phrase not in local, (shot.id, phrase) def test_emotion_shots_are_closeup() -> None: @@ -115,6 +137,17 @@ def test_apply_wardrobe_injects_into_both_prompts() -> None: assert out.cloud_prompt.index("wearing") < out.cloud_prompt.index("Keep the same") +def test_apply_wardrobe_lands_before_the_style_sentence_in_real_prompts() -> None: + """The outfit used to be appended AFTER the style sentence, producing + "…not a photograph., wearing a red raincoat Show only…".""" + pose = next(x for x in default_plan() if x.kind == "pose") + out = apply_wardrobe(pose.model_copy(update={"outfit": "a red raincoat"})) + for prompt in (out.local_prompt, out.cloud_prompt): + head, _, tail = prompt.partition(". ") + assert head.endswith(", wearing a red raincoat") + assert tail.startswith("Match the reference image's medium") + + def test_apply_wardrobe_idempotent() -> None: shot = Shot(id="pose-x", kind="pose", local_prompt="walking, wearing a hat", cloud_prompt="x, wearing a hat",