From 421833e70451d1a309fa324bd05b0cc4c6679c3f Mon Sep 17 00:00:00 2001 From: Refinement Systems Date: Sun, 13 Sep 2026 00:21:53 +0200 Subject: [PATCH] flux2 klein guidance replacement --- NOTES.md | 82 ++++++++++++++++++++++++++++++++++++++---- README.md | 6 ++-- README_RUNPOD.md | 7 ++-- scripts/sweep-klein.sh | 37 ++++++++++--------- src/dltb/args.py | 9 ++--- src/dltb/klein.py | 20 +++++++++-- 6 files changed, 127 insertions(+), 34 deletions(-) diff --git a/NOTES.md b/NOTES.md index d5a0f36..942e979 100644 --- a/NOTES.md +++ b/NOTES.md @@ -229,10 +229,80 @@ never reaches the model, via three independent points: is a partial duplicate of `prompt-enhance-slight/` — delete it (and `guidance4.0/` if it ever started). -**Repo follow-ups (not yet applied):** default `GUIDANCES` to empty in -`sweep-klein.sh` (+ header note); make `dltb-klein` refuse or warn on -`--guidance-scale > 1`; candidate replacement axis that *does* reach klein: -`--num-inference-steps` (e.g. 2 / 4 / 8) as the per-pass edit-intensity probe; -optional 2-frame hash A/B (`--max-frames 2`, with vs without the flag) if an -empirical confirmation is ever wanted on a future diffusers. +**Follow-ups applied 2026-09-13:** the guidance leg of `sweep-klein.sh` is +replaced by a `--num-inference-steps` probe (`STEPS="2 8"`, bracketing the +default 4 that the ladder legs already run) — steps are the one remaining +direct per-pass edit-intensity knob that actually reaches klein. `dltb-klein` +now prints a one-time warning when `--guidance-scale > 1` is passed (warn, +not refuse, so a future diffusers that implements a real guidance path does +not break the tool). The 2-frame hash A/B remains optional and only worth +doing after a diffusers upgrade. Next-experiment sketch for klein's control +problem: dual-reference conditioning, see the last section. + +## Sketch: klein dual-reference conditioning (`--conditioning dual-ref`) + +**Status: DESIGN ONLY (2026-09-13), not implemented.** The candidate fix for +klein's control problem, replacing the pixel-blend proxy. + +**Why:** klein's calibration trouble (too weak at blend 0.6–0.8, "melting" at +0.1 — see the calibration section) is plausibly an artifact of pixel-blending +two frames into ONE reference image. Klein is trained as a (multi-)reference +editor, and the pipeline natively accepts a LIST of reference images: +verified in diffusers 0.40.0 `Flux2KleinPipeline.__call__` (step 4 — each +image is preprocessed, downscaled to ≤ 1 MP if needed, packed, and the packed +latents are `torch.cat([latents, image_latents], dim=1)`-ed on the SEQUENCE +axis; batch size comes from the prompt, not the image count). So the carried +state and the fresh frame can both be conditioning inputs: + + blend (today) : P_n = f(image = (1-a)*R(P_{n-1}) + a*N_n) + dual-ref (new) : P_n = f(image = [R(P_{n-1}), N_n]) # two references + +**Design:** + +- CLI in `dltb-klein`: `--conditioning {blend,dual-ref}` (default `blend` + until validated). `--anchor-blend` applies to `blend` only; dual-ref run + tag: `_dualref[-norepro]_tails…` (no blend component). +- `imaging.run_pass` needs NO change — it forwards `image=source`, and a + list of two PIL images flows straight through. `--width/--height` still + set the output canvas; references are resized/packed per-image by the + pipeline (our 768² frames are under the 1 MP auto-resize cap). +- `klein.py` needs its own stateful loop for dual-ref (it currently delegates + to `continuous.run`, whose blend is baked in). Either fork the loop, or — + cleaner — generalize `continuous.run` to accept a + `condition(carried, new_frame) -> source` callable, with the current + blend/reproject logic as the default implementation. +- Reprojection: probably UNNECESSARY in dual-ref (the fresh frame is an + explicit reference; the model aligns content, not pixel coordinates) — but + keep it probeable: warping the carried reference may still help temporal + stability. A/B `--reproject` / `--no-reproject`. +- Reference order is a real variable: `[P, N]` vs `[N, P]` — likely encodes + "primary vs target"; cheap 2-frame A/Bs answer it empirically. +- Tails: freeze = `[P, last_source]`; free = `[P]` alone (single-reference + regeneration from state — the pure buffer-echo case); black = `[P, black]`. + +**Open questions / risks:** + +- Token budget: each 768² reference packs to ~2.3k sequence tokens + (2×2-packed VAE latents), so two references + text is a modest sequence — + but measure the real VRAM/speed delta with the README_RUNPOD VRAM-probe + pattern before scheduling runs. +- Does klein weight multiple references equally, or is there an implicit + "first = primary" convention? (The order A/B above answers this.) +- Interaction with the steps probe: re-run the steps axis under dual-ref — + intensity may interact with conditioning strength. + +**Suggested first runs (once implemented):** + + uv run dltb-klein --model flux2-klein-4b --input input/video_cropped.mp4 \ + --conditioning dual-ref --prompt "slightly enhance the fine details" \ + --max-frames 30 --tail-frames 10 --save-every 1 + + # order A/B (2 frames each, compare): EXTRA_ARGS='--reproject' etc. + uv run dltb-klein --model flux2-klein-4b --input input/video_cropped.mp4 \ + --conditioning dual-ref --ref-order state-first --max-frames 2 + +Then add a `SMOKE_KLEIN=1` leg to `scripts/smoke.sh` (2 frames + 1 tail frame, +flux2-klein-4b; off by default because of the 15 GB download) and a +`CONDITIONING=dual-ref` toggle to `scripts/sweep-klein.sh` so the prompt +ladder + steps probe re-run under the new conditioning. diff --git a/README.md b/README.md index 4bb540f..d92d908 100644 --- a/README.md +++ b/README.md @@ -28,8 +28,10 @@ Three scripts share the library code in `src/dltb/` (`models`, `imaging`, (`flux2-klein-4b`, ungated, default / `flux2-klein-9b`, gated). Klein is a reference-image editor: no `--strength`, per-pass change scales ~linearly with the blend (default `0.1`, far below the img2img models), and the prompt - is the de-facto per-pass edit-strength knob. `scripts/sweep-klein.sh` walks - its prompt ladder and `--guidance-scale` probes. + is the de-facto per-pass edit-strength knob (`--num-inference-steps` is the + other). `scripts/sweep-klein.sh` walks its prompt ladder and steps probes + (guidance is inert for klein — CFG is disabled and there is no guidance + embedding in the distilled checkpoints). ## Models diff --git a/README_RUNPOD.md b/README_RUNPOD.md index 0f34636..9259d07 100644 --- a/README_RUNPOD.md +++ b/README_RUNPOD.md @@ -231,9 +231,10 @@ Useful `sweep.sh` env overrides: `MODELS`, `BLENDS`, `BASELINE`, `MAX_FRAMES`, Klein regime (`scripts/sweep-klein.sh`, drives `dltb-klein`): `MODEL` (default `flux2-klein-9b`; `4b` is ungated), `BLEND` (default `0.1`), -`GUIDANCES` (default "2.0 4.0"), plus the shared `CLIP`/`MAX_FRAMES`/ -`TAIL_FRAMES`/`TAIL_MODES`/`SAVE_EVERY`/`EXTRA_ARGS`/`DRY_RUN`/ -`SKIP_GPU_CHECK`. Single-model, so no hf-cache eviction between runs. +`STEPS` (default "2 8", bracketing the default 4), plus the shared +`CLIP`/`MAX_FRAMES`/`TAIL_FRAMES`/`TAIL_MODES`/`SAVE_EVERY`/`EXTRA_ARGS`/ +`DRY_RUN`/`SKIP_GPU_CHECK`. Single-model, so no hf-cache eviction between +runs. ## 5. Model cache management diff --git a/scripts/sweep-klein.sh b/scripts/sweep-klein.sh index 2345005..fba0d20 100755 --- a/scripts/sweep-klein.sh +++ b/scripts/sweep-klein.sh @@ -19,10 +19,14 @@ # as trained, the loop converges toward "maximally # weathered" instead of melting - directed attractor # vs. undirected collapse. -# 3. Guidance probes : the mild prompt at --guidance-scale 2.0 / 4.0. -# Experimental: klein's guidance is an embedded -# conditioning signal (card default 1.0); higher values -# may strengthen prompt adherence per pass. +# 3. Steps probes : the mild prompt at --num-inference-steps 2 / 8 +# (bracketing the klein card default of 4, which the +# ladder legs already run). Steps are the one direct +# per-pass edit-intensity knob that actually reaches +# klein -- real scheduler steps, no distillation +# short-circuit. (Supersedes the guidance probes, +# removed 2026-09-13: --guidance-scale > 1 is provably +# inert for step-wise distilled klein; see NOTES.md.) # # Each prompt gets its own --output-dir subtree (output//prompt-/), # because dltb-klein's directory tag encodes only mode/blend/tails - without @@ -42,7 +46,8 @@ # TAIL_FRAMES frames per tail (default 60; 0 = no tails) # TAIL_MODES tail scenario list (default freeze; "freeze,free" etc.) # SAVE_EVERY save every Nth frame (default 10) -# GUIDANCES guidance probe values (default "2.0 4.0"; empty = skip) +# STEPS steps-probe values (default "2 8", bracketing the default 4; +# empty = skip the probe) # EXTRA_ARGS extra flags, word-split, appended to every run # DRY_RUN=1 print commands without executing anything # SKIP_GPU_CHECK=1 bypass the CUDA preflight @@ -65,7 +70,7 @@ MAX_FRAMES="${MAX_FRAMES-300}" TAIL_FRAMES="${TAIL_FRAMES:-60}" TAIL_MODES="${TAIL_MODES:-freeze}" SAVE_EVERY="${SAVE_EVERY:-10}" -GUIDANCES="${GUIDANCES-2.0 4.0}" +STEPS="${STEPS-2 8}" EXTRA_ARGS="${EXTRA_ARGS:-}" DRY_RUN="${DRY_RUN:-0}" @@ -80,9 +85,9 @@ PROMPT_TABLE=( "weathering|add more weathering, moss and water stains to the stone" ) -# Guidance probes reuse the mild-enhancement prompt. -GUIDANCE_SLUG="enhance-slight" -GUIDANCE_PROMPT="slightly enhance the fine details" +# Steps probes reuse the mild-enhancement prompt. +PROBE_SLUG="enhance-slight" +PROBE_PROMPT="slightly enhance the fine details" # -------------------------------------------------------------- preflight ---- if [[ "$DRY_RUN" != "1" ]]; then @@ -129,7 +134,7 @@ run() { } log "sweep-klein: model=$MODEL clip=$CLIP blend=$BLEND" -log "sweep-klein: max_frames=${MAX_FRAMES:-} tail=${TAIL_FRAMES}x${TAIL_MODES} guidances='${GUIDANCES:-}'" +log "sweep-klein: max_frames=${MAX_FRAMES:-} tail=${TAIL_FRAMES}x${TAIL_MODES} steps='${STEPS:-}'" log "sweep-klein: log=$LOG" # ------------------------------------------------------- 1+2. prompt ladder ---- @@ -145,14 +150,14 @@ for entry in "${PROMPT_TABLE[@]}"; do run "${args[@]}" ${EXTRA_ARGS:+$EXTRA_ARGS} done -# ------------------------------------------------------- 3. guidance probe ---- -if [[ -n "${GUIDANCES:-}" ]]; then - for G in $GUIDANCES; do +# --------------------------------------------------------- 3. steps probe ---- +if [[ -n "${STEPS:-}" ]]; then + for N in $STEPS; do log "" - log "########## guidance probe: '$GUIDANCE_SLUG' @ guidance=$G ##########" + log "########## steps probe: '$PROBE_SLUG' @ num-inference-steps=$N ##########" run ${common[@]+"${common[@]}"} \ - --prompt "$GUIDANCE_PROMPT" --guidance-scale "$G" \ - --output-dir "output/$MODEL/guidance$G" ${EXTRA_ARGS:+$EXTRA_ARGS} + --prompt "$PROBE_PROMPT" --num-inference-steps "$N" \ + --output-dir "output/$MODEL/steps$N" ${EXTRA_ARGS:+$EXTRA_ARGS} done fi diff --git a/src/dltb/args.py b/src/dltb/args.py index 6438c31..310f967 100644 --- a/src/dltb/args.py +++ b/src/dltb/args.py @@ -28,10 +28,11 @@ def add_pass_args(p: argparse.ArgumentParser) -> None: p.add_argument("--num-inference-steps", type=int, default=4, help="Denoise schedule length (all models distilled for 1-4 steps)") p.add_argument("--guidance-scale", type=float, default=None, - help="Override the model's default guidance scale. Experimental: " - "klein takes guidance as an embedded conditioning signal " - "(card default 1.0), so raising it may strengthen prompt " - "adherence per pass.") + help="Override the model's default guidance scale (per-model " + "defaults; klein 1.0). NOTE: values > 1 are IGNORED by the " + "step-distilled klein models (no CFG, no guidance embedding; " + "diffusers warns per pass) -- dltb-klein warns once at " + "startup. See NOTES.md.") p.add_argument("--prompt", default="", help="Optional text prompt. A faithful description acts as a " "semantic anchor (analogue of DLSS 5's artistic-direction " diff --git a/src/dltb/klein.py b/src/dltb/klein.py index 9f84bd8..9066b04 100644 --- a/src/dltb/klein.py +++ b/src/dltb/klein.py @@ -11,9 +11,14 @@ flux-schnell). Consequences for the loop: --prompt the de-facto per-pass edit-strength knob: an empty or neutral prompt preserves, an edit instruction compounds every pass (scripts/sweep-klein.sh walks that ladder) - --guidance-scale klein takes guidance as an embedded conditioning signal - (model card default 1.0); raising it may strengthen prompt - adherence per pass. Experimental. + --num-inference-steps real scheduler steps (klein card default 4) -- the + direct per-pass edit-intensity knob alongside the prompt + (sweep-klein probes 2 / 8 around it) + + Guidance: --guidance-scale > 1 is INERT for klein -- step-wise distilled + models get no CFG (hard-disabled) and no guidance embedding; diffusers + warns and ignores the value. dltb-klein prints its own warning; see + NOTES.md, "FLUX.2 klein: --guidance-scale is inert". This runs the same loop topologies as dltb-continuous (anchored boil test / stateful blend with optical-flow reprojection / failure tails), restricted to @@ -84,6 +89,15 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: def run(args: argparse.Namespace) -> None: + if args.guidance_scale is not None and args.guidance_scale > 1.0: + print( + f"dltb-klein: WARNING: --guidance-scale {args.guidance_scale:g} > 1 is " + "ignored by step-wise distilled klein models (CFG is hard-disabled " + "and there is no guidance embedding); the run would be identical to " + "the default-guidance one. See NOTES.md, 'FLUX.2 klein: " + "--guidance-scale is inert'.", + flush=True, + ) # Identical loop to dltb-continuous; only the argument surface above # differs (model restriction, klein blend default, guidance emphasis). run_continuous(args) -- 2.51.2