From 8f02ca6b0d847360bb5515180bcace304b8b5c8f Mon Sep 17 00:00:00 2001 From: Refinement Systems Date: Sun, 13 Sep 2026 08:54:49 +0200 Subject: [PATCH] flux klein upgrade --- AGENTS.md | 18 ++++-- NOTES.md | 29 +++++---- pyproject.toml | 1 + scripts/sweep-klein.sh | 77 +++++++++++++++++++---- src/dltb/__init__.py | 8 ++- src/dltb/assemble.py | 138 +++++++++++++++++++++++++++++++++++++++++ src/dltb/continuous.py | 66 +++++++++++++++----- src/dltb/klein.py | 115 ++++++++++++++++++++++++++++------ src/dltb/models.py | 5 +- 9 files changed, 386 insertions(+), 71 deletions(-) create mode 100644 src/dltb/assemble.py diff --git a/AGENTS.md b/AGENTS.md index 77e8d4d..6929111 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -59,7 +59,8 @@ in `pyproject.toml`): | `args.py` | argparse flag groups shared by the tools | | `oneshot.py` / `iterate.py` | single pass / free-running self-iteration | | `continuous.py` | video loop: anchored boil test vs stateful blend, reproject, freeze/free/black tails | -| `klein.py` | klein-restricted wrapper; delegates to `continuous.run()` | +| `assemble.py` | CPU-only local mp4 assembly from a run's saved frames (split workflow; `dltb-assemble`) | +| `klein.py` | klein-restricted wrapper; dual-ref conditioning hook; delegates to `continuous.run()` | Key invariants: @@ -68,11 +69,16 @@ Key invariants: - Model differences are encoded in `ModelSpec` (`uses_strength`, `pass_size`, `gated`, dtype/variant), not in `if model == ...` at call sites. Adding a model = adding a `MODELS` entry (plus docs/tables in README.md). -- `dltb-klein` shares `continuous.run()` wholesale — only its argument surface - (model restriction, blend default 0.1, guidance warning) differs. -- Run-directory tags encode only mode/blend/tails. Runs differing in other - knobs (prompt, strength, steps) must use `--output-dir` subtrees or they - silently overwrite earlier results (see how `sweep.sh` does it). +- `dltb-klein` shares `continuous.run()` wholesale but customizes conditioning + via the `run(args, make_conditioning=...)` hook: `--conditioning dual-ref` + passes `[state, frame]` as two clean reference images instead of the pixel + blend (`--conditioning blend`, the default, exercises the default + `_blend_conditioning`). Its argument surface also restricts models, defaults + blend to 0.1, and warns about inert guidance. +- Run-directory tags encode only mode/blend-or-conditioning/tails + (dual-ref: `..._dualref[-norepro]_tails…`). Runs differing in other knobs + (prompt, strength, steps, `--ref-order`) must use `--output-dir` subtrees or + they silently overwrite earlier results (see how `sweep.sh` does it). - User-facing errors fail fast via `raise SystemExit("message")`. - Most modules and scripts carry the ISC license header; keep it on new files in `src/dltb/` and `scripts/`. diff --git a/NOTES.md b/NOTES.md index 942e979..2355c45 100644 --- a/NOTES.md +++ b/NOTES.md @@ -239,10 +239,14 @@ not break the tool). The 2-frame hash A/B remains optional and only worth doing after a diffusers upgrade. Next-experiment sketch for klein's control problem: dual-reference conditioning, see the last section. -## Sketch: klein dual-reference conditioning (`--conditioning dual-ref`) +## Klein dual-reference conditioning (`--conditioning dual-ref`) -**Status: DESIGN ONLY (2026-09-13), not implemented.** The candidate fix for -klein's control problem, replacing the pixel-blend proxy. +**Status: IMPLEMENTED 2026-09-13.** The candidate fix for klein's control +problem, replacing the pixel-blend proxy. The design below is what landed +(one deviation: `continuous.run` takes a `make_conditioning` factory that +returns `(combine, tail_source)` function pairs, rather than a single +`condition(...)` callable, because tails need their own source construction). +Runs validating it (prompt ladder + order A/B) are still pending. **Why:** klein's calibration trouble (too weak at blend 0.6–0.8, "melting" at 0.1 — see the calibration section) is plausibly an artifact of pixel-blending @@ -266,11 +270,10 @@ state and the fresh frame can both be conditioning inputs: list of two PIL images flows straight through. `--width/--height` still set the output canvas; references are resized/packed per-image by the pipeline (our 768² frames are under the 1 MP auto-resize cap). -- `klein.py` needs its own stateful loop for dual-ref (it currently delegates - to `continuous.run`, whose blend is baked in). Either fork the loop, or — - cleaner — generalize `continuous.run` to accept a - `condition(carried, new_frame) -> source` callable, with the current - blend/reproject logic as the default implementation. +- `continuous.run` was generalized (no fork): it accepts + `make_conditioning(args, estimate_flow, warp) -> (combine, tail_source)`, + with the former blend/reproject logic as the default + (`_blend_conditioning`); `klein.py` supplies `_dual_ref_conditioning`. - Reprojection: probably UNNECESSARY in dual-ref (the fresh frame is an explicit reference; the model aligns content, not pixel coordinates) — but keep it probeable: warping the carried reference may still help temporal @@ -291,7 +294,7 @@ state and the fresh frame can both be conditioning inputs: - Interaction with the steps probe: re-run the steps axis under dual-ref — intensity may interact with conditioning strength. -**Suggested first runs (once implemented):** +**Suggested first runs:** uv run dltb-klein --model flux2-klein-4b --input input/video_cropped.mp4 \ --conditioning dual-ref --prompt "slightly enhance the fine details" \ @@ -301,8 +304,8 @@ state and the fresh frame can both be conditioning inputs: uv run dltb-klein --model flux2-klein-4b --input input/video_cropped.mp4 \ --conditioning dual-ref --ref-order state-first --max-frames 2 -Then add a `SMOKE_KLEIN=1` leg to `scripts/smoke.sh` (2 frames + 1 tail frame, -flux2-klein-4b; off by default because of the 15 GB download) and a -`CONDITIONING=dual-ref` toggle to `scripts/sweep-klein.sh` so the prompt -ladder + steps probe re-run under the new conditioning. +Remaining follow-ups: add a `SMOKE_KLEIN=1` leg to `scripts/smoke.sh` (2 frames ++ 1 tail frame, flux2-klein-4b; off by default because of the 15 GB download); +the `CONDITIONING=dual-ref REF_ORDER={state-first,frame-first}` toggle in +`scripts/sweep-klein.sh` is done (it also swaps in role-naming prompts). diff --git a/pyproject.toml b/pyproject.toml index 9e6e99c..04eac53 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,6 +26,7 @@ dltb-oneshot = "dltb.oneshot:main" dltb-iterate = "dltb.iterate:main" dltb-continuous = "dltb.continuous:main" dltb-klein = "dltb.klein:main" +dltb-assemble = "dltb.assemble:main" [build-system] requires = ["uv_build>=0.12.7,<0.13.0"] diff --git a/scripts/sweep-klein.sh b/scripts/sweep-klein.sh index fba0d20..5a2a49c 100755 --- a/scripts/sweep-klein.sh +++ b/scripts/sweep-klein.sh @@ -9,6 +9,16 @@ # the blend alpha (~linear, no constant rewrite bias like the img2img models), # and the PROMPT is the de-facto per-pass edit-strength knob. # +# CONDITIONING (stateful mode): +# blend (default) carried state and fresh frame pixel-blended into ONE +# reference; --anchor-blend applies (BLEND env). +# dual-ref carried state and fresh frame as TWO separate clean references -- +# no ghosted blend for the editor to parse; state/anchor weighting +# is done by the model. BLEND is ignored; REF_ORDER picks [P,N] +# (state-first, default) or [N,P]. The prompt table switches to +# role-naming prompts ("image 2 is the current frame; ...", which +# assumes state-first order -- swap the roles if REF_ORDER=frame-first). +# # What this sweep runs (all --mode stateful, one source pass per prompt): # # 1. Prompt ladder : neutral -> slight -> photo -> dramatic enhancement, @@ -30,21 +40,29 @@ # # Each prompt gets its own --output-dir subtree (output//prompt-/), # because dltb-klein's directory tag encodes only mode/blend/tails - without -# the subtree, prompt runs would silently overwrite each other. +# the subtree, prompt runs would silently overwrite each other. (Dual-ref runs +# are tagged ..._dualref[-norepro] instead of ..._stateful-a.) # # Usage (from any directory inside the repo): # scripts/sweep-klein.sh # DRY_RUN=1 scripts/sweep-klein.sh # MODEL=flux2-klein-4b BLEND=0.2 scripts/sweep-klein.sh +# CONDITIONING=dual-ref scripts/sweep-klein.sh +# CONDITIONING=dual-ref REF_ORDER=frame-first scripts/sweep-klein.sh # # Environment overrides: # MODEL klein model key (default flux2-klein-9b; 4b is ungated) # CLIP source video (default input/video_cropped.mp4) +# CONDITIONING blend | dual-ref (default blend) +# REF_ORDER state-first | frame-first (default state-first; dual-ref only) # BLEND anchor-blend for all runs (default 0.1 - klein's active range -# is far below the img2img models; try 0.03 0.05 0.2 manually) +# is far below the img2img models; try 0.03 0.05 0.2 manually; +# IGNORED under CONDITIONING=dual-ref) # MAX_FRAMES source frames per run (default 300; empty = whole clip) # TAIL_FRAMES frames per tail (default 60; 0 = no tails) -# TAIL_MODES tail scenario list (default freeze; "freeze,free" etc.) +# TAIL_MODES tail scenario list (default freeze; "freeze,free" etc. +# NOTE: under dual-ref, free = single-reference regeneration +# from state alone) # SAVE_EVERY save every Nth frame (default 10) # STEPS steps-probe values (default "2 8", bracketing the default 4; # empty = skip the probe) @@ -65,6 +83,8 @@ cd "$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" MODEL="${MODEL:-flux2-klein-9b}" CLIP="${CLIP:-input/video_cropped.mp4}" +CONDITIONING="${CONDITIONING:-blend}" +REF_ORDER="${REF_ORDER:-state-first}" BLEND="${BLEND:-0.1}" MAX_FRAMES="${MAX_FRAMES-300}" TAIL_FRAMES="${TAIL_FRAMES:-60}" @@ -74,20 +94,46 @@ STEPS="${STEPS-2 8}" EXTRA_ARGS="${EXTRA_ARGS:-}" DRY_RUN="${DRY_RUN:-0}" +case "$CONDITIONING" in + blend|dual-ref) ;; + *) echo "sweep-klein: CONDITIONING must be blend or dual-ref (got '$CONDITIONING')" >&2; exit 1 ;; +esac +case "$REF_ORDER" in + state-first|frame-first) ;; + *) echo "sweep-klein: REF_ORDER must be state-first or frame-first (got '$REF_ORDER')" >&2; exit 1 ;; +esac + # ---------------------------------------------------------------- prompts ---- # slug|prompt pairs. The slug becomes the output subdirectory; keep slugs short, # lowercase, hyphenated. Empty prompt = preservation baseline (no flag passed). -PROMPT_TABLE=( - "neutral|" - "enhance-slight|slightly enhance the fine details" - "enhance-photo|enhance details and lighting, make it photorealistic" - "enhance-dramatic|dramatically enhance every texture and surface detail" - "weathering|add more weathering, moss and water stains to the stone" -) +if [[ "$CONDITIONING" == "dual-ref" ]]; then + # Role-naming prompts. Phrasing assumes state-first order (image 1 = carried + # state/appearance reference, image 2 = current frame/content target); + # swap the roles if REF_ORDER=frame-first. + PROMPT_TABLE=( + "neutral|" + "enhance-slight|image 2 is the current frame; keep the appearance of image 1, slightly enhancing fine details" + "enhance-photo|image 2 is the current frame; keep the appearance of image 1, enhancing details and lighting to look photorealistic" + "enhance-dramatic|image 2 is the current frame; keep the appearance of image 1, dramatically enhancing every texture and surface detail" + "weathering|image 2 is the current frame; keep the appearance of image 1, adding more weathering, moss and water stains to the stone" + ) +else + PROMPT_TABLE=( + "neutral|" + "enhance-slight|slightly enhance the fine details" + "enhance-photo|enhance details and lighting, make it photorealistic" + "enhance-dramatic|dramatically enhance every texture and surface detail" + "weathering|add more weathering, moss and water stains to the stone" + ) +fi # Steps probes reuse the mild-enhancement prompt. PROBE_SLUG="enhance-slight" -PROBE_PROMPT="slightly enhance the fine details" +if [[ "$CONDITIONING" == "dual-ref" ]]; then + PROBE_PROMPT="image 2 is the current frame; keep the appearance of image 1, slightly enhancing fine details" +else + PROBE_PROMPT="slightly enhance the fine details" +fi # -------------------------------------------------------------- preflight ---- if [[ "$DRY_RUN" != "1" ]]; then @@ -115,7 +161,12 @@ LOG="output/sweep_klein_$(date -u +%Y%m%d-%H%M%S).log" # Flags shared by every run. common=(--model "$MODEL" --input "$CLIP" --save-every "$SAVE_EVERY" - --mode stateful --anchor-blend "$BLEND") + --mode stateful --conditioning "$CONDITIONING") +if [[ "$CONDITIONING" == "dual-ref" ]]; then + common+=(--ref-order "$REF_ORDER") # --anchor-blend is ignored under dual-ref +else + common+=(--anchor-blend "$BLEND") +fi if [[ -n "$MAX_FRAMES" ]]; then common+=(--max-frames "$MAX_FRAMES"); fi if [[ "$TAIL_FRAMES" -gt 0 ]]; then common+=(--tail-frames "$TAIL_FRAMES" --tail-modes "$TAIL_MODES") @@ -133,7 +184,7 @@ run() { uv run dltb-klein "$@" 2>&1 | tee -a "$LOG" } -log "sweep-klein: model=$MODEL clip=$CLIP blend=$BLEND" +log "sweep-klein: model=$MODEL clip=$CLIP conditioning=$CONDITIONING ref_order=$REF_ORDER blend=$BLEND" log "sweep-klein: max_frames=${MAX_FRAMES:-} tail=${TAIL_FRAMES}x${TAIL_MODES} steps='${STEPS:-}'" log "sweep-klein: log=$LOG" diff --git a/src/dltb/__init__.py b/src/dltb/__init__.py index 6ede318..34a0e5c 100644 --- a/src/dltb/__init__.py +++ b/src/dltb/__init__.py @@ -13,5 +13,11 @@ Tools (installed as console scripts; see pyproject.toml): dltb-iterate free-running image self-iteration + mp4 timelapse dltb-continuous video pipeline simulation (anchored/stateful, tails) dltb-klein the continuous loop restricted to the FLUX.2 klein editors - (no --strength; prompt = per-pass edit strength) + (no --strength; prompt = per-pass edit strength; + --conditioning blend|dual-ref) + +CPU-only local post-processing (no torch/CUDA; run on saved frames): + + analyze_drift.py drift metrics for a frames dir + assemble.py dltb-assemble: encode mp4s from a run's saved frames """ diff --git a/src/dltb/assemble.py b/src/dltb/assemble.py new file mode 100644 index 0000000..d3e8c0b --- /dev/null +++ b/src/dltb/assemble.py @@ -0,0 +1,138 @@ +# Permission to use, copy, modify, and/or distribute this software for +# any purpose with or without fee is hereby granted. +# +# THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL +# WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES +# OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE +# FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN +# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF +# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. + +"""dltb-assemble: encode mp4s from a run's saved PNG frames. + +CPU-only post-processing for the split workflow: frames are computed on a +CUDA pod (dltb-continuous / dltb-klein / dltb-iterate; --save-every 1 saves +the full sequence), the run directory is copied home, and the videos are +encoded locally -- no torch, no CUDA (the same local-only idea as +analyze_drift.py). The pod already encodes its own mp4s as it goes, so this +tool is for re-encoding: different --fps, --every subsampling, or frames +copied without the videos. + +Frame layout as written by the tools: + + /frames/frame_0001.png ... main loop + /frames/ frame_0001.png ... tail phases -- one prefix per + --tail-modes entry (the space + comes from continuous.py's + per-mode save tag) + /end_state.png, frame_0000_source.png (never part of any video) + +The main frames become /assembled.mp4; each tail prefix becomes +/assembled_tail_.mp4. Accepts either the run directory or the +frames directory itself. + +CAUTION: frames saved with --save-every N > 1 form a timelapse -- at any +fps the result plays N times faster than the source run. The detected +stride is printed per video; pass a proportionally lower --fps (e.g. +--fps 3 for stride 10 saved from 30 fps source) for real-time pacing. + +Example: + uv run dltb-assemble output_flux2_klein_4b/ --fps 30 + uv run dltb-assemble output_sdxl_turbo//frames --fps 12 --every 2 + python3 src/dltb/assemble.py --fps 24 # no install +""" + +from __future__ import annotations + +import argparse +import re +from pathlib import Path + +# continuous.py saves the main loop as frame_NNNN.png and each tail phase as +# " frame_NNNN.png" (the tag passed to one_pass is f"{mode} "). +_FRAME_RE = re.compile(r"^(?:(?P[a-z]+) )?frame_(?P\d+)\.png$") + + +def discover_groups(frames_dir: Path) -> dict[str | None, list[tuple[int, Path]]]: + """Group frame files by tail-mode prefix: {mode | None: [(idx, path)]}, + each list sorted by frame index (numeric -- robust past 9999 frames, + unlike a plain filename sort). None is the main loop.""" + groups: dict[str | None, list[tuple[int, Path]]] = {} + for p in frames_dir.iterdir(): + m = _FRAME_RE.match(p.name) + if m: + groups.setdefault(m.group("mode"), []).append((int(m.group("idx")), p)) + for frames in groups.values(): + frames.sort(key=lambda ip: ip[0]) + return groups + + +def uniform_stride(indices: list[int]) -> int | None: + """The common index gap if it is uniform (e.g. 10 for --save-every 10), + else None.""" + if len(indices) < 2: + return 1 + gaps = {b - a for a, b in zip(indices, indices[1:])} + return gaps.pop() if len(gaps) == 1 else None + + +def encode(frames: list[Path], video_path: Path, fps: float) -> None: + """libx264-encode the given frame files (CPU-only; imageio-ffmpeg's + bundled binary, the same encoder the pod-side tools use).""" + import imageio.v2 as imageio + + try: + with imageio.get_writer(video_path, fps=fps, codec="libx264") as w: + for f in frames: + w.append_data(imageio.imread(f)) + except Exception as e: # encoding is this tool's whole job: fail loudly + raise SystemExit(f"video encoding failed for {video_path}: {e}") + + +def main(argv: list[str] | None = None) -> None: + p = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + p.add_argument("run_dir", type=Path, + help="Run directory (containing frames/) or the frames " + "directory itself") + p.add_argument("--fps", type=float, default=30.0, + help="Encoding framerate (fractional ok, e.g. 29.97). With " + "--save-every N frames this is the TIMELAPSE playback " + "speed -- see the stride note printed per video") + p.add_argument("--every", type=int, default=1, + help="Use every Nth of the found frames (like " + "analyze_drift's --every)") + args = p.parse_args(argv) + + frames_dir = args.run_dir + if (frames_dir / "frames").is_dir(): + frames_dir = frames_dir / "frames" + if not frames_dir.is_dir(): + raise SystemExit(f"{args.run_dir} is neither a run directory nor a frames " + "directory") + + groups = discover_groups(frames_dir) + if not groups: + raise SystemExit(f"no frame_*.png / ' frame_*.png' files in {frames_dir}") + + every = max(1, args.every) + for mode, frames in sorted(groups.items(), + key=lambda kv: (kv[0] is not None, kv[0] or "")): + selected = frames[::every] + indices = [idx for idx, _ in selected] + stride = uniform_stride(indices) + video = frames_dir.parent / ("assembled.mp4" if mode is None + else f"assembled_tail_{mode}.mp4") + print(f"{'main' if mode is None else f'tail {mode!r}'}: " + f"{len(selected)} frames -> {video}" + + (f" (frame stride {stride}: timelapse, {stride}x faster than " + f"the source run)" if stride and stride > 1 else "")) + encode([path for _, path in selected], video, args.fps) + + print("Done.") + + +if __name__ == "__main__": + main() diff --git a/src/dltb/continuous.py b/src/dltb/continuous.py index fd55063..e007f22 100644 --- a/src/dltb/continuous.py +++ b/src/dltb/continuous.py @@ -45,9 +45,18 @@ the freshly rendered frame from the engine. black : engine submits BLACK frames; source = blend(P_{n-1}, black) (renderer-crash scenario; a decay driver, NOT free-running) -NOTE - FLUX.2 klein models are reference-image editors (regenerate from pure -noise, NO --strength). Refinement idea for klein stateful mode: pass -[P_{n-1}, N] as two separate reference images instead of pixel-blending. + CONDITIONING STRATEGIES: run() accepts make_conditioning, a factory + (args, estimate_flow, warp) -> (combine, tail_source), so tools can replace + HOW carried state and fresh frame become the model input: + + combine(carried, new_frame, prev_source) -> source (main loop) + tail_source(mode, current, last_source, black) -> source (tail phases) + + The default is the pixel-blend above (_blend_conditioning). dltb-klein uses + the hook for dual-reference conditioning ([P, N] as two clean reference + images instead of one blended one; see NOTES.md, "klein dual-reference + conditioning"). When args.conditioning == "dual-ref" the output tag is + dualref[-norepro] (no blend component). Example: uv run dltb-continuous --model flux-schnell --input clip.mp4 --mode stateful \\ @@ -116,7 +125,31 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: return p.parse_args(argv) -def run(args: argparse.Namespace) -> None: +def _blend_conditioning(args, estimate_flow, warp): + """Default conditioning: pixel-blend the (optionally reprojected) carried + state with the fresh frame into ONE model input.""" + from PIL import Image + + def combine(carried, new_frame, prev_source): + c = carried + if estimate_flow is not None and prev_source is not None: + c = warp(c, estimate_flow(prev_source, new_frame)) + return Image.blend(c, new_frame, args.anchor_blend) + + def tail_source(mode, current, last_source, black): + if mode == "free": + return current # anchor dropped + if mode == "black": + return Image.blend(current, black, args.anchor_blend) + return Image.blend(current, last_source, args.anchor_blend) # freeze + + return combine, tail_source + + +def run(args: argparse.Namespace, make_conditioning=None) -> None: + """Run the video loop. make_conditioning (optional) is a factory + (args, estimate_flow, warp) -> (combine, tail_source) replacing the + default pixel-blend conditioning; see the module docstring.""" spec = MODELS[args.model] width, height = resolve_geometry(spec, args.width, args.height) check_requirements(spec, args.num_inference_steps, args.strength) @@ -127,9 +160,13 @@ def run(args: argparse.Namespace) -> None: import numpy as np from PIL import Image + dual_ref = getattr(args, "conditioning", "blend") == "dual-ref" mode_tag = args.mode if args.mode == "stateful": - mode_tag = f"stateful-a{args.anchor_blend:g}" + if dual_ref: + mode_tag = "dualref" # no blend component + else: + mode_tag = f"stateful-a{args.anchor_blend:g}" if not args.reproject: mode_tag += "-norepro" if args.tail_frames: @@ -142,6 +179,11 @@ def run(args: argparse.Namespace) -> None: if use_reproject: estimate_flow, warp = make_reprojector() + if make_conditioning is None: + combine, tail_source = _blend_conditioning(args, estimate_flow, warp) + else: + combine, tail_source = make_conditioning(args, estimate_flow, warp) + frames_dir = out_dir / "frames" frames_dir.mkdir(parents=True, exist_ok=True) @@ -154,6 +196,7 @@ def run(args: argparse.Namespace) -> None: if total in (None, float("inf")): total = "?" print(f"source: {args.input} | fps={src_fps} | frames={total} | mode={args.mode} | " + f"conditioning={'dual-ref' if dual_ref else 'blend'} | " f"reproject={'on' if use_reproject else 'off'} | " f"tails={args.tail_frames}x{','.join(args.tail_modes) if args.tail_frames else 'off'}") @@ -194,11 +237,7 @@ def run(args: argparse.Namespace) -> None: if args.mode == "anchored" or current is None: source = new_frame # independent per frame -> boil test else: # stateful: carry processed state forward - carried = current - if use_reproject and prev_source is not None: - flow = estimate_flow(prev_source, new_frame) - carried = warp(current, flow) # reproject history (motion vectors) - source = Image.blend(carried, new_frame, args.anchor_blend) + source = combine(current, new_frame, prev_source) prev_source = new_frame one_pass(source, n, writer) finally: @@ -220,12 +259,7 @@ def run(args: argparse.Namespace) -> None: tail_writer = imageio.get_writer(tail_video, fps=src_fps, codec="libx264") try: for t in range(1, args.tail_frames + 1): - if mode == "free": - source = current # anchor dropped - elif mode == "black": - source = Image.blend(current, black, args.anchor_blend) - else: # freeze: re-submit last real frame; zero flow -> no warp - source = Image.blend(current, last_source, args.anchor_blend) + source = tail_source(mode, current, last_source, black) one_pass(source, t, tail_writer, tag=f"{mode} ") finally: tail_writer.close() diff --git a/src/dltb/klein.py b/src/dltb/klein.py index 9066b04..99987b4 100644 --- a/src/dltb/klein.py +++ b/src/dltb/klein.py @@ -1,3 +1,14 @@ +# Permission to use, copy, modify, and/or distribute this software for +# any purpose with or without fee is hereby granted. +# +# THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL +# WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES +# OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE +# FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY +# DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN +# AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT +# OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. + """dltb-klein: video pipeline simulation with the FLUX.2 klein editors. Klein (FLUX.2-klein-4B / -9B) is a reference-image EDITOR, not a partial-noise @@ -7,7 +18,8 @@ alpha (roughly linear -- no constant rewrite bias like sd/sdxl-turbo and flux-schnell). Consequences for the loop: --anchor-blend klein's active range is far below the img2img models - (default here 0.1; try 0.03 / 0.05 / 0.2) + (default here 0.1; try 0.03 / 0.05 / 0.2). Applies to + --conditioning blend only; IGNORED under dual-ref. --prompt the de-facto per-pass edit-strength knob: an empty or neutral prompt preserves, an edit instruction compounds every pass (scripts/sweep-klein.sh walks that ladder) @@ -20,18 +32,37 @@ flux-schnell). Consequences for the loop: warns and ignores the value. dltb-klein prints its own warning; see NOTES.md, "FLUX.2 klein: --guidance-scale is inert". -This runs the same loop topologies as dltb-continuous (anchored boil test / -stateful blend with optical-flow reprojection / failure tails), restricted to -the klein pair, with the klein-appropriate blend default. It shares -continuous.run() -- only argument defaults and model choice differ. +CONDITIONING (--conditioning, stateful mode only): + + blend (default) pixel-blend the (reprojected) carried state and the fresh + frame into ONE reference image: image = (1-a)*R(P_{n-1}) + a*N_n + + dual-ref pass the carried state and the fresh frame as TWO SEPARATE clean + reference images: image = [R(P_{n-1}), N_n]. Flux2KleinPipeline + accepts a list natively (each reference is preprocessed and packed + onto the sequence axis), and each reference stays in-distribution -- + no ghosted blend mush for the editor to parse. The model itself + decides how to weight state vs. anchor through attention, which is + categorically closer to DLSS 5's multi-input conditioning than the + pixel blend. --anchor-blend is ignored in this mode. -Refinement idea: pass [P_{n-1}, N] as two separate reference images instead -of pixel-blending. + --ref-order {state-first,frame-first} (default state-first) selects + [P, N] vs [N, P]; whether klein treats reference order as + "primary vs target" is an open question -- A/B it (see NOTES.md). + + Tails under dual-ref: freeze = [P, last_source], + free = [P] ALONE (single-reference regeneration from state -- the + pure buffer-echo case), black = [P, black]. + + Reprojection stays probeable: with --reproject (default) the + CARRIED reference is warped by the source-frame flow before + pairing; A/B with --no-reproject. Example: uv run dltb-klein --model flux2-klein-4b --input clip.mp4 \\ - --prompt "slightly enhance the fine details" \\ - --tail-frames 60 --tail-modes freeze + --conditioning dual-ref \\ + --prompt "image 2 is the current frame; keep the appearance of image 1" \\ + --tail-frames 60 --tail-modes freeze,free """ from __future__ import annotations @@ -45,6 +76,31 @@ from .models import model_key KLEIN_MODELS = ("flux2-klein-4b", "flux2-klein-9b") +def _dual_ref_conditioning(args, estimate_flow, warp): + """Klein dual-reference conditioning: carried state and fresh frame as two + separate clean references (a list flows through imaging.run_pass unchanged). + If reprojection is on, the CARRIED reference is warped by the source-frame + flow before pairing; the fresh frame is never warped.""" + + def ordered(state_img, frame_img): + if args.ref_order == "state-first": + return [state_img, frame_img] + return [frame_img, state_img] + + def combine(carried, new_frame, prev_source): + c = carried + if estimate_flow is not None and prev_source is not None: + c = warp(c, estimate_flow(prev_source, new_frame)) + return ordered(c, new_frame) + + def tail_source(mode, current, last_source, black): + if mode == "free": + return [current] # anchor dropped: single-reference regeneration + return ordered(current, black if mode == "black" else last_source) + + return combine, tail_source + + def parse_args(argv: list[str] | None = None) -> argparse.Namespace: p = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter @@ -61,17 +117,29 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: p.add_argument("--mode", choices=["anchored", "stateful"], default="stateful", help="Loop topology (see module docstring). anchored = " "independent per-frame boil test; stateful = carried " - "state blended with each new frame.") + "state conditioned with each new frame.") + p.add_argument("--conditioning", choices=["blend", "dual-ref"], default="blend", + help="stateful mode only: how carried state + fresh frame become " + "model input. blend = one pixel-blended reference " + "(--anchor-blend applies); dual-ref = two separate clean " + "references, model-weighted (blend ignored)") + p.add_argument("--ref-order", choices=["state-first", "frame-first"], + default="state-first", + help="dual-ref only: reference order [P, N] vs [N, P]; whether " + "klein treats order as primary/target is an open question " + "-- A/B it") p.add_argument("--anchor-blend", type=float, default=0.1, - help="stateful mode: weight alpha of the fresh frame in the blend " - "(1-a)*previous_output + a*new_frame. Klein's active range " - "is far below the img2img models (try 0.03/0.05/0.2); " - "0.0 = free-running, 1.0 = fully re-anchored every frame") + help="stateful+blend only: weight alpha of the fresh frame in " + "the blend (1-a)*previous_output + a*new_frame. Klein's " + "active range is far below the img2img models (try " + "0.03/0.05/0.2); 0.0 = free-running, 1.0 = fully " + "re-anchored every frame. IGNORED under dual-ref.") p.add_argument("--reproject", action=argparse.BooleanOptionalAction, default=True, help="stateful mode only: warp the carried state by optical flow " - "estimated between consecutive source frames before blending " - "(emulates engine motion vectors; needs opencv-python-headless). " - "--no-reproject gives the naive history blend (ghosting).") + "estimated between consecutive source frames before " + "combining (emulates engine motion vectors; needs " + "opencv-python-headless). Under dual-ref only the CARRIED " + "reference is warped. --no-reproject gives the naive variant.") p.add_argument("--max-frames", type=int, default=None, help="Stop after this many SOURCE frames (tail phases, if " "any, come after)") @@ -98,9 +166,16 @@ def run(args: argparse.Namespace) -> None: "--guidance-scale is inert'.", flush=True, ) - # Identical loop to dltb-continuous; only the argument surface above - # differs (model restriction, klein blend default, guidance emphasis). - run_continuous(args) + if args.conditioning == "dual-ref": + if args.anchor_blend != 0.1: + print("dltb-klein: NOTE: --anchor-blend is ignored under " + "--conditioning dual-ref (state/anchor weighting is done by " + "the model, not by pixel blending).", flush=True) + run_continuous(args, make_conditioning=_dual_ref_conditioning) + else: + # Identical loop to dltb-continuous; only the argument surface above + # differs (model restriction, klein blend default, guidance emphasis). + run_continuous(args) def main(argv: list[str] | None = None) -> None: diff --git a/src/dltb/models.py b/src/dltb/models.py index 3ff2488..4c6c426 100644 --- a/src/dltb/models.py +++ b/src/dltb/models.py @@ -20,8 +20,9 @@ between families: flux2klein (Flux2KleinPipeline) reference-image editors: regenerate from pure noise, NO --strength -NOTE - FLUX.2 klein models: refinement idea for stateful loops is to pass -[P_{n-1}, N] as two separate reference images instead of pixel-blending. +NOTE - FLUX.2 klein models: stateful loops can pass [P_{n-1}, N] as two +separate reference images instead of pixel-blending (dltb-klein +--conditioning dual-ref; the pipeline resizes/packs each reference). """ from __future__ import annotations -- 2.51.2