diff --git a/AGENTS.md b/AGENTS.md index 5aebd35..5921c78 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -51,6 +51,8 @@ scripts/smoke-local.sh # single-frame pass per locally-via # model (sd-turbo, sdxl-turbo, klein-4b) uv run dltb-distance # DreamSim perceptual drift/convergence # (SMOKE_DISTANCE=1 adds it to smoke.sh) +uv run dltb-distance-feedback --model sd-turbo --input input_example/test_512.png \ + --anchor-blend 0.3 --frames 200 # static-source feedback loop + chart uv run python src/dltb/analyze_drift.py # CPU-only drift metrics ``` @@ -84,6 +86,7 @@ in `pyproject.toml`): | `assemble.py` | CPU-only local mp4 assembly from a run's saved frames (split workflow; `dltb-assemble`) | | `analyze_distance.py` | DreamSim perceptual drift/convergence over a run's frames (`dltb-distance`); companion to `analyze_drift.py`, needs torch + first-run weights download | | `distance_loop.py` | iterate + measure in one run (`dltb-distance-loop`): free-running loop with a strength-capable model, then DreamSim on the saved frames -> `distance_metrics.csv` + `distance_plot.png`; the pipeline is unloaded and released before DreamSim loads | +| `distance_feedback.py` | continuous' stateful loop with a static source (`dltb-distance-feedback`): one image repeated for `--frames` frames, `--anchor-blend 0` = distance-loop, `1` = repeated fresh passes; saves every frame and every model input blend, then the same three-series DreamSim CSV + plot (input-vs-original is the third series) | | `klein.py` | klein-restricted wrapper; dual-ref conditioning hook; delegates to `continuous.run()` | Key invariants: diff --git a/README.md b/README.md index 49b794f..7b2d13c 100644 --- a/README.md +++ b/README.md @@ -57,6 +57,17 @@ The console scripts share the library code in `src/dltb/` (`models`, `imaging`, `distance_metrics.csv`, and both series are charted to `distance_plot.png` (frame number on X, distance on Y). The diffusion pipeline is unloaded before DreamSim loads, so the two models never share accelerator memory. +- `dltb-distance-feedback` — the same measurement around `dltb-continuous`'s + stateful loop with a *static* source: the "video" is one image repeated for + `--frames` frames, so `source_n = (1-a)*P_{n-1} + a*N` unrolls with no motion + (no reprojection, no tails — flow between identical frames is zero). The + endpoints are free correctness checks: `--anchor-blend 0` is exactly + `dltb-distance-loop` (free-running), `--anchor-blend 1` is repeated + independent passes (boil test); in between, the loop shows how far the model's + carried state drags a run away from a fresh render of its own input. Every + pass's actual model input (the blend) is saved under `inputs/`, and DreamSim + scores three series into `distance_metrics.csv` / `distance_plot.png`: frame + vs. original, frame vs. previous frame, and model input vs. original. ## Documentation @@ -126,7 +137,8 @@ https://huggingface.co/black-forest-labs/FLUX.2-klein-9B, then set an `HF_TOKEN` environment variable with an access token from https://huggingface.co/settings/tokens (read-only is enough). -`dltb-distance` and `dltb-distance-loop` download DreamSim weights (~1.2 GB +`dltb-distance`, `dltb-distance-loop` and `dltb-distance-feedback` download +DreamSim weights (~1.2 GB checkpoint zip plus backbone checkpoints, ~2.7 GB for the default ensemble) from GitHub releases into `untracked/models` (gitignored) on first use; `--cache-dir` moves it and `--dreamsim-type` picks a cheaper single-backbone variant. `untracked/` is not @@ -163,14 +175,20 @@ uv run dltb-distance untracked/output_sd_turbo/menu_free-running/frames --every # Self-iteration with the DreamSim drift chart in one run uv run dltb-distance-loop --model sd-turbo --input menu.png --strength 0.4 \ --prompt "a bronze lion sculpture" --iterations 60 + +# Static-source feedback (one image as a 200-frame video, carried state +# blended in): 0.0 = dltb-distance-loop, 1.0 = repeated fresh passes +uv run dltb-distance-feedback --model sd-turbo --input menu.png --strength 0.4 \ + --anchor-blend 0.3 --frames 200 --prompt "a bronze lion sculpture" ``` Runs land in `untracked/output_/_/` (e.g. `menu_free-running/`, `clip_stateful-a0.3_tailsfreeze-free-black60/`): the untouched input frame, -the saved frames, and the output video(s). `dltb-distance-loop` uses a fixed -`distance_s/` tag (no input stem) under the same model tree; since -prompt/steps/seed are not part of that tag, pass `--output-dir` subtrees when -varying them — its `run.json` records the settings of each run. +the saved frames, and the output video(s). `dltb-distance-loop` and +`dltb-distance-feedback` use fixed tags (no input stem) under the same model +tree — `distance_s/` and `feedback_s_a/`; since +prompt/steps/seed are not part of those tags, pass `--output-dir` subtrees when +varying them — their `run.json` records the settings of each run. Run `uv run --help` for all options (strength, steps, seed handling, resolution, prompt, save frequency, video FPS, ...). diff --git a/pyproject.toml b/pyproject.toml index 9a232ef..39737e8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -31,6 +31,7 @@ dltb-klein = "dltb.klein:main" dltb-assemble = "dltb.assemble:main" dltb-distance = "dltb.analyze_distance:main" dltb-distance-loop = "dltb.distance_loop:main" +dltb-distance-feedback = "dltb.distance_feedback:main" [build-system] requires = ["uv_build>=0.12.7,<0.13.0"] diff --git a/scripts/bundle.sh b/scripts/bundle.sh index 551b5f5..fd4cae8 100755 --- a/scripts/bundle.sh +++ b/scripts/bundle.sh @@ -54,7 +54,7 @@ fi for f in pyproject.toml uv.lock \ src/dltb/models.py src/dltb/imaging.py src/dltb/output.py src/dltb/args.py \ src/dltb/oneshot.py src/dltb/iterate.py src/dltb/continuous.py src/dltb/klein.py \ - src/dltb/distance_loop.py; do + src/dltb/distance_loop.py src/dltb/distance_feedback.py; do [[ -f "${stage}/$f" ]] || { echo "bundle: missing $f" >&2; exit 1; } done diff --git a/src/dltb/analyze_distance.py b/src/dltb/analyze_distance.py index 00f01af..ddfe488 100644 --- a/src/dltb/analyze_distance.py +++ b/src/dltb/analyze_distance.py @@ -194,16 +194,22 @@ def load_model(dreamsim_type: str, patch: bool, retries: int, device: str, "fetch the checkpoint into that directory by hand.") -def write_metrics_csv(rows: list[tuple[str, float, float]], out_csv: Path) -> None: - """Write (frame, dreamsim_to_ref, dreamsim_to_prev) rows as CSV. - - Shared with dltb-distance-loop so both tools emit identical columns - (the distance convention: 0.0 = identical, higher = more different). +def write_metrics_csv(rows: list[tuple], out_csv: Path, + extra_header: str | None = None) -> None: + """Write (frame, dreamsim_to_ref, dreamsim_to_prev[, extra]) rows as CSV. + + Shared with dltb-distance-loop and dltb-distance-feedback so the tools + emit identical columns where they measure the same thing (the distance + convention: 0.0 = identical, higher = more different). extra_header names + the 4th column when the rows carry one (compute(extra=...) fills it). """ + header = ["frame", "dreamsim_to_ref", "dreamsim_to_prev"] + if extra_header is not None: + header.append(extra_header) out_csv.parent.mkdir(parents=True, exist_ok=True) with out_csv.open("w", newline="") as fh: w = csv.writer(fh) - w.writerow(["frame", "dreamsim_to_ref", "dreamsim_to_prev"]) + w.writerow(header) w.writerows(rows) @@ -216,20 +222,28 @@ def batch_tensors(preprocess, paths: list[Path], device: str): def compute(frames: list[Path], reference: Path, model, preprocess, - device: str, batch_size: int) -> list[tuple[str, float, float]]: + device: str, batch_size: int, + extra: list[Path] | None = None) -> list[tuple]: """(frame, dreamsim_to_ref, dreamsim_to_prev) for every frame. The reference is embedded once and the batch is compared against a view of it, so each frame costs two model calls worth of images (ref + prev). The first frame is compared against itself for dreamsim_to_prev (0.0), matching analyze_drift.py's convention. + + extra (optional, same length as frames) is a second image per frame that + is ALSO compared against the same reference; its distance becomes a 4th + tuple element. dltb-distance-feedback uses it for the model input (the + blend the pass actually consumed), which is not any saved frame. """ import torch from PIL import Image + if extra is not None and len(extra) != len(frames): + raise SystemExit("compute(extra=...): extra must match frames one-to-one") ref = preprocess(Image.open(reference)).to(device) prevs = [frames[0], *frames[:-1]] - rows: list[tuple[str, float, float]] = [] + rows: list[tuple] = [] for start in range(0, len(frames), batch_size): chunk = frames[start:start + batch_size] cur = batch_tensors(preprocess, chunk, device) @@ -237,14 +251,59 @@ def compute(frames: list[Path], reference: Path, model, preprocess, with torch.no_grad(): to_ref = model(ref.expand(len(chunk), -1, -1, -1), cur) to_prev = model(prv, cur) + to_extra = None + if extra is not None: + ex = batch_tensors(preprocess, extra[start:start + batch_size], + device) + to_extra = model(ref.expand(len(ex), -1, -1, -1), ex) # 1 - cos() can undershoot 0 by a float epsilon on identical frames # (device-dependent, e.g. -2.4e-07 on CPU); the metric is [0, 2]. - rows.extend((p.stem, float(max(0.0, a)), float(max(0.0, b))) - for p, a, b in zip(chunk, to_ref, to_prev)) + if to_extra is None: + rows.extend((p.stem, float(max(0.0, a)), float(max(0.0, b))) + for p, a, b in zip(chunk, to_ref, to_prev)) + else: + rows.extend((p.stem, float(max(0.0, a)), float(max(0.0, b)), + float(max(0.0, c))) + for p, a, b, c in zip(chunk, to_ref, to_prev, to_extra)) print(f" {len(rows)}/{len(frames)} frames", flush=True) return rows +def plot_metrics(rows: list[tuple], out_png: Path, title: str, + extra_label: str | None = None) -> None: + """DreamSim chart: frame number on X, distance on Y. + + Rows are (frame, dreamsim_to_ref, dreamsim_to_prev[, extra_to_ref]); the + 4th series is plotted (and labelled with extra_label) only when it is + present in the rows. Shared by dltb-distance-loop and + dltb-distance-feedback. Agg is selected before pyplot so this also works + headless on the pod. + """ + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + from matplotlib.ticker import MaxNLocator + + x = range(len(rows)) + fig, ax = plt.subplots(figsize=(10, 5), dpi=150) + ax.plot(x, [r[1] for r in rows], linewidth=1.5, + label="to original (drift)") + ax.plot(x, [r[2] for r in rows], linewidth=1.5, + label="to previous frame (fixed point)") + if extra_label is not None and len(rows[0]) > 3: + ax.plot(x, [r[3] for r in rows], linewidth=1.5, label=extra_label) + ax.set_xlabel("iteration (frame number; 0 = prepared original)") + ax.set_ylabel("DreamSim distance (1 - cosine similarity; lower = more similar)") + ax.set_title(title) + ax.xaxis.set_major_locator(MaxNLocator(integer=True)) # frame numbers + ax.grid(alpha=0.3) + ax.legend() + fig.tight_layout() + out_png.parent.mkdir(parents=True, exist_ok=True) + fig.savefig(out_png) + plt.close(fig) + + def main() -> None: args = parse_args() if args.every < 1: diff --git a/src/dltb/distance_feedback.py b/src/dltb/distance_feedback.py new file mode 100644 index 0000000..05f4e3d --- /dev/null +++ b/src/dltb/distance_feedback.py @@ -0,0 +1,275 @@ +# Permission to use, copy, modify, and/or distribute this software for +# any purpose with or without fee is hereby granted. +# +# THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL +# WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES +# OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE +# FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY +# DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN +# AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT +# OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. + +"""dltb-distance-feedback: static-source feedback loop with a DreamSim chart. + +dltb-continuous' stateful loop, but the "video" is one image repeated: every +simulated source frame N is the SAME prepared input, so the loop is + + source_n = (1-a) * P_{n-1} + a * N (a = --anchor-blend) + P_n = f(source_n) + +with no motion at all -- no reprojection, no tails (flow between identical +frames is exactly zero, so the engine-motion-vector warp would be the +identity and only cost CPU; dltb-continuous is where those phases live). +What is left is a one-parameter family between two already-existing +experiments, which makes the endpoints a free correctness check: + + a = 0.0 blend(P, N, 0) = P -> exactly dltb-distance-loop (free-running; + same frames and same numbers for the same strength/steps/seed) + a = 1.0 the model always gets the untouched input -> independent + repeated passes = the boil test + +and in between "static scene, carried state partially re-anchored": how much +does the model's own history drag the output away from a fresh render of the +same image? The third series answers that directly (see below). + +Each pass is saved (frames/frame_NNNN.png) AND its input blend is saved +(inputs/input_NNNN.png), then DreamSim measures three things per frame: + + dreamsim_to_ref frame_n vs the prepared original (drift) + dreamsim_to_prev frame_n vs frame_{n-1} (perceptual fixed point) + dreamsim_input_to_ref the pass's model INPUT vs the prepared original + (how far the conditioning itself moved -- for + a<1 the blend drags toward the carried state; + frame 0 is the original itself, so it starts 0.0) + +Distances are DreamSim's 1 - cosine similarity (0.0 = identical, higher = +more different), the same metric and CSV columns as dltb-distance and +dltb-distance-loop, so curves from the tools are directly comparable. + +Artifacts land in the run directory: + + frame_0000_original.png the model-sized input (the reference frame) + frames/frame_NNNN.png every pass's output + inputs/input_NNNN.png every pass's actual model input (the blend) + distance_metrics.csv frame + the three series above + distance_plot.png all three series, frame number on X (matched + Y scale, lower = more similar) + timelapse.mp4 best-effort, from frames/ + timelapse_inputs.mp4 best-effort, from inputs/ + run.json settings, since the tag encodes only + strength and blend (pass --output-dir + subtrees when varying prompt/steps/seed) + +Default output: untracked/output_/feedback_s_a (the +shared --output-dir flag puts runs under /feedback_s..._a... +instead; keep pod runs out of the local tree, per AGENTS.md). + +Only the three strength-capable img2img models are accepted (sd-turbo, +sdxl-turbo, flux-schnell -- the klein editors take no --strength, and their +dual-reference conditioning is dltb-klein's loop over real video). With the +defaults (4 steps x 0.4) every pass runs exactly one denoise step. + +INPUT RESOLUTION: as in dltb-distance-loop -- if the input already is exactly +the output size it is used untouched, otherwise it goes through +imaging.prepare_frame()'s rule (largest centered region of the target aspect, +LANCZOS-resized to width x height). Size defaults to the model's native +geometry (512 for sd-turbo, 768 otherwise); --width/--height override it. + +TWO PHASES, ONE ACCELERATOR: the diffusion pipeline is unloaded and its +cached blocks are returned to the driver (models.release_accelerator) before +DreamSim is loaded, so peak VRAM is the larger of the two models, not their +sum. DreamSim runs on the same --device; its weights (~2.7 GB for the default +ensemble) download into --cache-dir on first use. + +Usage: + uv run dltb-distance-feedback --model sd-turbo --input menu.png \\ + --strength 0.4 --anchor-blend 0.3 --frames 200 \\ + --prompt "a bronze lion sculpture" + + # Control: must reproduce dltb-distance-loop's numbers for the same seed + uv run dltb-distance-feedback --model sd-turbo --input menu.png \\ + --anchor-blend 0.0 --output-dir untracked/output_lab/parity +""" + +from __future__ import annotations + +import argparse +import json +import statistics +from pathlib import Path + +from .analyze_distance import (add_dreamsim_args, check_dreamsim_args, + compute, load_model, plot_metrics, + write_metrics_csv) +from .args import (add_geometry_args, add_model_args, add_output_args, + add_pass_args, non_empty_path) +from .imaging import PassSettings, fit_to_model_size, make_generator, run_pass +from .models import (MODELS, STRENGTH_MODELS, check_requirements, + load_pipeline, release_accelerator, resolve_device, + resolve_geometry) +from .output import assemble_timelapse, run_dir + +INPUT_SERIES = "dreamsim_input_to_ref" +INPUT_LABEL = "model input to original (blend drag)" + + +def parse_args(argv: list[str] | None = None) -> argparse.Namespace: + p = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + add_model_args(p, models=STRENGTH_MODELS) + add_pass_args(p) + add_geometry_args(p) + add_output_args(p) + p.add_argument("--input", required=True, type=non_empty_path, + help="Path to the input image (png/jpg/...); every simulated " + "source frame is this one image, prepared once") + p.add_argument("--frames", type=int, default=200, + help="Number of simulated video frames (one model pass each; " + "every pass is saved and measured)") + p.add_argument("--anchor-blend", type=float, default=0.3, + help="Weight alpha of the fresh frame (the input image) in the " + "blend (1-a)*previous_output + a*fresh_frame. 0.0 = " + "free-running (same loop as dltb-distance-loop), 1.0 = " + "fully re-anchored every frame (repeated independent " + "passes = boil test)") + p.add_argument("--video-fps", type=int, default=30, + help="FPS of the assembled mp4 timelapses") + add_dreamsim_args(p) + return p.parse_args(argv) + + +def run(args: argparse.Namespace) -> None: + spec = MODELS[args.model] + width, height = resolve_geometry(spec, args.width, args.height) + if not 0.0 < args.strength <= 1.0: + raise SystemExit("--strength must be in (0, 1] (magnitude of noise per pass)") + if not 0.0 <= args.anchor_blend <= 1.0: + raise SystemExit("--anchor-blend must be in [0, 1]") + if args.frames < 1: + raise SystemExit("--frames must be >= 1") + check_requirements(spec, args.num_inference_steps, args.strength) + check_dreamsim_args(args) + + out_dir = run_dir(args.output_dir, args.model, + f"feedback_s{args.strength:g}_a{args.anchor_blend:g}") + frames_dir = out_dir / "frames" + inputs_dir = out_dir / "inputs" + frames_dir.mkdir(parents=True, exist_ok=True) + inputs_dir.mkdir(parents=True, exist_ok=True) + device = resolve_device(args.device) + + (out_dir / "run.json").write_text(json.dumps({ + "model": args.model, + "model_id": spec.model_id, + "input": str(args.input), + "width": width, + "height": height, + "frames": args.frames, + "prompt": args.prompt, + "strength": args.strength, + "num_inference_steps": args.num_inference_steps, + "guidance_scale": (spec.guidance_scale if args.guidance_scale is None + else args.guidance_scale), + "seed": args.seed, + "fixed_seed": args.fixed_seed, + "anchor_blend": args.anchor_blend, + "device": device, + "dreamsim_type": args.dreamsim_type + ("_patch" if args.patch else ""), + "batch_size": args.batch_size, + "cache_dir": str(args.cache_dir), + }, indent=2) + "\n") + + # --- phase 1: the diffusion model alone on the accelerator ------------- + from PIL import Image + + pipe = load_pipeline(spec, args.offload, device) + settings = PassSettings(prompt=args.prompt, + num_inference_steps=args.num_inference_steps, + strength=args.strength, + guidance_scale=args.guidance_scale) + + original_path = out_dir / "frame_0000_original.png" + original = fit_to_model_size(Image.open(args.input), width, height) + original.save(original_path) + print(f"input: {args.input} -> {width}x{height} ({original_path})", flush=True) + + current = None # previous PROCESSED frame (carried state) + frame_paths: list[Path] = [] + input_paths: list[Path] = [] + for i in range(1, args.frames + 1): + # A static source has no "previous source frame" to estimate flow + # from, and N is the same image every pass. + source = (original if current is None + else Image.blend(current, original, args.anchor_blend)) + source_path = inputs_dir / f"input_{i:04d}.png" + source.save(source_path) + input_paths.append(source_path) + + current = run_pass(pipe, spec, settings, source, + make_generator(args.seed, args.fixed_seed, i, + device=device), + width, height) + path = frames_dir / f"frame_{i:04d}.png" + current.save(path) + frame_paths.append(path) + if i % 10 == 0 or i == 1: + print(f"frame {i}/{args.frames}", flush=True) + + # The two models must never share the accelerator: drop the pipeline and + # return its cached blocks before DreamSim loads (peak VRAM = max, not sum). + del pipe + release_accelerator(device) + assemble_timelapse(frames_dir, out_dir / "timelapse.mp4", args.video_fps) + assemble_timelapse(inputs_dir, out_dir / "timelapse_inputs.mp4", args.video_fps, + pattern="input_*.png") + + # --- phase 2: DreamSim over the original plus every saved frame -------- + cache_dir = Path(args.cache_dir) + # dreamsim's own downloader mkdirs a single level (os.mkdir), so nested + # --cache-dir needs this mkdir -p, exactly as in dltb-distance. + cache_dir.mkdir(parents=True, exist_ok=True) + label = args.dreamsim_type + ("_patch" if args.patch else "") + print(f"loading dreamsim {label} on {device} (weights: {cache_dir}; the " + "first run downloads them)", flush=True) + model, preprocess = load_model(args.dreamsim_type, args.patch, args.retries, + device, cache_dir) + # compute() compares each frame against its predecessor, so seeding the + # list with the original gives frame 1 dreamsim_to_prev = vs original. The + # extra list is index-aligned: row 0's "model input" is the original + # itself (there is no pass before it), hence extra[0] = original. + rows = compute([original_path, *frame_paths], original_path, model, + preprocess, device, args.batch_size, + extra=[original_path, *input_paths]) + del model, preprocess + release_accelerator(device) + + out_csv = out_dir / "distance_metrics.csv" + out_plot = out_dir / "distance_plot.png" + write_metrics_csv(rows, out_csv, extra_header=INPUT_SERIES) + prompt = args.prompt if len(args.prompt) <= 60 else args.prompt[:57] + "..." + title = (f"{args.model} feedback: anchor-blend={args.anchor_blend:g} " + f"strength={args.strength:g} steps={args.num_inference_steps}") + if prompt: + title += f" prompt={prompt!r}" + plot_metrics(rows, out_plot, title, extra_label=INPUT_LABEL) + + tail = rows[-10:] + print(f"Analyzed {len(rows)} frames (original + {args.frames} passes); " + f"dreamsim {label} on {device}") + print(f" dreamsim_to_ref, last {len(tail)} frames: " + f"{statistics.mean(r[1] for r in tail):8.4f}") + print(f" dreamsim_to_prev, last {len(tail)} frames: " + f"{statistics.mean(r[2] for r in tail):8.4f}") + print(f" {INPUT_SERIES}, last {len(tail)} frames: " + f"{statistics.mean(r[3] for r in tail):8.4f}") + print(f" CSV : {out_csv}") + print(f" plot: {out_plot}") + + +def main(argv: list[str] | None = None) -> None: + run(parse_args(argv)) + + +if __name__ == "__main__": + main() diff --git a/src/dltb/distance_loop.py b/src/dltb/distance_loop.py index c0676c1..908922f 100644 --- a/src/dltb/distance_loop.py +++ b/src/dltb/distance_loop.py @@ -78,7 +78,8 @@ import statistics from pathlib import Path from .analyze_distance import (add_dreamsim_args, check_dreamsim_args, - compute, load_model, write_metrics_csv) + compute, load_model, plot_metrics, + write_metrics_csv) from .args import (add_geometry_args, add_model_args, add_output_args, add_pass_args, non_empty_path) from .imaging import PassSettings, fit_to_model_size, make_generator, run_pass @@ -107,35 +108,6 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: return p.parse_args(argv) -def plot_metrics(rows: list[tuple[str, float, float]], out_png: Path, - title: str) -> None: - """Two-line DreamSim chart: frame number on X, distance on Y. - - Agg is selected before pyplot so this also works headless on the pod. - """ - import matplotlib - matplotlib.use("Agg") - import matplotlib.pyplot as plt - from matplotlib.ticker import MaxNLocator - - x = range(len(rows)) - fig, ax = plt.subplots(figsize=(10, 5), dpi=150) - ax.plot(x, [r[1] for r in rows], linewidth=1.5, - label="to original (drift)") - ax.plot(x, [r[2] for r in rows], linewidth=1.5, - label="to previous frame (fixed point)") - ax.set_xlabel("iteration (frame number; 0 = prepared original)") - ax.set_ylabel("DreamSim distance (1 - cosine similarity; lower = more similar)") - ax.set_title(title) - ax.xaxis.set_major_locator(MaxNLocator(integer=True)) # frame numbers - ax.grid(alpha=0.3) - ax.legend() - fig.tight_layout() - out_png.parent.mkdir(parents=True, exist_ok=True) - fig.savefig(out_png) - plt.close(fig) - - def run(args: argparse.Namespace) -> None: spec = MODELS[args.model] width, height = resolve_geometry(spec, args.width, args.height) diff --git a/src/dltb/output.py b/src/dltb/output.py index 6b31f4e..8a5f5c6 100644 --- a/src/dltb/output.py +++ b/src/dltb/output.py @@ -38,11 +38,15 @@ def run_dir(output_dir: str | None, model: str, tag: str) -> Path: return d -def assemble_timelapse(frames_dir: Path, video_path: Path, fps: int) -> None: - """Assemble frame_*.png from frames_dir into an mp4 (best effort).""" +def assemble_timelapse(frames_dir: Path, video_path: Path, fps: int, + pattern: str = "frame_*.png") -> None: + """Assemble PNGs matching pattern from frames_dir into an mp4 (best + effort). The default pattern skips the inputs/ dirs and tail copies the + tools save next to the frames; dltb-distance-feedback passes + "input_*.png" for its model-input timelapse.""" try: import imageio.v2 as imageio - frames = sorted(frames_dir.glob("frame_*.png")) + frames = sorted(frames_dir.glob(pattern)) if frames: with imageio.get_writer(video_path, fps=fps, codec="libx264") as w: for f in frames: