Something went wrong. Try again.
Deep Learning Tripping Balls
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141#!/usr/bin/env bash
# Permission to use, copy, modify, and/or distribute this software for# any purpose with or without fee is hereby granted.## THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL# WARRANTIES WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES# OF MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE# FOR ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY# DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN# AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT# OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
## smoke-local.sh -- local single-frame smoke test: one dltb-oneshot pass per# model that fits this machine, on whatever accelerator the tools auto-select# (Apple MPS on a Mac, CUDA if present).## scripts/smoke.sh is the pod acceptance test (full tool surface, one model);# this is its local counterpart: it proves that each locally-viable model# loads and generates ONE frame, and reports wall-clock per model (the first# run of a model includes its download, so rerun for a warm timing).## Default model list -- what fits a 16 GB unified-memory Mac:# sd-turbo, sdxl-turbo, flux2-klein-4b# flux2-klein-4b WORKS but is slow (~5 min per 4-step 768^2 frame, heavy# swapping): single frames only, not video. Deliberately excluded (too large# for 16 GB): flux-schnell (~34 GB bf16) and flux2-klein-9b (gated, ~20-29 GB).# Override MODELS to try them anyway; the model cache is the default# ~/.cache/huggingface (see hf-cache.sh).## Every model runs with a minimal budget: 1 denoise step x strength 1.0 for# the img2img models (they are 1-step distilled); klein runs its model-card# default 4 steps (no --strength). The produced frame is checked for# existence and sanity (non-uniform pixels -- a black/uniform frame means a# broken decode, not a generation).## Usage:# scripts/smoke-local.sh# MODELS="sd-turbo" scripts/smoke-local.sh # subset / single model# OFFLOAD=1 scripts/smoke-local.sh # add --offload per pass# SKIP_GPU_CHECK=1 scripts/smoke-local.sh # bypass the accelerator preflight## Environment:# MODELS model keys to smoke (default: "sd-turbo sdxl-turbo# flux2-klein-4b")# IMG input image (default: untracked/input/inputs.env if# present, else the tracked# input_example/test_512.png)# OFFLOAD=1 pass --offload to every run (smaller unified memory)# SKIP_GPU_CHECK=1 bypass the accelerator preflight## Log: untracked/output/smoke-local_<UTC timestamp>.log
set -euo pipefail
cd "$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
# Input files: env > untracked/input/inputs.env (user) > input_example/inputs.env.source scripts/inputs.sh
MODELS="${MODELS:-sd-turbo sdxl-turbo flux2-klein-4b}"IMG="${IMG:-input_example/test_512.png}"
[[ -f "$IMG" ]] || { echo "smoke-local: input not found: $IMG" >&2; exit 1; }
if [[ "${SKIP_GPU_CHECK:-0}" != "1" ]]; then if ! uv run python -c 'import sys, torch; sys.exit(0 if (torch.cuda.is_available() or torch.backends.mps.is_available()) else 1)'; then echo "smoke-local: torch reports no CUDA or MPS device -- run this on Apple Silicon or a GPU machine" >&2 echo " (SKIP_GPU_CHECK=1 to bypass)" >&2 exit 1 fifi
mkdir -p untracked/outputLOG="untracked/output/smoke-local_$(date -u +%Y%m%d-%H%M%S).log"OUT="untracked/output/smoke-local"rm -rf "$OUT"mkdir -p "$OUT"
img_stem="$(basename "$IMG")"; img_stem="${img_stem%.*}"
log() { echo "$@" | tee -a "$LOG"; }
# One frame's pixel sanity: uniform output (all black/saturated) would pass a# mere existence check but means the decode broke, e.g. the fp16-VAE-on-MPS# failure mode (NOTES.md has the fallback).check_frame() { local f="$1" model="$2" stats if [[ ! -s "$f" ]]; then log "SMOKE-LOCAL FAIL: missing or empty $f" exit 1 fi if ! stats="$(uv run python -c 'import sysimport numpy as npfrom PIL import Imagea = np.asarray(Image.open(sys.argv[1]).convert("RGB"), dtype=np.float32)print(f"mean={a.mean():.1f} std={a.std():.1f}")sys.exit(0 if a.std() > 1.0 else 1)' "$f" 2>&1 | tee -a "$LOG")"; then log "SMOKE-LOCAL FAIL: degenerate (uniform) frame for $model: $f" exit 1 fi log "ok: $model $f ($(du -h "$f" | cut -f1), $stats)"}
SUMMARY=""log "smoke-local: models=$MODELS img=$IMG offload=${OFFLOAD:-0}"log "smoke-local: log=$LOG"
for model in $MODELS; do extra="" [[ "${OFFLOAD:-0}" == "1" ]] && extra="--offload"
# klein has no --strength and runs its full model-card schedule; the # img2img models are 1-step distilled (1 step x strength 1.0 = 1 denoise # step, the same minimal budget smoke.sh uses). $strength/$extra are # word-split on purpose (bash 3.2 on macOS has no safe empty arrays). steps=1; strength="--strength 1.0" case "$model" in flux2-klein-*) steps=4; strength="" ;; esac
log "" log "== $(date -u +%Y-%m-%dT%H:%M:%SZ) dltb-oneshot model=$model steps=$steps" start=$SECONDS uv run dltb-oneshot --model "$model" --input "$IMG" \ --num-inference-steps "$steps" $strength $extra \ --output-dir "$OUT/$model" 2>&1 | tee -a "$LOG" elapsed=$((SECONDS - start))
check_frame "$OUT/$model/${img_stem}_oneshot/frames/frame_0001.png" "$model" SUMMARY="${SUMMARY}${model}: ${elapsed}s (wall clock${OFFLOAD:+ +offload})\n"done
log ""log "smoke-local: summary (wall clock includes one-time downloads on first run):"printf "%b" "$SUMMARY" | tee -a "$LOG"log "smoke-local: PASS -- all requested models produced a frame under $OUT"