Something went wrong. Try again.
Monorepo for Aesthetic.Computer aesthetic.computer
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475#!/usr/bin/env python3"""boundfix.py — refine whisper word boundaries to energy valleys.
Whisper's word edges drift; real word boundaries in singing sit atenergy minima (stop closures, breaths). For each adjacent matched-wordpair, this slides the shared boundary to the RMS valley within ±0.28s,and extends the last word's tail to where energy actually dies. Outputis a windows file holyvox.mjs merges over the whisper timings.
pop/.venv/bin/python pop/imab/bin/boundfix.py <stem.wav> <syllnote.json> <out.json>"""import json, os, sys, reimport numpy as npimport librosa
wav, sylj, outp = sys.argv[1], sys.argv[2], sys.argv[3]doc = json.load(open(sylj))y, sr = librosa.load(wav, sr=22050, mono=True)hop = 128rms = librosa.feature.rms(y=y, frame_length=1024, hop_length=hop)[0]times = librosa.times_like(rms, sr=sr, hop_length=hop)sm = np.convolve(rms, np.ones(9) / 9, mode="same")
TMPL = "i'm a butterfly flapping for you guys just a costume i put on in my room".split(" ")def norm(w): return re.sub(r"[^a-z']", "", w.lower())def fuzzy(a, b): return a == b or (len(a) > 3 and len(b) > 3 and (a.startswith(b[:4]) or b.startswith(a[:4]))) \ or (a in ("a", "the") and b in ("a", "the"))seq = []ti = 0for w in doc["words"]: if ti < len(TMPL) and fuzzy(TMPL[ti], norm(w["text"])): seq.append({"ti": ti, "text": TMPL[ti], "fromMs": w["fromMs"], "toMs": w["toMs"]}) ti += 1
def valley(t, lo, hi, span=0.35): """energy minimum near t, clamped so neither word loses its core""" sel = (times > max(t - span, lo)) & (times < min(t + span, hi)) idx = np.where(sel)[0] if not len(idx): return t return float(times[idx[np.argmin(sm[idx])]])
for i in range(len(seq) - 1): t = (seq[i]["toMs"] + seq[i + 1]["fromMs"]) / 2000 lo = seq[i]["fromMs"] / 1000 + 0.12 # keep ≥120ms of each word hi = seq[i + 1]["toMs"] / 1000 - 0.12 if hi <= lo: continue b = valley(t, lo, hi) seq[i]["toMs"] = int(b * 1000) seq[i + 1]["fromMs"] = int(b * 1000)# the last word rings until its energy truly dies (max +1.2 s)last = seq[-1]t = last["toMs"] / 1000sel = (times > t) & (times < t + 1.2)idx = np.where(sel)[0]th = sm[(times > last["fromMs"] / 1000) & (times < t)].max() * 0.06 if idx.size else 0for i in idx: if sm[i] < th: last["toMs"] = int(times[i] * 1000); breakelse: if idx.size: last["toMs"] = int(times[idx[-1]] * 1000)
# hand overrides (tracked, singer-confirmed) win over everythingov_path = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), f"boundaries-{wav.split('whistlegraph-')[-1].split('/')[0]}.json")if os.path.exists(ov_path): ov = json.load(open(ov_path))["overrides"] for w in seq: o = ov.get(str(w["ti"])) if o: w.update(o) print(f" (+{len(ov)} hand overrides from {os.path.basename(ov_path)})")json.dump({"source": wav.split("/")[-1], "words": seq}, open(outp, "w"), indent=1)for w in seq: print(f" {w['text']:<12}{w['fromMs']/1000:6.2f} – {w['toMs']/1000:6.2f}")print(f"✓ {outp}")