From 77182d132e82e15f1c6372a290375714cb29788b Mon Sep 17 00:00:00 2001 From: "prompt.ac/@jeffrey" Date: Sat, 25 Apr 2026 23:56:40 -0700 Subject: [PATCH] recap: new pipeline for narrated, captioned monorepo activity videos MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit audience/.mjs declares narration + segment markers + slide HTML (or query-driven slide functions) + transcript fixes; jeffrey-pvc TTS via /api/say is the source of truth, whisper word-timestamps drive slide durations and word-synced subtitle pills, and bin/scout.mjs resolves per-slide content queries (file globs, PDF→PNG via pdftoppm, git --grep commit lists, recent-files mtime queries) into base64 data URLs that slide bodies consume. fia is the first audience; runs end-to-end via fish pipeline.fish fia. Co-Authored-By: Claude Opus 4.7 (1M context) --- recap/.gitignore | 3 + recap/SCORE.md | 186 +++++++++++++++++++++++++++++ recap/audience/fia.mjs | 237 +++++++++++++++++++++++++++++++++++++ recap/bin/align.mjs | 72 +++++++++++ recap/bin/build-filter.mjs | 47 ++++++++ recap/bin/compose.fish | 40 +++++++ recap/bin/scout.mjs | 120 +++++++++++++++++++ recap/bin/slides.mjs | 155 ++++++++++++++++++++++++ recap/bin/subtitles.mjs | 126 ++++++++++++++++++++ recap/bin/transcribe.mjs | 38 ++++++ recap/bin/tts.mjs | 37 ++++++ recap/pipeline.fish | 40 +++++++ 12 files changed, 1101 insertions(+) create mode 100644 recap/.gitignore create mode 100644 recap/SCORE.md create mode 100644 recap/audience/fia.mjs create mode 100755 recap/bin/align.mjs create mode 100755 recap/bin/build-filter.mjs create mode 100755 recap/bin/compose.fish create mode 100755 recap/bin/scout.mjs create mode 100755 recap/bin/slides.mjs create mode 100755 recap/bin/subtitles.mjs create mode 100755 recap/bin/transcribe.mjs create mode 100755 recap/bin/tts.mjs create mode 100755 recap/pipeline.fish diff --git a/recap/.gitignore b/recap/.gitignore new file mode 100644 index 0000000000..7edfc96258 --- /dev/null +++ b/recap/.gitignore @@ -0,0 +1,3 @@ +out/ +models/*.bin +node_modules/ diff --git a/recap/SCORE.md b/recap/SCORE.md new file mode 100644 index 0000000000..259523cb24 --- /dev/null +++ b/recap/SCORE.md @@ -0,0 +1,186 @@ +# Recap + +Generates narrated, captioned video recaps of monorepo activity for a chosen +audience (currently `fia`, jas's girlfriend; trivially extendable to others). +The audio is the source of truth — whisper word-level timestamps drive slide +durations, so visuals stay in sync with what the voice is actually saying. + +The default voice is `jeffrey-pvc` (the same Professional Voice Clone used in +the `say` piece and the LACMA grant pitch video), called via `/api/say` on +production. + +## Pipeline + +``` +audience/.mjs (narration + segment markers + slide HTML/queries + voice + transcriptFixes) + │ + ▼ bin/tts.mjs +out/recap.mp3 (jeffrey-pvc TTS via /api/say) + │ + ▼ bin/transcribe.mjs (whisper-cli, models/ggml-base.en.bin) +out/words.json ([{text, fromMs, toMs}, ...]) + │ + ▼ bin/align.mjs (matches audience.segments[].marker) +out/segments.json ([{name, startSec, endSec, durationSec}, ...]) + │ + ▼ bin/scout.mjs (resolves per-slide content queries; pdftoppm for PDFs) +out/assets.json (slide-name → {queryName: dataUrl|commits|paths}) + │ + ▼ bin/slides.mjs (puppeteer + ywft-processing + purple-pals + scouted assets) +out/slides/*.png (1080×1920 PNG per segment) +out/concat.txt (ffmpeg concat demuxer w/ real durations) +out/duration.txt (total seconds, including trailing silence) + │ + ▼ bin/subtitles.mjs (chunks words.json, applies transcriptFixes, renders pill PNGs) +out/subs/*.png (1080×220 transparent subtitle pill per chunk) +out/subs.json ([{file, startSec, endSec, text}, ...]) + │ + ▼ bin/build-filter.mjs (emits filter graph: showwaves + drawbox + per-sub overlay) + ▼ bin/compose.fish +out/recap.mp4 (1080×1920, h264 + aac, faststart, baked subs) +``` + +Run end-to-end: + +```fish +./pipeline.fish fia # fresh tts + everything +./pipeline.fish fia --skip-tts # reuse existing out/recap.mp3 (re-align/re-render) +``` + +First run only — download the whisper model (~141 MB): + +```fish +curl -L -o models/ggml-base.en.bin \ + https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin +``` + +## Architecture decisions + +- **Audio is the source of truth.** Slide durations come from whisper word + timestamps, not from hand-tuned guesses. Re-recording the audio (e.g. a + re-edit of the narration) automatically retimes the visuals. +- **Markers are anchor phrases**, not paraphrases. Each `audience.segments[]` + has a `marker` field that must appear in the narration verbatim (modulo + whisper transcription quirks — match is case-insensitive and punctuation- + stripped). `align.mjs` fails loud if any marker is missing. +- **End card sits in trailing silence.** The last segment uses a synthetic + `__END__` marker; the audio is padded with `apad` so the silent end card + has time to breathe without truncating the narration. +- **Slides are HTML rendered by Chrome.** Reuses the `oven/` puppeteer install + to avoid taking on a new dep. ywft-processing-bold/regular fonts are + inlined as base64; `unicode-range: U+0020-007E` constrains the AC font to + ASCII so Chrome falls back to system fonts for `ñ`, `中文`, `日本語`, + `·`, `×`, etc. +- **Progress bar is `drawbox` with `w='iw*t/$TOTAL'`.** This ffmpeg build + lacks `drawtext` and `subtitles`, so visible captions live in the slide + PNGs; only the bar (no text) is composited at runtime. + +## Content queries (scout) + +Slide bodies can be **functions** of resolved query results. `scout.mjs` runs +every query declared on a slide and writes data URLs / commit lists / file +paths into `out/assets.json`. The slide function then receives those values +and produces HTML. + +Three query shapes are supported: + +| Shape | Result | +| -------------------------------------------------------------------- | --------------------------------------------------------- | +| `{ glob: "" }` | base64 data URL of the first matching image (PNG/JPG/WebP/SVG) | +| `{ glob: ".pdf", pdfPage: 1, pdfWidth: 600 }` | base64 data URL of one PDF page rendered via pdftoppm | +| `{ commits: "", since: "48 hours ago", limit: 5 }` | `[{hash, subject}, ...]` from `git log --grep -E` | +| `{ files: "", sinceHours: 48, limit: 60 }` | matching paths newer than N hours, sorted newest first | + +Globs are repo-relative or absolute. PDF rendering uses 150 DPI by default; +`pdfWidth` scales the longer side. Commit grep is POSIX extended (`|` +alternation works without escaping). Failed queries log a warning and skip +the value; the slide function should defensively handle missing results +(e.g. `${(commits || []).map(...)}`). + +Example slide entry in an audience config: + +```js +"02_notepat": { + queries: { + icon: { glob: "ac-electron/build/icon.png" }, + paper: { glob: "system/public/papers.aesthetic.computer/notepat-26-arxiv-cards.pdf", + pdfPage: 1, pdfWidth: 600 }, + commits: { commits: "^notepat|^build-notepat", since: "48 hours ago", limit: 5 }, + }, + body: ({ icon, paper, commits }) => ` +
+ + + ${(commits || []).map(c => `
${c.hash} ${c.subject}
`).join("")} +
`, +}, +``` + +A slide entry can also still be a plain HTML string when no scouting is +needed (see `01_title`, `03_arena`, etc. in `audience/fia.mjs`). + +## Subtitle transcript fixes + +Whisper renders dictionary-style — `notepat` becomes `Notepad`, `baktok` +becomes `Backtalk`, `menubar` becomes `menu bar`. Fix per-audience without +re-running whisper: + +```js +transcriptFixes: { + "Notepad": "notepat", + "Backtalk": "baktok", + "menu bar": "menubar", +} +``` + +Match is case-insensitive and applied to each subtitle chunk's joined text +(so multi-word fixes like `"laid on Linux": "late on Linux"` work). + +## Adding a new audience + +Drop `audience/.mjs` exporting `audience` (and a `PALETTE` if you want +to deviate from fia's). Required shape: + +```js +export const audience = { + name: "", + handle: "", + voice: { provider: "jeffrey", voice: "neutral:0" }, + narration: "", + segments: [ + { name: "01_title", marker: "" }, + { name: "02_topic1", marker: "" }, + // ... + { name: "10_end", marker: "__END__", trailingSilenceSec: 3 }, + ], + slides: { "01_title": "", /* ...one per segment */ }, +}; +``` + +Then `./pipeline.fish `. + +## Files + +| File | Role | +| ------------------------ | ------------------------------------------------------------- | +| `audience/fia.mjs` | narration, markers, slide HTML/queries, palette, fixes | +| `bin/tts.mjs` | POST narration → `/api/say` → `out/recap.mp3` | +| `bin/transcribe.mjs` | `whisper-cli` → `out/words.json` | +| `bin/align.mjs` | match markers in transcript → `out/segments.json` | +| `bin/scout.mjs` | resolve per-slide content queries → `out/assets.json` | +| `bin/slides.mjs` | puppeteer-render slide PNGs (consume assets) + `concat.txt` | +| `bin/subtitles.mjs` | chunk words into pills (apply transcriptFixes) → `subs.json` | +| `bin/build-filter.mjs` | emit ffmpeg filter graph for compose (one overlay per sub) | +| `bin/compose.fish` | ffmpeg compose final mp4 | +| `pipeline.fish` | runs all six stages | +| `models/ggml-base.en.bin`| whisper model (gitignored, downloaded on first run) | +| `out/` | all generated artifacts (gitignored) | + +## Dependencies + +- `whisper-cli` (homebrew `whisper-cpp`) +- `ffmpeg` with `libx264`, `aac`, `showwaves`, `drawbox`, `apad`, `movie`, `overlay` (homebrew default) +- `pdftoppm` (homebrew `poppler`) for PDF → PNG in scout +- `node` (uses `oven/node_modules/puppeteer` to avoid extra installs) +- Google Chrome at `/Applications/Google Chrome.app` (puppeteer driver) +- Network access to `aesthetic.computer/api/say` (jeffrey-pvc TTS) diff --git a/recap/audience/fia.mjs b/recap/audience/fia.mjs new file mode 100644 index 0000000000..4acab672ee --- /dev/null +++ b/recap/audience/fia.mjs @@ -0,0 +1,237 @@ +// Audience config: fia (jas's girlfriend, non-technical). +// Voice: jeffrey-pvc (default in /api/say). Style: lowercase, warm, ends in a rhyme. +// +// `narration` is the verbatim text POSTed to /api/say. +// `segments` anchor each slide to a phrase in the narration; whisper word-level +// timestamps determine real durations. The marker is matched case-insensitively +// against the transcript with punctuation stripped, so it must appear in the +// audio (not be a paraphrase of it). + +export const PALETTE = { + bg: "#201040", + accent: "#ff69b4", + cyan: "#70f0e0", + lime: "#a0f070", + magenta: "#ff70d0", + yellow: "#ffd860", + cream: "#fcf7c5", + off: "#ffffffcc", + dim: "#9080c0", +}; + +export const audience = { + name: "fia", + handle: "@fifi", + voice: { provider: "jeffrey", voice: "neutral:0" }, + + // Narration as one paragraph for clean TTS. Lowercase house-voice for jas. + // The voice says "fifi" (her handle, sans @ — easier to pronounce) instead + // of the nickname "fi", since she goes by @fifi on AC. + narration: `hey fifi, here's the last two days at the keyboard for you. the big one — notepat finally has a little remote that lives inside ableton, so when i'm playing, anyone can pop a tiny keyboard panel right into their session. the keys glow in piano colors, octaves stack, and the whole thing feels like a real instrument now. the multiplayer arena got dressed up too — a minimap in the top right, plain english labels instead of jargon, and every spectator gets their own quiet color. the camera piece, the one we call cap, learned to hold-to-record like baktok, so a tap doesn't startle it anymore. on the side, i wrote a little paper for parag about why audio feels late on linux — mostly love letters to interrupts — retranslated sucking on the complex into spanish, danish, chinese, and japanese, and rebuilt the menubar slab in proper swift so the mail submenu actually works. and through it all the oven kept its rhythm, baking pdfs in the background. fifi, you make the keystrokes kinder — every quiet line of code is a little you-reminder.`, + + // Each slide is anchored to a phrase that appears in the narration. + // align.mjs finds the phrase in the whisper transcript and uses the start + // timestamp of its first word as the slide start. The slide ends when the + // next slide starts (or at audio end + trailingSilenceSec for the last). + // Whisper substitutions for the displayed subtitle text. Match is + // case-insensitive; replacement is verbatim. Use these to fix word forms + // that whisper renders dictionary-style ("Notepad" for the AC piece + // "notepat", "menu bar" → "menubar", etc.) or to correct mishears. + transcriptFixes: { + "Notepad": "notepat", + "mini-map": "minimap", + "Backtalk": "baktok", + "menu bar": "menubar", + "sub-menu": "submenu", + "laid on Linux": "late on Linux", + "you reminder": "you-reminder", + "CAP": "cap", + }, + + segments: [ + { name: "01_title", marker: "hey fifi" }, + { name: "02_notepat", marker: "the big one" }, + { name: "03_arena", marker: "the multiplayer arena" }, + { name: "04_cap", marker: "the camera piece" }, + { name: "05_paper", marker: "i wrote a little paper" }, + { name: "06_translations", marker: "retranslated" }, + { name: "07_slab", marker: "rebuilt the" }, + { name: "08_oven", marker: "and through it all" }, + { name: "09_outro", marker: "fifi you make" }, + // End card runs after the audio ends; pipeline pads the audio with silence. + { name: "10_end", marker: "__END__", trailingSilenceSec: 3 }, + ], + + // Each slide's HTML body. Rendered into a 1080x1920 frame by slides.mjs. + // Use ${PALETTE.x} colors. CSS in slides.mjs handles font fallback. + slides: { + "01_title": ` +
+
+
+
aesthetic computer
+
@fifi
+
last 48 hours @ the keyboard
+
+
2026·04·24 → 2026·04·25
+
`, + "02_notepat": { + queries: { + icon: { glob: "ac-electron/build/icon.png" }, + paper: { glob: "system/public/papers.aesthetic.computer/notepat-26-arxiv-cards.pdf", pdfPage: 1, pdfWidth: 600 }, + // Match commits whose subject *starts with* notepat-related prefix so + // we don't pick up unrelated commits that mention notepat in passing. + commits: { commits: "^notepat|^build-notepat", since: "48 hours ago", limit: 5 }, + }, + body: ({ icon, paper, commits }) => ` +
+
01 / 08 · notepat
+
+ ${icon ? `` : ""} +
notepat × ableton
+
+
+ a little remote that lives inside your DAW. + piano colors, stacked octaves, square pads. +
+
+
+ ${(commits || []).map((c) => `
${c.hash.slice(0, 4)}${(c.subject.split(":").slice(1).join(":").trim() || c.subject).slice(0, 56)}
`).join("")} +
+ ${paper ? `` : ""} +
+
"feels like a real instrument now."
+
`, + }, + "03_arena": ` +
+
02 / 08 · arena
+
arena, dressed up
+
+ minimap, top-right.
+ plain english instead of jargon.
+ every spectator gets their own quiet color. +
+
+
dfa8minimap top-right + plain-English HUD
+
2fedcool per-handle specColor + 'specs' label
+
06dafix UDP snap drops + telemetry
+
+
`, + "04_cap": ` +
+
03 / 08 · cap
+
hold to record
+
+ the camera piece learned baktok manners.
+ a tap doesn't startle it anymore. +
+
+
6839cap + video: hold-to-record (BakTok-style)
+
a6b8slide-up-to-zoom while holding record
+
6f54portrait-native camera + drawn mic glyph
+
+
`, + "05_paper": ` +
+
04 / 08 · papers
+
a paper for parag
+
+ why audio feels late on linux —
+ mostly love letters to interrupts. +
+
+
arxiv-latency
+
IRQ + audio/input latency analysis
+
7-phase commit archaeology
+
+
`, + "06_translations": { + queries: { + es: { glob: "system/public/papers.aesthetic.computer/sucking-on-the-complex-26-arxiv-es.pdf", pdfPage: 1, pdfWidth: 400 }, + da: { glob: "system/public/papers.aesthetic.computer/sucking-on-the-complex-26-arxiv-da.pdf", pdfPage: 1, pdfWidth: 400 }, + zh: { glob: "system/public/papers.aesthetic.computer/sucking-on-the-complex-26-arxiv-zh.pdf", pdfPage: 1, pdfWidth: 400 }, + ja: { glob: "system/public/papers.aesthetic.computer/sucking-on-the-complex-26-arxiv-ja.pdf", pdfPage: 1, pdfWidth: 400 }, + }, + body: ({ es, da, zh, ja }) => ` +
+
05 / 08 · translations
+
sucking on the complex
+
+ ${[["es", "español", PALETTE.cyan, es], ["da", "dansk", PALETTE.lime, da], ["zh", "中文", PALETTE.yellow, zh], ["ja", "日本語", PALETTE.accent, ja]].map(([code, label, color, src]) => ` +
+ ${src ? `` : `
`} +
${code} ${label}
+
`).join("")} +
+
+ reframed as engine, not adversary. +
+
`, + }, + "07_slab": ` +
+
06 / 08 · slab
+
menubar in swift
+
+ native appkit rewrite.
+ sf-symbol icons, passphrase modal,
+ mail submenu wired to mbsync + mu. +
+
+
826dnative appkit rewrite with sf-symbol icons
+
585dNotification hook fades ambient + TTS
+
+
`, + "08_oven": { + queries: { + ovenCommits: { commits: "oven", since: "48 hours ago", limit: 6 }, + recentPdfs: { files: "system/public/papers.aesthetic.computer/*.pdf", sinceHours: 48, limit: 60 }, + }, + body: ({ ovenCommits, recentPdfs }) => { + // Dedupe by paper stem (drop -26-arxiv-{cards,da,es,zh,ja}.pdf) so + // the chip strip lists distinct papers, not translation duplicates. + const stems = new Set(); + for (const p of recentPdfs || []) { + stems.add(p.split("/").pop().replace(/-26-arxiv(-cards|-da|-es|-zh|-ja)?\.pdf$/, "")); + } + const distinct = [...stems]; + return ` +
+
07 / 08 · oven
+
the oven, baking
+
+ pdfs compiled while we sleep —
+ ${distinct.length} distinct papers refreshed,
+ ${(recentPdfs || []).length} files in 48 hrs. +
+
+ ${(ovenCommits || []).slice(0, 4).map((c) => `
${c.hash.slice(0, 4)} ${c.subject.replace(/^\[?papers\]?[\s:]*/i, "").slice(0, 60)}
`).join("")} +
+
+ ${distinct.slice(0, 12).map((s) => `
${s.replace(/-/g, " ")}
`).join("")} +
+
`; + }, + }, + "09_outro": ` +
+
08 / 08 · outro
+
+
@fifi, you make the
+
keystrokes kinder —
+
every quiet line of code
+
is a little you-reminder.
+
+
`, + "10_end": ` +
+
+
aesthetic·computer
+
narrated by jeffrey-pvc · @jeffrey
+
made with love · 2026·04·25
+
`, + }, +}; + +export default audience; diff --git a/recap/bin/align.mjs b/recap/bin/align.mjs new file mode 100755 index 0000000000..495338e595 --- /dev/null +++ b/recap/bin/align.mjs @@ -0,0 +1,72 @@ +#!/usr/bin/env node +// align.mjs — match audience.segments[].marker against out/words.json word +// timestamps and produce out/segments.json: +// [{name, startSec, endSec, durationSec}, ...] +// Each marker is normalized (lowercase, punctuation stripped) and matched as +// a contiguous run of N words. Unmatched markers fail loud. +// Usage: node bin/align.mjs [audience-name] + +import { readFileSync, writeFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const audienceName = process.argv[2] || "fia"; +const { audience } = await import(`${ROOT}/audience/${audienceName}.mjs`); + +const words = JSON.parse(readFileSync(`${ROOT}/out/words.json`, "utf8")); +const norm = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim(); +const wordTokens = words.map((w) => norm(w.text)); +const audioEndMs = words[words.length - 1].toMs; + +function findMarker(marker, fromIdx) { + if (marker === "__END__") return -1; + const tokens = norm(marker).split(/\s+/); + for (let i = fromIdx; i <= wordTokens.length - tokens.length; i++) { + let ok = true; + for (let j = 0; j < tokens.length; j++) { + if (wordTokens[i + j] !== tokens[j]) { ok = false; break; } + } + if (ok) return i; + } + return -2; // not found +} + +const starts = []; +let cursor = 0; +for (const seg of audience.segments) { + if (seg.marker === "__END__") { + starts.push({ ...seg, idx: -1, startMs: audioEndMs }); + continue; + } + const idx = findMarker(seg.marker, cursor); + if (idx === -2) { + console.error(`✗ marker not found in transcript: "${seg.marker}" (segment ${seg.name})`); + console.error(` cursor at word ${cursor}/${wordTokens.length}: "${wordTokens.slice(cursor, cursor + 8).join(" ")}"`); + process.exit(1); + } + starts.push({ ...seg, idx, startMs: words[idx].fromMs }); + cursor = idx + 1; +} + +const trailing = (audience.segments[audience.segments.length - 1].trailingSilenceSec || 0) * 1000; +const endMs = audioEndMs + trailing; + +const segments = starts.map((s, i) => { + const next = i + 1 < starts.length ? starts[i + 1].startMs : endMs; + return { + name: s.name, + startSec: +(s.startMs / 1000).toFixed(3), + endSec: +(next / 1000).toFixed(3), + durationSec: +((next - s.startMs) / 1000).toFixed(3), + marker: s.marker, + }; +}); + +writeFileSync(`${ROOT}/out/segments.json`, JSON.stringify(segments, null, 2)); +console.log(`✓ ${ROOT}/out/segments.json`); +for (const s of segments) { + console.log(` ${s.name.padEnd(18)} ${String(s.startSec).padStart(6)}s → ${String(s.endSec).padStart(6)}s (${s.durationSec.toFixed(2)}s) "${s.marker}"`); +} +console.log(` audio ends at ${(audioEndMs / 1000).toFixed(2)}s · video ends at ${(endMs / 1000).toFixed(2)}s`); diff --git a/recap/bin/build-filter.mjs b/recap/bin/build-filter.mjs new file mode 100755 index 0000000000..2f123900d6 --- /dev/null +++ b/recap/bin/build-filter.mjs @@ -0,0 +1,47 @@ +#!/usr/bin/env node +// build-filter.mjs — emit the ffmpeg filter_complex graph for compose.fish. +// Reads out/subs.json and stitches one overlay per subtitle chunk into the +// video chain so each chunk appears only between its [startSec, endSec]. +// Subs sit in the lower-middle, above the waveform, with a translucent pill. +// Usage: node bin/build-filter.mjs (writes graph to stdout) + +import { readFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const TOTAL = process.argv[2]; +if (!TOTAL) { + console.error("usage: build-filter.mjs "); + process.exit(1); +} + +const subs = JSON.parse(readFileSync(`${ROOT}/out/subs.json`, "utf8")); + +// Subtitle band sits at y=1180 (just above the waveform at y=1720). +// Centered horizontally — sub PNGs are 1080×220 so x=0. +const SUB_Y = 1180; + +const lines = []; +lines.push(`[0:v]format=yuv420p,fps=30,scale=1080:1920,setsar=1[bg]`); +lines.push(`[1:a]apad=whole_dur=${TOTAL},asplit=2[a1][a2]`); +lines.push(`[a2]showwaves=s=1080x96:colors=0xff70d0|0x70f0e0:mode=cline:rate=30,format=rgba,colorchannelmixer=aa=0.55[wave]`); +lines.push(`[bg][wave]overlay=x=0:y=1920-200:format=auto[bg2]`); +lines.push(`[bg2]drawbox=x=0:y=1912:w='iw*t/${TOTAL}':h=8:color=0xff69b4:t=fill[v0]`); + +let prev = "v0"; +for (let i = 0; i < subs.length; i++) { + const s = subs[i]; + const srcLabel = `s${i}`; + const nextLabel = `v${i + 1}`; + // movie filter loads PNG with alpha; format=rgba ensures alpha is preserved. + lines.push(`movie='${s.file}':loop=0,setpts=N/(FRAME_RATE*TB),format=rgba[${srcLabel}]`); + lines.push(`[${prev}][${srcLabel}]overlay=x=0:y=${SUB_Y}:format=auto:enable='between(t,${s.startSec},${s.endSec})'[${nextLabel}]`); + prev = nextLabel; +} + +// Final stream needs the canonical [final] label for compose.fish -map. +lines.push(`[${prev}]null[final]`); + +process.stdout.write(lines.join(";\n") + "\n"); diff --git a/recap/bin/compose.fish b/recap/bin/compose.fish new file mode 100755 index 0000000000..57f4105957 --- /dev/null +++ b/recap/bin/compose.fish @@ -0,0 +1,40 @@ +#!/usr/bin/env fish +# compose.fish — final ffmpeg pass: concat slides + audio (with trailing +# silence) + waveform + animated progress bar + word-synced subtitles +# (loaded as movie sources, overlaid with enable=between(t,a,b)). +# Reads out/concat.txt, out/recap.mp3, out/duration.txt, out/subs.json. + +set -l ROOT (realpath (dirname (status -f))/..) +set -l OUT $ROOT/out +set -l TOTAL (cat $OUT/duration.txt) +set -l AUDIO $OUT/recap.mp3 +set -l VIDEO $OUT/recap.mp4 +set -l FILTER $OUT/filter.txt + +if not test -f $OUT/concat.txt + echo "✗ missing $OUT/concat.txt — run bin/slides.mjs first" + exit 1 +end +if not test -f $OUT/subs.json + echo "✗ missing $OUT/subs.json — run bin/subtitles.mjs first" + exit 1 +end + +echo "→ ffmpeg compose · $TOTAL s · 1080x1920" + +# Build the filter graph in node so we can splice in one overlay per subtitle +# chunk without fish escape gymnastics around brackets and quotes. +node $ROOT/bin/build-filter.mjs $TOTAL > $FILTER + +ffmpeg -hide_banner -y \ + -f concat -safe 0 -i $OUT/concat.txt \ + -i $AUDIO \ + -filter_complex_script $FILTER \ + -map "[final]" -map "[a1]" \ + -c:v libx264 -preset medium -crf 20 -pix_fmt yuv420p \ + -c:a aac -b:a 192k \ + -movflags +faststart \ + -t $TOTAL \ + $VIDEO + +echo "✓ $VIDEO" diff --git a/recap/bin/scout.mjs b/recap/bin/scout.mjs new file mode 100755 index 0000000000..1666c7f177 --- /dev/null +++ b/recap/bin/scout.mjs @@ -0,0 +1,120 @@ +#!/usr/bin/env node +// scout.mjs — resolve per-slide content queries from the audience config +// into base64 data URLs that slide HTML can embed inline. Supports: +// { glob: "" } → image +// { glob: "...pdf", pdfPage: 1, pdfWidth: 800 } → first PDF page +// { commits: "", since: "48 hours" } → commit list strings +// { files: "", since: "48 hours", limit: 8 } → recent matching paths +// Output: out/assets.json mapping slide-name → resolved values keyed by query name. +// Usage: node bin/scout.mjs [audience-name] + +import { execFileSync, execSync } from "node:child_process"; +import { mkdirSync, readFileSync, writeFileSync, statSync } from "node:fs"; +import { resolve, dirname, basename, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { tmpdir } from "node:os"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const REPO = resolve(ROOT, ".."); +const audienceName = process.argv[2] || "fia"; +const { audience } = await import(`${ROOT}/audience/${audienceName}.mjs`); + +const TMP = `${tmpdir()}/recap-pdf-${process.pid}`; +mkdirSync(TMP, { recursive: true }); + +function expandGlob(pattern) { + const abs = pattern.startsWith("/") ? pattern : join(REPO, pattern); + // Use shell glob expansion for simplicity. + try { + const out = execSync(`ls -1 ${abs} 2>/dev/null || true`, { encoding: "utf8" }); + return out.split("\n").map((s) => s.trim()).filter(Boolean); + } catch { return []; } +} + +function pdfPageToDataUrl(pdfPath, page = 1, width = 800) { + const stem = `${TMP}/${basename(pdfPath, ".pdf")}-p${page}`; + // pdftoppm uses 1-based page index; -scale-to fits the longer side. + execFileSync("pdftoppm", [ + "-png", "-r", "150", + "-f", String(page), "-l", String(page), + "-scale-to", String(width), + pdfPath, stem, + ]); + // pdftoppm produces "-1.png" or "-01.png" — find it. + const candidates = expandGlob(`${stem}-*.png`); + if (!candidates.length) throw new Error(`pdftoppm produced no output for ${pdfPath}`); + const buf = readFileSync(candidates[0]); + return `data:image/png;base64,${buf.toString("base64")}`; +} + +function imageToDataUrl(p) { + const buf = readFileSync(p); + const ext = p.toLowerCase().split(".").pop(); + const mime = ext === "svg" ? "image/svg+xml" : ext === "webp" ? "image/webp" : ext === "jpg" || ext === "jpeg" ? "image/jpeg" : "image/png"; + return `data:${mime};base64,${buf.toString("base64")}`; +} + +function recentCommits(grep, sinceArg, limit = 8) { + const since = sinceArg || "48 hours ago"; + const out = execSync( + `git -C ${REPO} log --since="${since}" -E --grep="${grep}" --pretty=format:"%h|%s" -n ${limit}`, + { encoding: "utf8" }, + ); + return out.split("\n").filter(Boolean).map((l) => { + const i = l.indexOf("|"); + return { hash: l.slice(0, i), subject: l.slice(i + 1) }; + }); +} + +function recentFiles(glob, sinceHours = 168, limit = 12) { + // mtime within sinceHours; sorted newest first. + const abs = glob.startsWith("/") ? glob : join(REPO, glob); + const sinceMs = Date.now() - sinceHours * 3600 * 1000; + const matches = expandGlob(abs) + .map((f) => ({ f, mtime: statSync(f).mtimeMs })) + .filter((x) => x.mtime >= sinceMs) + .sort((a, b) => b.mtime - a.mtime) + .slice(0, limit); + return matches.map((x) => x.f); +} + +function resolveQuery(name, q) { + if (q.glob) { + const matches = expandGlob(q.glob); + if (!matches.length) { + console.warn(` ⚠ ${name}: glob "${q.glob}" matched nothing`); + return null; + } + const file = matches[0]; + if (q.pdfPage) return pdfPageToDataUrl(file, q.pdfPage, q.pdfWidth || 800); + return imageToDataUrl(file); + } + if (q.commits) return recentCommits(q.commits, q.since, q.limit); + if (q.files) return recentFiles(q.files, q.sinceHours || 168, q.limit || 12); + console.warn(` ⚠ ${name}: unknown query shape`); + return null; +} + +const out = {}; +for (const seg of audience.segments) { + const slide = audience.slides[seg.name]; + if (!slide || typeof slide !== "object" || !slide.queries) continue; + console.log(`▸ ${seg.name}`); + out[seg.name] = {}; + for (const [name, q] of Object.entries(slide.queries)) { + const value = resolveQuery(name, q); + if (value !== null) { + out[seg.name][name] = value; + const desc = typeof value === "string" && value.startsWith("data:") + ? `${value.slice(0, value.indexOf(";"))} (${(value.length / 1024).toFixed(0)} KB b64)` + : Array.isArray(value) + ? `${value.length} items` + : String(value).slice(0, 60); + console.log(` ✓ ${name}: ${desc}`); + } + } +} + +writeFileSync(`${ROOT}/out/assets.json`, JSON.stringify(out, null, 2)); +console.log(`✓ ${ROOT}/out/assets.json · ${Object.keys(out).length} slide(s)`); diff --git a/recap/bin/slides.mjs b/recap/bin/slides.mjs new file mode 100755 index 0000000000..2dc901b77a --- /dev/null +++ b/recap/bin/slides.mjs @@ -0,0 +1,155 @@ +#!/usr/bin/env node +// slides.mjs — render audience slides as 1080x1920 PNGs and write +// out/concat.txt with per-slide durations from out/segments.json. +// Usage: node bin/slides.mjs [audience-name] + +import puppeteer from "/Users/jas/aesthetic-computer/oven/node_modules/puppeteer/lib/esm/puppeteer/puppeteer.js"; +import { mkdirSync, writeFileSync, readFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const audienceName = process.argv[2] || "fia"; +const { audience, PALETTE } = await import(`${ROOT}/audience/${audienceName}.mjs`); + +const SLIDE_DIR = `${ROOT}/out/slides`; +mkdirSync(SLIDE_DIR, { recursive: true }); + +const FONT_BOLD = "/Users/jas/aesthetic-computer/system/public/type/webfonts/ywft-processing-bold.ttf"; +const FONT_REG = "/Users/jas/aesthetic-computer/system/public/type/webfonts/ywft-processing-regular.ttf"; +const PALS_SVG = "/Users/jas/aesthetic-computer/system/public/purple-pals.svg"; + +const fontBoldB64 = readFileSync(FONT_BOLD).toString("base64"); +const fontRegB64 = readFileSync(FONT_REG).toString("base64"); +const palsSvgB64 = Buffer.from(readFileSync(PALS_SVG, "utf8"), "utf8").toString("base64"); + +const segments = JSON.parse(readFileSync(`${ROOT}/out/segments.json`, "utf8")); +const assets = (() => { + try { return JSON.parse(readFileSync(`${ROOT}/out/assets.json`, "utf8")); } + catch { return {}; } +})(); + +const cssTemplate = ` +@font-face { + font-family: 'ProcessingB'; + src: url(data:font/ttf;base64,${fontBoldB64}) format('truetype'); + unicode-range: U+0020-007E; +} +@font-face { + font-family: 'ProcessingR'; + src: url(data:font/ttf;base64,${fontRegB64}) format('truetype'); + unicode-range: U+0020-007E; +} +* { box-sizing: border-box; margin: 0; padding: 0; } +html, body { + width: 1080px; + height: 1920px; + font-family: 'ProcessingR', 'Helvetica Neue', 'Hiragino Sans', monospace; + -webkit-font-smoothing: antialiased; +} +body { background: ${PALETTE.bg}; padding: 80px 70px; position: relative; overflow: hidden; } +.frame { width: 100%; height: 100%; display: flex; flex-direction: column; gap: 28px; position: relative; } +.pals { + background-image: url(data:image/svg+xml;base64,${palsSvgB64}); + background-size: contain; background-repeat: no-repeat; background-position: center; +} +.pals.big { width: 720px; height: 720px; align-self: center; margin-top: 60px; } +.pals.med { width: 480px; height: 480px; align-self: center; margin-top: 80px; } +.pals.bug { position: absolute; bottom: 180px; right: 70px; width: 120px; height: 120px; opacity: 0.55; } +.kicker { font-family: 'ProcessingB'; font-size: 38px; letter-spacing: 8px; text-transform: uppercase; text-align: center; } +.huge { font-family: 'ProcessingB'; font-size: 240px; letter-spacing: -8px; text-align: center; line-height: 0.95; margin-top: 10px; } +.sub { font-family: 'ProcessingB'; font-size: 46px; letter-spacing: 2px; text-align: center; margin-top: 24px; } +.datestamp { position: absolute; bottom: 0; left: 0; right: 0; text-align: center; font-family: 'ProcessingR'; font-size: 30px; letter-spacing: 4px; } +.chapter { font-family: 'ProcessingR'; font-size: 30px; letter-spacing: 6px; text-transform: uppercase; } +.title { font-family: 'ProcessingB'; font-size: 130px; letter-spacing: -3px; line-height: 1.0; margin-top: 6px; } +.body { font-family: 'ProcessingR'; font-size: 56px; line-height: 1.32; margin-top: 32px; } +.body em { font-style: normal; color: ${PALETTE.magenta}; } +.body.small { font-size: 44px; } +.cap { font-family: 'ProcessingB'; font-size: 44px; margin-top: auto; padding-top: 30px; } +.commits { margin-top: 36px; display: flex; flex-direction: column; gap: 14px; } +.commit { font-family: 'ProcessingR'; font-size: 30px; color: ${PALETTE.off}; line-height: 1.3; } +.commit .hash { display: inline-block; width: 90px; color: ${PALETTE.lime}; font-family: 'ProcessingB'; } +.paper-card { margin-top: 50px; padding: 50px 40px; border: 4px solid ${PALETTE.cyan}; border-radius: 6px; background: rgba(112, 240, 224, 0.06); } +.paper-title { font-family: 'ProcessingB'; font-size: 74px; color: ${PALETTE.cyan}; letter-spacing: -1px; } +.paper-sub { font-family: 'ProcessingR'; font-size: 38px; color: ${PALETTE.cream}; margin-top: 18px; } +.paper-meta { font-family: 'ProcessingR'; font-size: 28px; color: ${PALETTE.dim}; margin-top: 14px; letter-spacing: 2px; } +.lang-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 32px; margin-top: 50px; } +.lang { font-family: 'ProcessingB'; font-size: 86px; color: ${PALETTE.cream}; padding: 36px 30px; border: 3px solid rgba(255,255,255,0.15); border-radius: 4px; display: flex; align-items: baseline; gap: 22px; } +.lang .flag { font-family: 'ProcessingB'; font-size: 60px; letter-spacing: -2px; } +.ticker { margin-top: 40px; display: flex; flex-direction: column; gap: 18px; padding: 30px 36px; background: rgba(255, 216, 96, 0.07); border-left: 6px solid ${PALETTE.yellow}; } +.tick { font-family: 'ProcessingR'; font-size: 32px; color: ${PALETTE.cream}; } +.rhyme { margin-top: 220px; text-align: center; display: flex; flex-direction: column; gap: 30px; } +.rhyme .line1, .rhyme .line2 { font-family: 'ProcessingB'; font-size: 92px; line-height: 1.05; } +.rhyme .line2 { margin-top: 60px; } +.endline { font-family: 'ProcessingB'; font-size: 92px; letter-spacing: -2px; text-align: center; margin-top: 40px; } +.endsub { font-family: 'ProcessingR'; font-size: 36px; letter-spacing: 4px; text-align: center; margin-top: 14px; } +.cornerbug { position: absolute; bottom: 200px; left: 70px; font-family: 'ProcessingR'; font-size: 22px; color: ${PALETTE.dim}; letter-spacing: 4px; text-transform: uppercase; } + +/* asset-driven slide additions */ +.title-row { display: flex; align-items: center; gap: 24px; } +.brand-icon { width: 110px; height: 110px; border-radius: 18px; flex-shrink: 0; } +.row-with-aside { display: flex; gap: 30px; margin-top: 36px; align-items: flex-start; } +.row-with-aside .commits { flex: 1; margin-top: 0; } +.paper-thumb { width: 280px; border: 3px solid ${PALETTE.cyan}; border-radius: 6px; box-shadow: 0 6px 30px rgba(0,0,0,0.4); } +.cover-grid { display: grid; grid-template-columns: 1fr 1fr; gap: 28px; margin-top: 40px; } +.cover { display: flex; flex-direction: column; gap: 14px; align-items: center; } +.cover-img { width: 100%; max-width: 360px; border-radius: 6px; box-shadow: 0 4px 20px rgba(0,0,0,0.5); } +.cover-img.placeholder { aspect-ratio: 0.71; background: rgba(255,255,255,0.06); border: 2px dashed rgba(255,255,255,0.2); } +.cover-label { font-family: 'ProcessingB'; font-size: 38px; color: ${PALETTE.cream}; } +.cover-label span { font-size: 32px; margin-right: 10px; letter-spacing: -1px; } +.pdf-strip { margin-top: 30px; display: flex; flex-wrap: wrap; gap: 12px; } +.pdf-chip { font-family: 'ProcessingR'; font-size: 24px; color: ${PALETTE.cream}; padding: 10px 18px; background: rgba(255, 216, 96, 0.1); border: 2px solid rgba(255, 216, 96, 0.3); border-radius: 4px; } +.tick .hash { color: ${PALETTE.lime}; font-family: 'ProcessingB'; margin-right: 14px; } +`; + +const browser = await puppeteer.launch({ + executablePath: "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", + args: ["--no-sandbox"], +}); + +const slidesOrder = audience.segments.map((s) => s.name); +const showBug = (name) => name !== slidesOrder[0] && name !== slidesOrder[slidesOrder.length - 1]; + +for (const name of slidesOrder) { + const slide = audience.slides[name]; + if (!slide) { + console.error(`✗ no slide HTML for segment "${name}" in audience.slides`); + process.exit(1); + } + // Slide can be a string or an object { body, queries }; body can also be a + // function (assets) => string for query-driven slides. + const slideAssets = assets[name] || {}; + let body; + if (typeof slide === "string") body = slide; + else if (typeof slide.body === "function") body = slide.body(slideAssets); + else body = slide.body; + const page = await browser.newPage(); + await page.setViewport({ width: 1080, height: 1920, deviceScaleFactor: 1 }); + const html = `${body} + ${showBug(name) ? `
` : ""} + ${showBug(name) ? `
aesthetic·computer · for ${audience.handle || audience.name}
` : ""} + `; + await page.setContent(html, { waitUntil: "networkidle0" }); + await new Promise((r) => setTimeout(r, 200)); + const png = await page.screenshot({ type: "png", omitBackground: false }); + writeFileSync(`${SLIDE_DIR}/${name}.png`, png); + await page.close(); + const seg = segments.find((s) => s.name === name); + console.log(`✓ ${name}.png · ${seg.durationSec.toFixed(2)}s (${seg.startSec}s → ${seg.endSec}s)`); +} +await browser.close(); + +// concat.txt with real durations +const lines = []; +for (const seg of segments) { + lines.push(`file '${SLIDE_DIR}/${seg.name}.png'`); + lines.push(`duration ${seg.durationSec}`); +} +// concat demuxer needs the last file repeated without duration for proper end +lines.push(`file '${SLIDE_DIR}/${segments[segments.length - 1].name}.png'`); +writeFileSync(`${ROOT}/out/concat.txt`, lines.join("\n") + "\n"); + +const total = segments[segments.length - 1].endSec; +writeFileSync(`${ROOT}/out/duration.txt`, String(total)); +console.log(`✓ ${ROOT}/out/concat.txt · total ${total}s`); diff --git a/recap/bin/subtitles.mjs b/recap/bin/subtitles.mjs new file mode 100755 index 0000000000..4338ac7408 --- /dev/null +++ b/recap/bin/subtitles.mjs @@ -0,0 +1,126 @@ +#!/usr/bin/env node +// subtitles.mjs — group out/words.json into readable phrase chunks (~5 words, +// breaking on long pauses or punctuation), render each as a 1080×220 +// transparent PNG with a pill backdrop, and write out/subs.json: +// [{file, startSec, endSec, text}, ...] +// Compose.fish overlays each PNG at its time window using ffmpeg's `movie` +// + `overlay enable=between(t,a,b)` chain. +// Usage: node bin/subtitles.mjs + +import puppeteer from "/Users/jas/aesthetic-computer/oven/node_modules/puppeteer/lib/esm/puppeteer/puppeteer.js"; +import { mkdirSync, writeFileSync, readFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const SUB_DIR = `${ROOT}/out/subs`; +mkdirSync(SUB_DIR, { recursive: true }); + +const audienceName = process.argv[2] || "fia"; +const { audience } = await import(`${ROOT}/audience/${audienceName}.mjs`); +const fixes = audience.transcriptFixes || {}; +function escapeRegex(s) { return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } +function applyFixes(text) { + let out = text; + for (const [find, replace] of Object.entries(fixes)) { + out = out.replace(new RegExp(escapeRegex(find), "gi"), replace); + } + return out; +} + +const FONT_BOLD = "/Users/jas/aesthetic-computer/system/public/type/webfonts/ywft-processing-bold.ttf"; +const fontBoldB64 = readFileSync(FONT_BOLD).toString("base64"); + +const words = JSON.parse(readFileSync(`${ROOT}/out/words.json`, "utf8")); + +const MAX_WORDS = 6; +const MAX_GAP_MS = 380; // pause longer than this triggers a chunk break +const SENTENCE_END = /[.!?,—]$/; + +const chunks = []; +let cur = []; +let curStart = 0; +for (let i = 0; i < words.length; i++) { + const w = words[i]; + if (cur.length === 0) curStart = w.fromMs; + cur.push(w); + + const next = words[i + 1]; + const reachedMax = cur.length >= MAX_WORDS; + const longPause = next && next.fromMs - w.toMs > MAX_GAP_MS; + const sentenceEnd = SENTENCE_END.test(w.text.trim()) && cur.length >= 3; + const isLast = !next; + + if (reachedMax || longPause || sentenceEnd || isLast) { + chunks.push({ + startMs: curStart, + endMs: next ? next.fromMs : w.toMs, + text: applyFixes(cur.map((x) => x.text.trim()).join(" ")), + }); + cur = []; + } +} + +const cssTemplate = ` +@font-face { + font-family: 'ProcessingB'; + src: url(data:font/ttf;base64,${fontBoldB64}) format('truetype'); + unicode-range: U+0020-007E; +} +* { box-sizing: border-box; margin: 0; padding: 0; } +html, body { width: 1080px; height: 220px; background: transparent; -webkit-font-smoothing: antialiased; } +.wrap { width: 100%; height: 100%; display: flex; align-items: center; justify-content: center; padding: 0 60px; } +.pill { + background: rgba(16, 8, 32, 0.72); + backdrop-filter: blur(2px); + border: 3px solid rgba(255, 105, 180, 0.55); + border-radius: 14px; + padding: 26px 44px; + max-width: 100%; + text-align: center; + font-family: 'ProcessingB', 'Helvetica Neue', 'Hiragino Sans', monospace; + font-size: 64px; + line-height: 1.15; + color: #fcf7c5; + letter-spacing: -1px; + text-shadow: 0 2px 0 rgba(0,0,0,0.5); + word-wrap: break-word; +} +.pill em { font-style: normal; color: #ff70d0; } +`; + +const browser = await puppeteer.launch({ + executablePath: "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", + args: ["--no-sandbox"], +}); + +const out = []; +for (let i = 0; i < chunks.length; i++) { + const c = chunks[i]; + const file = `${SUB_DIR}/${String(i).padStart(3, "0")}.png`; + const page = await browser.newPage(); + await page.setViewport({ width: 1080, height: 220, deviceScaleFactor: 1 }); + const html = `
${escapeHtml(c.text)}
`; + await page.setContent(html, { waitUntil: "networkidle0" }); + await new Promise((r) => setTimeout(r, 80)); + const png = await page.screenshot({ type: "png", omitBackground: true }); + writeFileSync(file, png); + await page.close(); + out.push({ + file, + startSec: +(c.startMs / 1000).toFixed(3), + endSec: +(c.endMs / 1000).toFixed(3), + text: c.text, + }); +} +await browser.close(); + +writeFileSync(`${ROOT}/out/subs.json`, JSON.stringify(out, null, 2)); +console.log(`✓ ${out.length} subtitle chunks → ${SUB_DIR}/`); +for (const s of out.slice(0, 5)) console.log(` ${s.startSec.toFixed(2)}-${s.endSec.toFixed(2)} "${s.text}"`); +if (out.length > 5) console.log(` ... (+${out.length - 5} more)`); + +function escapeHtml(s) { + return s.replace(/&/g, "&").replace(//g, ">").replace(/"/g, """); +} diff --git a/recap/bin/transcribe.mjs b/recap/bin/transcribe.mjs new file mode 100755 index 0000000000..c0eca49c68 --- /dev/null +++ b/recap/bin/transcribe.mjs @@ -0,0 +1,38 @@ +#!/usr/bin/env node +// transcribe.mjs — run whisper-cli on out/recap.mp3 and emit out/words.json +// in a flat shape: [{text, fromMs, toMs}, ...]. +// Usage: node bin/transcribe.mjs + +import { execFileSync } from "node:child_process"; +import { readFileSync, writeFileSync, existsSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const MP3 = `${ROOT}/out/recap.mp3`; +const MODEL = `${ROOT}/models/ggml-base.en.bin`; + +if (!existsSync(MP3)) { + console.error(`✗ missing ${MP3} — run bin/tts.mjs first`); + process.exit(1); +} +if (!existsSync(MODEL)) { + console.error(`✗ missing ${MODEL} — download with:\n curl -L -o ${MODEL} https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin`); + process.exit(1); +} + +console.log(`→ whisper-cli · ${MP3}`); +execFileSync( + "whisper-cli", + ["-m", MODEL, "-f", MP3, "-ojf", "-of", `${ROOT}/out/recap`, "--max-len", "1", "-ml", "1", "-sow"], + { stdio: ["ignore", "ignore", "inherit"] }, +); + +const raw = JSON.parse(readFileSync(`${ROOT}/out/recap.json`, "utf8")); +const words = raw.transcription + .map((s) => ({ text: s.text.trim(), fromMs: s.offsets.from, toMs: s.offsets.to })) + .filter((w) => w.text.length > 0); + +writeFileSync(`${ROOT}/out/words.json`, JSON.stringify(words, null, 2)); +console.log(`✓ ${ROOT}/out/words.json · ${words.length} words · ${(words[words.length - 1].toMs / 1000).toFixed(2)}s`); diff --git a/recap/bin/tts.mjs b/recap/bin/tts.mjs new file mode 100755 index 0000000000..66dc060336 --- /dev/null +++ b/recap/bin/tts.mjs @@ -0,0 +1,37 @@ +#!/usr/bin/env node +// tts.mjs — POST audience narration to /api/say, save MP3 to out/recap.mp3. +// Usage: node bin/tts.mjs [audience-name] (default: fia) + +import { writeFileSync } from "node:fs"; +import { resolve, dirname } from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const ROOT = resolve(HERE, ".."); +const audienceName = process.argv[2] || "fia"; + +const { audience } = await import(`${ROOT}/audience/${audienceName}.mjs`); + +const body = { + from: audience.narration, + provider: audience.voice.provider, + voice: audience.voice.voice, +}; + +console.log(`→ POST /api/say · ${audience.narration.length} chars · ${audience.voice.provider}`); +const res = await fetch("https://aesthetic.computer/api/say", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body), + redirect: "follow", +}); + +if (!res.ok) { + console.error(`✗ /api/say returned ${res.status}: ${await res.text()}`); + process.exit(1); +} + +const buf = Buffer.from(await res.arrayBuffer()); +const out = `${ROOT}/out/recap.mp3`; +writeFileSync(out, buf); +console.log(`✓ ${out} (${(buf.length / 1024).toFixed(0)} KB)`); diff --git a/recap/pipeline.fish b/recap/pipeline.fish new file mode 100755 index 0000000000..cd6e79360f --- /dev/null +++ b/recap/pipeline.fish @@ -0,0 +1,40 @@ +#!/usr/bin/env fish +# pipeline.fish — full end-to-end recap build. +# Usage: ./pipeline.fish [audience-name] (default: fia) +# ./pipeline.fish fia --skip-tts (reuse existing out/recap.mp3) + +set -l ROOT (realpath (dirname (status -f))) +set -l AUDIENCE $argv[1] +test -z "$AUDIENCE"; and set AUDIENCE fia +set -l SKIP_TTS 0 +contains -- --skip-tts $argv; and set SKIP_TTS 1 + +cd $ROOT + +echo "━━━ recap pipeline · audience=$AUDIENCE ━━━" + +if test $SKIP_TTS -eq 0 + echo "▸ 1/6 tts" + node bin/tts.mjs $AUDIENCE; or exit 1 +else + echo "▸ 1/6 tts (skipped — reusing out/recap.mp3)" +end + +echo "▸ 2/6 transcribe + align" +node bin/transcribe.mjs; or exit 1 +node bin/align.mjs $AUDIENCE; or exit 1 + +echo "▸ 3/6 scout (resolve per-slide content queries)" +node bin/scout.mjs $AUDIENCE; or exit 1 + +echo "▸ 4/6 slides" +node bin/slides.mjs $AUDIENCE; or exit 1 + +echo "▸ 5/6 subtitles" +node bin/subtitles.mjs $AUDIENCE; or exit 1 + +echo "▸ 6/6 compose" +fish bin/compose.fish; or exit 1 + +echo "━━━ done · $ROOT/out/recap.mp4 ━━━" +ls -lh $ROOT/out/recap.mp4 -- 2.51.2