From 3ee8c0da61a80247fe139e9d38b42a36287aa1da Mon Sep 17 00:00:00 2001 From: Steve Date: Mon, 7 Sep 2026 22:06:33 -0400 Subject: [PATCH] fix: closes #62 --- packages/cli/src/commands/inject.ts | 337 +++++++++++++++++++++++----- packages/cli/test/inject.test.ts | 205 ++++++++++++++++- 2 files changed, 485 insertions(+), 57 deletions(-) diff --git a/packages/cli/src/commands/inject.ts b/packages/cli/src/commands/inject.ts index 495a557..4b36551 100644 --- a/packages/cli/src/commands/inject.ts +++ b/packages/cli/src/commands/inject.ts @@ -1,14 +1,15 @@ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; import { log } from "@clack/prompts"; import { command, flag, option, optional, string } from "cmd-ts"; import { glob } from "glob"; -import * as fs from "node:fs/promises"; -import * as path from "node:path"; import { + DEFAULT_PUBLISHER_CONFIG, findConfig, loadConfig, loadState, - DEFAULT_PUBLISHER_CONFIG, } from "../lib/config"; +import type { PublisherState } from "../lib/types"; export const injectCommand = command({ name: "inject", @@ -49,36 +50,17 @@ export const injectCommand = command({ // Load state to get atUri mappings const state = await loadState(configDir); - // Build a map of slug to atUri from state - // The slug is stored in state by the publish command, using the configured slug options - const slugToAtUri = new Map(); - for (const [filePath, postState] of Object.entries(state.posts)) { - if (postState.atUri && postState.slug) { - // Use the slug stored in state (computed by publish with config options) - slugToAtUri.set(postState.slug, postState.atUri); - - // Also add the last segment for simpler matching - // e.g., "other/my-other-post" -> also map "my-other-post" - const lastSegment = postState.slug.split("/").pop(); - if (lastSegment && lastSegment !== postState.slug) { - slugToAtUri.set(lastSegment, postState.atUri); - } - } else if (postState.atUri) { - // Fallback for older state files without slug field - // Extract slug from file path (e.g., ./content/blog/my-post.md -> my-post) - const basename = path.basename(filePath, path.extname(filePath)); - slugToAtUri.set(basename.toLowerCase(), postState.atUri); - } - } + // Published posts, keyed in state by their source file path + const posts = collectPublishedPosts(state); - if (slugToAtUri.size === 0) { + if (posts.length === 0) { log.warn( "No published posts found in state. Run 'sequoia publish' first.", ); return; } - log.info(`Found ${slugToAtUri.size} slug mappings from published posts`); + log.info(`Found ${posts.length} published posts in state`); // Scan for HTML files const htmlFiles = await glob("**/*.html", { @@ -93,46 +75,36 @@ export const injectCommand = command({ log.info(`Found ${htmlFiles.length} HTML files`); + // The source file path is stored relative to the config directory, so the + // content directory has to be expressed the same way to line the two up. + const contentDir = path.isAbsolute(config.contentDir) + ? path.relative(configDir, config.contentDir) + : config.contentDir; + + const { matches, warnings } = matchPostsToHtmlFiles(htmlFiles, posts, { + contentDir, + pathPrefix: config.pathPrefix, + pathTemplate: config.pathTemplate, + }); + + for (const warning of warnings) { + log.warn(` ${warning}`); + } + let injectedCount = 0; let skippedCount = 0; let alreadyHasCount = 0; for (const file of htmlFiles) { - const htmlPath = path.join(resolvedOutputDir, file); - // Try to match this HTML file to a published post - const relativePath = file; - const htmlDir = path.dirname(relativePath); - const htmlBasename = path.basename(relativePath, ".html"); - - // Try different matching strategies - let atUri: string | undefined; - - // Strategy 1: Direct basename match (e.g., my-post.html -> my-post) - atUri = slugToAtUri.get(htmlBasename); - - // Strategy 2: For index.html, try the directory path - // e.g., posts/40th-puzzle-box/what-a-gift/index.html -> 40th-puzzle-box/what-a-gift - if (!atUri && htmlBasename === "index" && htmlDir !== ".") { - // Try full directory path (for nested subdirectories) - atUri = slugToAtUri.get(htmlDir); - - // Also try just the last directory segment - if (!atUri) { - const lastDir = path.basename(htmlDir); - atUri = slugToAtUri.get(lastDir); - } - } - - // Strategy 3: Full path match (e.g., blog/my-post.html -> blog/my-post) - if (!atUri && htmlDir !== ".") { - atUri = slugToAtUri.get(`${htmlDir}/${htmlBasename}`); - } - + const atUri = matches.get(file); if (!atUri) { skippedCount++; continue; } + const htmlPath = path.join(resolvedOutputDir, file); + const relativePath = file; + // Read the HTML file let content = await fs.readFile(htmlPath, "utf-8"); @@ -180,6 +152,259 @@ export const injectCommand = command({ }, }); +/** A post from state that has been published and therefore has an AT URI. */ +export interface PublishedPost { + /** Source file path, as stored in state (relative to the config directory). */ + filePath: string; + /** Slug recorded by `publish` (or derived from the filename for old state). */ + slug: string; + atUri: string; +} + +export interface MatchOptions { + /** Content directory, relative to the config directory (e.g. "./content"). */ + contentDir: string; + pathPrefix?: string; + pathTemplate?: string; +} + +export interface MatchResult { + /** HTML file (relative to the output directory) -> AT URI to inject. */ + matches: Map; + warnings: string[]; +} + +// Candidate routes are ranked so the most specific match wins. A slug like +// "blog" can match both a section index (blog/index.html) and the article +// itself (blog/blog/index.html); only the latter is the generated post path. +const RANK_CONFIGURED_PATH = 4; // pathPrefix + slug, i.e. the canonical URL +const RANK_GENERATED_PATH = 3; // source directory + slug +const RANK_SLUG = 2; // the slug on its own +const RANK_SLUG_SEGMENT = 1; // the last segment of the slug, anywhere in the tree + +interface RouteCandidate { + route: string; + rank: number; + /** + * "route" matches the whole route a file is built to; "segment" matches only + * its last segment, for sites whose output layout follows neither the slug + * nor the content tree. + */ + match: "route" | "segment"; +} + +export function collectPublishedPosts(state: PublisherState): PublishedPost[] { + const posts: PublishedPost[] = []; + + for (const [filePath, postState] of Object.entries(state.posts)) { + if (!postState.atUri) continue; + + // Older state files have no slug field, so fall back to the filename + // (e.g. ./content/blog/my-post.md -> my-post) + const slug = + postState.slug ?? + path.basename(filePath, path.extname(filePath)).toLowerCase(); + + posts.push({ filePath, slug, atUri: postState.atUri }); + } + + return posts; +} + +/** + * Match built HTML files to published posts. + * + * Every post claims only the files matching its most specific candidate route, + * so a document is never injected into a page that merely shares a name with it. + */ +export function matchPostsToHtmlFiles( + htmlFiles: string[], + posts: PublishedPost[], + options: MatchOptions, +): MatchResult { + const filesByRoute = new Map(); + const filesBySegment = new Map(); + for (const file of htmlFiles) { + const route = getHtmlRoute(file); + addToIndex(filesByRoute, route, file); + const lastSegment = route.split("/").pop(); + if (lastSegment) { + addToIndex(filesBySegment, lastSegment, file); + } + } + + const claims = new Map< + string, + Array<{ rank: number; post: PublishedPost }> + >(); + for (const post of posts) { + for (const candidate of getRouteCandidates(post, options)) { + const index = + candidate.match === "segment" ? filesBySegment : filesByRoute; + const files = index.get(candidate.route); + if (!files) continue; + + for (const file of files) { + const fileClaims = claims.get(file); + if (fileClaims) { + fileClaims.push({ rank: candidate.rank, post }); + } else { + claims.set(file, [{ rank: candidate.rank, post }]); + } + } + + // Stop at the most specific route this post matches + break; + } + } + + const matches = new Map(); + const warnings: string[] = []; + + for (const [file, fileClaims] of claims) { + const best = fileClaims.reduce((a, b) => (b.rank > a.rank ? b : a)); + const rivals = fileClaims.filter( + (claim) => + claim.rank === best.rank && claim.post.atUri !== best.post.atUri, + ); + + if (rivals.length > 0) { + const slugs = [best, ...rivals] + .map((claim) => claim.post.slug) + .join(", "); + warnings.push( + `Multiple published posts match ${file} (${slugs}), skipping`, + ); + continue; + } + + matches.set(file, best.post.atUri); + } + + return { matches, warnings }; +} + +/** The route a built HTML file serves, without leading or trailing slashes. */ +export function getHtmlRoute(htmlFile: string): string { + const withoutExtension = normalizeRoute(htmlFile).replace(/\.html?$/i, ""); + + // posts/my-post/index.html and posts/my-post.html serve the same route + if (path.posix.basename(withoutExtension) === "index") { + return normalizeRoute(path.posix.dirname(withoutExtension)); + } + + return withoutExtension; +} + +/** Routes a post could have been built to, most specific first. */ +function getRouteCandidates( + post: PublishedPost, + options: MatchOptions, +): RouteCandidate[] { + const slug = normalizeRoute(post.slug); + const lastSegment = slug.split("/").pop() ?? slug; + const candidates: RouteCandidate[] = []; + + // A path template can only be resolved with the full post, which is not in + // state, so fall back to slug matching for those sites. + if (!options.pathTemplate) { + const prefix = normalizeRoute( + options.pathPrefix ?? DEFAULT_PUBLISHER_CONFIG.pathPrefix, + ); + if (prefix) { + candidates.push({ + route: `${prefix}/${slug}`, + rank: RANK_CONFIGURED_PATH, + match: "route", + }); + } + } + + // Most generators build a page next to its siblings in the content tree, + // with the slug replacing the last path segment: content/blog/2024-12-17-blog + // with slug "blog" is built to blog/blog/index.html. + const sectionDir = getSectionDir(post.filePath, options.contentDir); + if (sectionDir !== undefined) { + candidates.push({ + route: sectionDir ? `${sectionDir}/${lastSegment}` : lastSegment, + rank: RANK_GENERATED_PATH, + match: "route", + }); + } + + candidates.push({ route: slug, rank: RANK_SLUG, match: "route" }); + candidates.push({ + route: lastSegment, + rank: RANK_SLUG_SEGMENT, + match: "segment", + }); + + // Keep the highest rank for routes reached more than one way + const byRoute = new Map(); + for (const candidate of candidates) { + const key = `${candidate.match}:${candidate.route}`; + const existing = byRoute.get(key); + if (!existing || candidate.rank > existing.rank) { + byRoute.set(key, candidate); + } + } + + return [...byRoute.values()].sort((a, b) => b.rank - a.rank); +} + +/** + * Directory a source file's page is built into, relative to the content + * directory. Page bundles (index.md/_index.md) resolve to their parent. + * Returns undefined when the file is not inside the content directory. + */ +function getSectionDir( + filePath: string, + contentDir: string, +): string | undefined { + const normalizedFile = normalizeRoute(filePath); + const normalizedContentDir = normalizeRoute(contentDir); + + let relative = normalizedFile; + if (normalizedContentDir) { + relative = path.posix.relative(normalizedContentDir, normalizedFile); + if (relative.startsWith("../")) return undefined; + } + + let dir = path.posix.dirname(relative); + const basename = path.posix + .basename(relative) + .replace(/\.[^.]+$/, "") + .toLowerCase(); + if (basename === "index" || basename === "_index") { + dir = path.posix.dirname(dir); + } + + return normalizeRoute(dir); +} + +function addToIndex( + index: Map, + key: string, + file: string, +): void { + const files = index.get(key); + if (files) { + files.push(file); + } else { + index.set(key, [file]); + } +} + +/** Strip "./", duplicate slashes, and leading/trailing slashes from a path. */ +function normalizeRoute(value: string): string { + const normalized = value + .replace(/\\/g, "/") + .replace(/\/+/g, "/") + .replace(/^\.\//, "") + .replace(/^\/+|\/+$/g, ""); + return normalized === "." ? "" : normalized; +} + export enum Injected { AlreadyPresent = 0, Skipped, diff --git a/packages/cli/test/inject.test.ts b/packages/cli/test/inject.test.ts index 93bbc02..2982337 100644 --- a/packages/cli/test/inject.test.ts +++ b/packages/cli/test/inject.test.ts @@ -1,6 +1,11 @@ import { describe, expect, it, spyOn } from "bun:test"; import { log } from "@clack/prompts"; -import { Injected, injectLinkTags } from "../src/commands/inject"; +import { + collectPublishedPosts, + Injected, + injectLinkTags, + matchPostsToHtmlFiles, +} from "../src/commands/inject"; const atUri = "at://did:plc:abc123/app.bsky.feed.post/xyz"; const publicationUri = "at://did:plc:def456/app.bsky.feed.generator/main"; @@ -139,3 +144,201 @@ describe("injectLinkTags", () => { }); }); }); + +describe("matchPostsToHtmlFiles", () => { + const otherAtUri = "at://did:plc:abc123/site.standard.document/other"; + + it("prefers the generated post path over a section index with the same slug", () => { + // https://tangled.org/stevedylan.dev/sequoia/issues/62 + // Zola: content/blog/2024-12-17-blog/index.md with slug = "blog" + // builds to public/blog/blog/index.html, alongside the section index + // public/blog/index.html + const { matches } = matchPostsToHtmlFiles( + ["index.html", "blog/index.html", "blog/blog/index.html"], + [ + { + filePath: "content/blog/2024-12-17-blog/index.md", + slug: "blog", + atUri, + }, + ], + { contentDir: "./content", pathPrefix: "/blog" }, + ); + + expect(matches.get("blog/blog/index.html")).toBe(atUri); + expect(matches.has("blog/index.html")).toBe(false); + expect(matches.has("index.html")).toBe(false); + }); + + it("matches the generated path when pathPrefix does not describe the output", () => { + const { matches } = matchPostsToHtmlFiles( + ["blog/index.html", "blog/blog/index.html"], + [ + { + filePath: "content/blog/2024-12-17-blog/index.md", + slug: "blog", + atUri, + }, + ], + { contentDir: "./content", pathPrefix: "/posts" }, + ); + + expect(matches.get("blog/blog/index.html")).toBe(atUri); + expect(matches.has("blog/index.html")).toBe(false); + }); + + it("matches the configured pathPrefix path", () => { + const { matches } = matchPostsToHtmlFiles( + ["posts/my-post/index.html", "my-post/index.html"], + [{ filePath: "content/my-post.md", slug: "my-post", atUri }], + { contentDir: "./content", pathPrefix: "/posts" }, + ); + + expect(matches.get("posts/my-post/index.html")).toBe(atUri); + expect(matches.has("my-post/index.html")).toBe(false); + }); + + it("matches flat HTML files", () => { + const { matches } = matchPostsToHtmlFiles( + ["posts/my-post.html"], + [{ filePath: "content/my-post.md", slug: "my-post", atUri }], + { contentDir: "./content", pathPrefix: "/posts" }, + ); + + expect(matches.get("posts/my-post.html")).toBe(atUri); + }); + + it("matches nested slugs", () => { + const { matches } = matchPostsToHtmlFiles( + ["other/my-other-post/index.html"], + [ + { + filePath: "content/other/my-other-post.md", + slug: "other/my-other-post", + atUri, + }, + ], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.get("other/my-other-post/index.html")).toBe(atUri); + }); + + it("falls back to the slug when the output layout is unrelated to the source", () => { + const { matches } = matchPostsToHtmlFiles( + ["archive/my-post/index.html"], + [{ filePath: "content/2024/my-post.md", slug: "archive/my-post", atUri }], + { contentDir: "./content" }, + ); + + expect(matches.get("archive/my-post/index.html")).toBe(atUri); + }); + + it("falls back to the last slug segment", () => { + const { matches } = matchPostsToHtmlFiles( + ["writing/my-post/index.html"], + [{ filePath: "content/blog/my-post.md", slug: "blog/my-post", atUri }], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.get("writing/my-post/index.html")).toBe(atUri); + }); + + it("skips files two posts match equally well", () => { + const { matches, warnings } = matchPostsToHtmlFiles( + ["writing/my-post/index.html"], + [ + { filePath: "content/blog/my-post.md", slug: "blog/my-post", atUri }, + { + filePath: "content/notes/my-post.md", + slug: "notes/my-post", + atUri: otherAtUri, + }, + ], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.size).toBe(0); + expect(warnings).toHaveLength(1); + expect(warnings[0]).toContain("writing/my-post/index.html"); + }); + + it("still matches the better of two posts claiming one file", () => { + const { matches, warnings } = matchPostsToHtmlFiles( + ["blog/my-post/index.html", "notes/my-post/index.html"], + [ + { filePath: "content/blog/my-post.md", slug: "blog/my-post", atUri }, + { + filePath: "content/notes/my-post.md", + slug: "notes/my-post", + atUri: otherAtUri, + }, + ], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.get("blog/my-post/index.html")).toBe(atUri); + expect(matches.get("notes/my-post/index.html")).toBe(otherAtUri); + expect(warnings).toHaveLength(0); + }); + + it("falls back to the last slug segment for a flat output layout", () => { + const { matches } = matchPostsToHtmlFiles( + ["my-post.html"], + [{ filePath: "content/blog/my-post.md", slug: "blog/my-post", atUri }], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.get("my-post.html")).toBe(atUri); + }); + + it("matches a templated path the slug alone does not describe", () => { + const { matches } = matchPostsToHtmlFiles( + ["2024/12/my-post/index.html"], + [{ filePath: "content/2024/my-post.md", slug: "my-post", atUri }], + { + contentDir: "./content", + pathPrefix: "/posts", + pathTemplate: "{year}/{month}/{slug}", + }, + ); + + expect(matches.get("2024/12/my-post/index.html")).toBe(atUri); + }); + + it("matches a root index.html for a top-level page", () => { + const { matches } = matchPostsToHtmlFiles( + ["index.html", "about/index.html"], + [{ filePath: "content/about.md", slug: "about", atUri }], + { contentDir: "./content", pathPrefix: "" }, + ); + + expect(matches.get("about/index.html")).toBe(atUri); + expect(matches.has("index.html")).toBe(false); + }); +}); + +describe("collectPublishedPosts", () => { + it("skips unpublished posts and derives a slug from old state entries", () => { + const posts = collectPublishedPosts({ + posts: { + "./content/blog/My-Post.md": { contentHash: "a", atUri }, + "./content/blog/other.md": { + contentHash: "b", + atUri: "at://did:plc:abc123/site.standard.document/other", + slug: "blog/other", + }, + "./content/blog/draft.md": { contentHash: "c" }, + }, + }); + + expect(posts).toEqual([ + { filePath: "./content/blog/My-Post.md", slug: "my-post", atUri }, + { + filePath: "./content/blog/other.md", + slug: "blog/other", + atUri: "at://did:plc:abc123/site.standard.document/other", + }, + ]); + }); +}); -- 2.51.2