diff --git a/README.md b/README.md index 42ee149..637866e 100644 --- a/README.md +++ b/README.md @@ -39,9 +39,9 @@ Notes are plain markdown: CommonMark and GFM, plus the Obsidian extensions peopl | Backlinks | ✅ | ✅ | — | | YouTube embed from a bare URL | ✅ | ✅ | — | | Transclusion `![[note]]` | — | ✅ | — | -| Block references `[[note#^id]]` | — | ✅ | — | +| Block references `[[note#^id]]` | ✅ ids you type | ✅ auto-minted too | — | | Tags `#tag` | — | ✅ | — | -| YAML frontmatter properties | — | ✅ | shown as a table | +| YAML frontmatter properties | — hidden, not rendered | ✅ | shown as a table | | Raw HTML | — by design | ✅ | sanitized subset | diff --git a/src/lib/block-ids.ts b/src/lib/block-ids.ts new file mode 100644 index 0000000..6a9bca2 --- /dev/null +++ b/src/lib/block-ids.ts @@ -0,0 +1,16 @@ +// Obsidian's charset for author-typed block ids: letters, digits and dashes. +const BLOCK_ID = "[A-Za-z0-9][A-Za-z0-9-]*"; + +// "some text ^my-id" — the marker closing a block. +export const TRAILING_BLOCK_ID_RE = new RegExp(`\\s\\^(${BLOCK_ID})$`); +// "^my-id" alone on the line after a block. +export const STANDALONE_BLOCK_ID_RE = new RegExp(`^\\^(${BLOCK_ID})$`); + +const BLOCK_ID_RE = new RegExp(`^${BLOCK_ID}$`); + +// The anchor keeps the caret: toSlug eats it, so no heading anchor can ever collide +// with a block id. Browsers percent-decode a fragment before matching ids, so `#^id` +// still finds `id="^id"`. +export function blockAnchor(id: string): string | null { + return BLOCK_ID_RE.test(id) ? `^${id}` : null; +} diff --git a/src/lib/frontmatter.ts b/src/lib/frontmatter.ts new file mode 100644 index 0000000..3f9026b --- /dev/null +++ b/src/lib/frontmatter.ts @@ -0,0 +1,8 @@ +// A leading YAML block, as Obsidian writes it. Only a document-opening `---` counts; +// anywhere else `---` stays a horizontal rule. +const FRONTMATTER_RE = /^---[ \t]*\r?\n(?:[\s\S]*?\r?\n)?---[ \t]*(?:\r?\n|$)/; + +export function stripFrontmatter(source: string): string { + if (!source.startsWith("---")) return source; + return source.replace(FRONTMATTER_RE, ""); +} diff --git a/src/lib/headings.ts b/src/lib/headings.ts index 069307a..7b82210 100644 --- a/src/lib/headings.ts +++ b/src/lib/headings.ts @@ -1,3 +1,4 @@ +import { stripFrontmatter } from "./frontmatter.ts"; import { toSlug } from "./slug.ts"; interface Heading { @@ -18,12 +19,13 @@ export function makeAnchor(text: string, seen: Map): string { const HEADING_RE = /^(#{1,6})\s+(.+?)\s*#*\s*$/; const FENCE_RE = /^(`{3,}|~{3,})/; -// Extracts ATX headings, skipping fenced code blocks. +// Extracts ATX headings, skipping fenced code blocks and frontmatter (whose YAML +// comments are "#" lines the renderer never turns into headings). export function extractHeadings(md: string): Heading[] { const out: Heading[] = []; const seen = new Map(); let fence: string | null = null; - for (const line of md.split("\n")) { + for (const line of stripFrontmatter(md).split("\n")) { const fenceMatch = line.match(FENCE_RE); if (fenceMatch) { const marker = fenceMatch[1]?.[0]; diff --git a/src/lib/import-export/markdown-transform.ts b/src/lib/import-export/markdown-transform.ts index 2354240..cc2d572 100644 --- a/src/lib/import-export/markdown-transform.ts +++ b/src/lib/import-export/markdown-transform.ts @@ -42,18 +42,16 @@ export function rewriteForImport( }, ); - // 3. Rewrite wikilinks: [[Note Title]] and [[Note Title|display]] + // 3. Rewrite wikilinks: [[Note Title]], [[Note Title#Section]], [[Note Title|display]] result = result.replace(/\[\[([^\]]+)\]\]/g, (_match, inner: string) => { - const pipeIdx = inner.indexOf("|"); - const title = pipeIdx >= 0 ? inner.slice(0, pipeIdx) : inner; - const display = pipeIdx >= 0 ? inner.slice(pipeIdx + 1) : null; - if (title.includes("/")) { + const { target, section, display } = splitWikilink(inner); + if (target.includes("/")) { return _match; } - const slug = findSlug(slugMap, title.trim()); + const slug = findSlug(slugMap, target.trim()); if (slug) { - return display ? `[[${slug}|${display}]]` : `[[${slug}]]`; + return `[[${slug}${section}${display ? `|${display}` : ""}]]`; } return _match; // leave as-is if no matching note }); @@ -82,18 +80,16 @@ export function rewriteForExport( }, ); - // 2. Rewrite wikilinks: [[slug]] and [[slug|display]] + // 2. Rewrite wikilinks: [[slug]], [[slug#section]], [[slug|display]] result = result.replace(/\[\[([^\]]+)\]\]/g, (_match, inner: string) => { - const pipeIdx = inner.indexOf("|"); - const slug = pipeIdx >= 0 ? inner.slice(0, pipeIdx) : inner; - const display = pipeIdx >= 0 ? inner.slice(pipeIdx + 1) : null; - if (slug.includes("/")) { + const { target, section, display } = splitWikilink(inner); + if (target.includes("/")) { return _match; } - const title = slugToTitle.get(slug.trim()); + const title = slugToTitle.get(target.trim()); if (title) { - return display ? `[[${title}|${display}]]` : `[[${title}]]`; + return `[[${title}${section}${display ? `|${display}` : ""}]]`; } return _match; }); @@ -127,6 +123,24 @@ export function extractLocalImageRefs(content: string): string[] { return [...refs]; } +// Splits [[target#section|display]]; the "#section" keeps its hash so it can be +// concatenated back verbatim — a heading or a ^block-id, neither of which we rewrite. +function splitWikilink(inner: string): { + target: string; + section: string; + display: string | null; +} { + const pipeIdx = inner.indexOf("|"); + const beforePipe = pipeIdx >= 0 ? inner.slice(0, pipeIdx) : inner; + const display = pipeIdx >= 0 ? inner.slice(pipeIdx + 1) : null; + const hashIdx = beforePipe.indexOf("#"); + return { + target: hashIdx >= 0 ? beforePipe.slice(0, hashIdx) : beforePipe, + section: hashIdx >= 0 ? beforePipe.slice(hashIdx) : "", + display, + }; +} + function isAbsoluteUrl(s: string): boolean { return ( s.startsWith("http://") || s.startsWith("https://") || s.startsWith("//") diff --git a/src/lib/markdown/base-plugins.ts b/src/lib/markdown/base-plugins.ts index ad265de..b276e33 100644 --- a/src/lib/markdown/base-plugins.ts +++ b/src/lib/markdown/base-plugins.ts @@ -2,8 +2,10 @@ import type MarkdownIt from "markdown-it"; import footnotePlugin from "markdown-it-footnote"; import markPlugin from "markdown-it-mark"; import taskListPlugin from "markdown-it-task-lists"; +import { blockIdPlugin } from "./block-id-plugin.ts"; import { calloutPlugin } from "./callout-plugin.ts"; import { commentPlugin } from "./comment-plugin.ts"; +import { frontmatterPlugin } from "./frontmatter-plugin.ts"; import { headingAnchorPlugin } from "./heading-anchor-plugin.ts"; import { imagePlugin } from "./image-plugin.ts"; import { katexPlugin } from "./katex-plugin.ts"; @@ -13,8 +15,10 @@ import { youtubePlugin } from "./youtube-plugin.ts"; // shiki/viz are deliberately server-only (heavy WASM grammars / separate D3 bundle), so they live in markdown.ts, not here. export function useBaseMarkdownPlugins(md: MarkdownIt): void { + md.use(frontmatterPlugin); md.use(commentPlugin); md.use(wikilinkPlugin); + md.use(blockIdPlugin); md.use(headingAnchorPlugin); md.use(footnotePlugin); md.use(markPlugin); diff --git a/src/lib/markdown/block-id-plugin.ts b/src/lib/markdown/block-id-plugin.ts new file mode 100644 index 0000000..6b8d856 --- /dev/null +++ b/src/lib/markdown/block-id-plugin.ts @@ -0,0 +1,67 @@ +import type MarkdownIt from "markdown-it"; +import type Token from "markdown-it/lib/token.mjs"; +import { + blockAnchor, + STANDALONE_BLOCK_ID_RE, + TRAILING_BLOCK_ID_RE, +} from "../block-ids.ts"; + +// Tight list items hide their paragraph, so an id set there would never reach the HTML. +function anchorToken(tokens: Token[], inlineIndex: number): Token | null { + let i = inlineIndex - 1; + while (tokens[i]?.hidden && tokens[i - 1]?.nesting === 1) i--; + return tokens[i] ?? null; +} + +function isBlockStart(token: Token): boolean { + return token.nesting !== -1 && token.type !== "inline"; +} + +// Obsidian block ids: "text ^id" ending a block, or "^id" alone on the line after one. +// Headings are excluded — Obsidian addresses those with "#Heading", and their id is +// already the heading anchor. +export function blockIdPlugin(mdi: MarkdownIt): void { + mdi.core.ruler.after("block", "block_id", (state) => { + const tokens = state.tokens; + const standalone: number[] = []; + let previousBlock: Token | null = null; + let currentBlock: Token | null = null; + + for (let i = 0; i < tokens.length; i++) { + const token = tokens[i]; + if (!token) continue; + if (token.level === 0 && isBlockStart(token)) { + previousBlock = currentBlock; + currentBlock = token; + } + if (token.type !== "inline") continue; + + const content = token.content.trimEnd(); + + const alone = STANDALONE_BLOCK_ID_RE.exec(content); + if (alone?.[1] && token.level === 1 && previousBlock) { + const anchor = blockAnchor(alone[1]); + if (anchor) { + previousBlock.attrSet("id", anchor); + standalone.push(i - 1); + currentBlock = previousBlock; + continue; + } + } + + const trailing = TRAILING_BLOCK_ID_RE.exec(content); + if (!trailing?.[1]) continue; + const target = anchorToken(tokens, i); + if (!target || target.type === "heading_open") continue; + const anchor = blockAnchor(trailing[1]); + if (!anchor) continue; + target.attrSet("id", anchor); + token.content = content.slice(0, trailing.index); + } + + // Back to front: the marker paragraph itself leaves no output. + for (let i = standalone.length - 1; i >= 0; i--) { + tokens.splice(standalone[i] as number, 3); + } + }); +} diff --git a/src/lib/markdown/frontmatter-plugin.ts b/src/lib/markdown/frontmatter-plugin.ts new file mode 100644 index 0000000..be075a3 --- /dev/null +++ b/src/lib/markdown/frontmatter-plugin.ts @@ -0,0 +1,11 @@ +import type MarkdownIt from "markdown-it"; +import { stripFrontmatter } from "../frontmatter.ts"; + +// Without this a pasted Obsidian note renders its properties as
+ a setext

. +// Stripping the source before tokenization keeps one definition of "what frontmatter is", +// shared with extractHeadings. +export function frontmatterPlugin(mdi: MarkdownIt): void { + mdi.core.ruler.after("normalize", "frontmatter", (state) => { + state.src = stripFrontmatter(state.src); + }); +} diff --git a/src/lib/markdown/wikilink-plugin.ts b/src/lib/markdown/wikilink-plugin.ts index 371853b..8ec195c 100644 --- a/src/lib/markdown/wikilink-plugin.ts +++ b/src/lib/markdown/wikilink-plugin.ts @@ -1,5 +1,6 @@ import type MarkdownIt from "markdown-it"; import type StateInline from "markdown-it/lib/rules_inline/state_inline.mjs"; +import { blockAnchor } from "../block-ids.ts"; import { escapeHtml } from "../html.ts"; import { toSlug } from "../slug.ts"; import { noteUrl, wikiUrl } from "../urls.ts"; @@ -95,6 +96,16 @@ export function parseWikilinkTarget( }; } +// [[note#^id]] points at a block id, which toSlug would eat the caret of — leaving the +// link aimed at an anchor no note emits. +function sectionAnchor(section: string): string { + if (section.startsWith("^")) { + const anchor = blockAnchor(section.slice(1)); + if (anchor) return anchor; + } + return toSlug(section); +} + export function wikilinkPlugin(mdi: MarkdownIt): void { mdi.inline.ruler.push("wikilink", (state: StateInline, silent: boolean) => { const src = state.src; @@ -116,7 +127,7 @@ export function wikilinkPlugin(mdi: MarkdownIt): void { // Must match what note-validation mints, or [[日本語のノート]] resolves to // a slug no note has. - const anchor = target.section ? `#${toSlug(target.section)}` : ""; + const anchor = target.section ? `#${sectionAnchor(target.section)}` : ""; const noteSlug = target.noteSlug ? toSlug(target.noteSlug) : ""; const wikiSlug = target.wikiSlug ? toSlug(target.wikiSlug) : ""; let href: string; diff --git a/tests/lib/headings.test.ts b/tests/lib/headings.test.ts index 0091871..7eb86e4 100644 --- a/tests/lib/headings.test.ts +++ b/tests/lib/headings.test.ts @@ -54,6 +54,11 @@ describe("extractHeadings", () => { expect(extractHeadings(md).map((h) => h.text)).toEqual(["Real"]); }); + test("ignores YAML comments in frontmatter", () => { + const md = "---\n# not a heading\ntitle: x\n---\n\n# Real"; + expect(extractHeadings(md).map((h) => h.text)).toEqual(["Real"]); + }); + test("disambiguates duplicate anchors", () => { const md = "# Intro\n\n## Intro\n\n### Intro"; expect(extractHeadings(md).map((h) => h.anchor)).toEqual([ diff --git a/tests/lib/import-export/markdown-transform.test.ts b/tests/lib/import-export/markdown-transform.test.ts index 2b33494..1494660 100644 --- a/tests/lib/import-export/markdown-transform.test.ts +++ b/tests/lib/import-export/markdown-transform.test.ts @@ -28,6 +28,17 @@ describe("rewriteForImport", () => { expect(result).toBe("Check [[my-notes|my personal notes]] here."); }); + test("keeps heading and block-id sections attached to the slug", () => { + const result = rewriteForImport( + "See [[Getting Started#Install]] and [[My Notes#^ref-1|notes]].", + slugMap, + imageMap, + ); + expect(result).toBe( + "See [[getting-started#Install]] and [[my-notes#^ref-1|notes]].", + ); + }); + test("leaves unmatched wikilinks as-is", () => { const input = "See [[Unknown Page]] for info."; const result = rewriteForImport(input, slugMap, imageMap); @@ -101,6 +112,16 @@ describe("rewriteForExport", () => { expect(content).toBe("Check [[My Notes|notes]] here."); }); + test("keeps heading and block-id sections attached to the title", () => { + const { content } = rewriteForExport( + "See [[getting-started#Install]] and [[my-notes#^ref-1|notes]].", + slugToTitle, + ); + expect(content).toBe( + "See [[Getting Started#Install]] and [[My Notes#^ref-1|notes]].", + ); + }); + test("leaves unmatched wikilinks as-is", () => { const input = "See [[unknown-slug]] for info."; const { content } = rewriteForExport(input, slugToTitle); diff --git a/tests/lib/markdown.test.ts b/tests/lib/markdown.test.ts index d59fa66..f434518 100644 --- a/tests/lib/markdown.test.ts +++ b/tests/lib/markdown.test.ts @@ -500,6 +500,88 @@ describe("heading anchors", () => { }); }); +describe("frontmatter", () => { + test("strips a leading YAML block instead of rendering it as a heading", () => { + const { html } = renderMarkdown( + "---\ntitle: Hello\ntags: [a]\n---\n\nBody", + ); + expect(html).not.toContain(" { + const { html } = renderMarkdown("---\n---\nBody"); + expect(html).not.toContain(" { + const { html } = renderMarkdown("---\ntitle: Hello\n\nBody"); + expect(html).toContain("title: Hello"); + }); + + test("leaves a horizontal rule further down the note alone", () => { + const { html } = renderMarkdown("text\n\n---\n\nmore"); + expect(html).toContain(" { + test("moves a trailing ^id onto the paragraph and out of the text", () => { + const { html } = renderMarkdown("A paragraph. ^my-id"); + expect(html).toContain('

'); + expect(html).not.toContain("^my-id<"); + }); + + test("attaches a standalone ^id to the block above it", () => { + const { html } = renderMarkdown("| a |\n| - |\n| 1 |\n\n^tbl"); + expect(html).toContain(''); + expect(html).not.toContain("

^tbl

"); + }); + + test("attaches a list item id to the item, not its hidden paragraph", () => { + const { html } = renderMarkdown("- first ^one\n- second"); + expect(html).toContain('
  • '); + }); + + test("leaves headings to their heading anchor", () => { + const { html } = renderMarkdown("# Title ^h1"); + expect(html).toContain('

    '); + }); + + test("ignores carets that are not block markers", () => { + const { html } = renderMarkdown("2^10 and `a ^id` and x ^bad_id"); + expect(html).not.toContain("id="); + expect(html).toContain("^bad_id"); + }); + + test("ignores a ^id inside a fenced code block", () => { + const { html } = renderMarkdown("```\nfoo ^id\n```"); + expect(html).toContain("^id"); + expect(html).not.toContain('id="^id"'); + }); + + test("links [[note#^id]] to the block anchor, caret intact", () => { + const { html } = renderMarkdown("[[note#^my-id]]", "wiki", "alice.test"); + expect(html).toContain('href="/@alice.test/wiki/note#^my-id"'); + }); + + test("links [[#^id]] within the same note", () => { + const { html } = renderMarkdown("[[#^local]]", "wiki", "alice.test"); + expect(html).toContain('href="#^local"'); + }); + + test("still slugifies heading sections", () => { + const { html } = renderMarkdown( + "[[note#Some Heading]]", + "wiki", + "alice.test", + ); + expect(html).toContain('href="/@alice.test/wiki/note#some-heading"'); + }); +}); + describe("extractWikilinks", () => { test("returns empty array for content with no wikilinks", () => { expect(extractWikilinks("no links here")).toEqual([]);