diff --git a/src/lib/import-export/markdown-transform.ts b/src/lib/import-export/markdown-transform.ts new file mode 100644 index 0000000..33c1740 --- /dev/null +++ b/src/lib/import-export/markdown-transform.ts @@ -0,0 +1,184 @@ +import { basename } from "node:path"; + +interface ExportResult { + content: string; + blobRefs: { did: string; cid: string }[]; +} + +/** + * Rewrite Obsidian-style markdown to Lichen format. + * - [[Note Title]] -> [[slug]] + * - [[Note Title|display]] -> [[slug|display]] + * - ![[image.png]] -> ![image](blobUrl) + * - ![alt](relative/image.png) -> ![alt](blobUrl) (basename match) + * - Absolute URLs and /blob/ refs left untouched + */ +export function rewriteForImport( + content: string, + slugMap: Map, + imageMap: Map, +): string { + // 1. Rewrite Obsidian image embeds: ![[image.png]] + let result = content.replace( + /!\[\[([^\]]+)\]\]/g, + (_match, filename: string) => { + const name = basename(filename.trim()); + const url = findImageUrl(imageMap, name); + if (url) { + const alt = name.replace(/\.[^.]+$/, ""); + return `![${alt}](${url})`; + } + return _match; // leave as-is if image not found + }, + ); + + // 2. Rewrite standard markdown images with relative paths + result = result.replace( + /!\[([^\]]*)\]\(([^)]+)\)/g, + (_match, alt: string, href: string) => { + if (isAbsoluteUrl(href) || href.startsWith("/blob/")) { + return _match; + } + const name = basename(href.trim()); + const url = findImageUrl(imageMap, name); + if (url) { + return `![${alt}](${url})`; + } + return _match; + }, + ); + + // 3. Rewrite wikilinks: [[Note Title]] and [[Note Title|display]] + result = result.replace(/\[\[([^\]]+)\]\]/g, (_match, inner: string) => { + const pipeIdx = inner.indexOf("|"); + const title = pipeIdx >= 0 ? inner.slice(0, pipeIdx) : inner; + const display = pipeIdx >= 0 ? inner.slice(pipeIdx + 1) : null; + + const slug = findSlug(slugMap, title.trim()); + if (slug) { + return display ? `[[${slug}|${display}]]` : `[[${slug}]]`; + } + return _match; // leave as-is if no matching note + }); + + return result; +} + +/** + * Rewrite Lichen markdown to Obsidian-compatible format. + * - [[slug]] -> [[Note Title]] + * - [[slug|display]] -> [[Note Title|display]] + * - ![alt](/blob/{did}/{cid}) -> ![alt](attachments/{cid}.ext), collects blob refs + */ +export function rewriteForExport( + content: string, + slugToTitle: Map, +): ExportResult { + const blobRefs: { did: string; cid: string }[] = []; + const seenCids = new Set(); + + // 1. Rewrite blob image refs to attachments/ + let result = content.replace( + /!\[([^\]]*)\]\(\/blob\/(did:[^/]+)\/([^)\s]+)\)/g, + (_match, alt: string, did: string, cid: string) => { + if (!seenCids.has(cid)) { + seenCids.add(cid); + blobRefs.push({ did, cid }); + } + // Use cid as filename — extension will be added when building the zip + return `![${alt}](attachments/${cid})`; + }, + ); + + // 2. Rewrite wikilinks: [[slug]] and [[slug|display]] + result = result.replace(/\[\[([^\]]+)\]\]/g, (_match, inner: string) => { + const pipeIdx = inner.indexOf("|"); + const slug = pipeIdx >= 0 ? inner.slice(0, pipeIdx) : inner; + const display = pipeIdx >= 0 ? inner.slice(pipeIdx + 1) : null; + + const title = slugToTitle.get(slug.trim()); + if (title) { + return display ? `[[${title}|${display}]]` : `[[${title}]]`; + } + return _match; + }); + + return { content: result, blobRefs }; +} + +/** + * Extract local image references from markdown content. + * Finds ![[filename]] and ![...](relative-path) where path is not a URL or /blob/ ref. + * Returns basenames only (deduplicated). + */ +export function extractLocalImageRefs(content: string): string[] { + const refs = new Set(); + + // Obsidian embeds: ![[filename]] + for (const match of content.matchAll(/!\[\[([^\]]+)\]\]/g)) { + const name = basename((match[1] as string).trim()); + if (isImageFilename(name)) { + refs.add(name.toLowerCase()); + } + } + + // Standard markdown images: ![alt](path) + for (const match of content.matchAll(/!\[[^\]]*\]\(([^)]+)\)/g)) { + const href = (match[1] as string).trim(); + if (!isAbsoluteUrl(href) && !href.startsWith("/blob/")) { + const name = basename(href); + if (isImageFilename(name)) { + refs.add(name.toLowerCase()); + } + } + } + + return [...refs]; +} + +function isAbsoluteUrl(s: string): boolean { + return ( + s.startsWith("http://") || s.startsWith("https://") || s.startsWith("//") + ); +} + +const IMAGE_EXTENSIONS = new Set([".jpg", ".jpeg", ".png", ".gif", ".webp"]); + +function isImageFilename(name: string): boolean { + const ext = name.slice(name.lastIndexOf(".")).toLowerCase(); + return IMAGE_EXTENSIONS.has(ext); +} + +/** + * Case-insensitive lookup in imageMap by basename. + */ +function findImageUrl( + imageMap: Map, + filename: string, +): string | undefined { + // Try exact match first + const exact = imageMap.get(filename); + if (exact) return exact; + // Case-insensitive fallback + const lower = filename.toLowerCase(); + for (const [key, value] of imageMap) { + if (key.toLowerCase() === lower) return value; + } + return undefined; +} + +/** + * Look up slug by note title. Tries exact match, then case-insensitive. + */ +function findSlug( + slugMap: Map, + title: string, +): string | undefined { + const exact = slugMap.get(title); + if (exact) return exact; + const lower = title.toLowerCase(); + for (const [key, value] of slugMap) { + if (key.toLowerCase() === lower) return value; + } + return undefined; +} diff --git a/src/lib/import-export/types.ts b/src/lib/import-export/types.ts new file mode 100644 index 0000000..f7f30ae --- /dev/null +++ b/src/lib/import-export/types.ts @@ -0,0 +1,18 @@ +export interface ImportedNote { + filename: string; + title: string; + slug: string; + content: string; +} + +export interface ImportedImage { + filename: string; + data: Uint8Array; + mimeType: string; +} + +export interface ImportResult { + notes: ImportedNote[]; + images: ImportedImage[]; + warnings: string[]; +} diff --git a/src/lib/import-export/zip-parse.ts b/src/lib/import-export/zip-parse.ts new file mode 100644 index 0000000..123b26a --- /dev/null +++ b/src/lib/import-export/zip-parse.ts @@ -0,0 +1,158 @@ +import { basename } from "node:path"; +import { unzipSync } from "fflate"; +import { ImportError } from "../errors.ts"; +import { isValidSlug, slugify } from "../slug.ts"; +import { extractLocalImageRefs } from "./markdown-transform.ts"; +import type { ImportedImage, ImportedNote, ImportResult } from "./types.ts"; + +const MAX_UNCOMPRESSED_BYTES = 50 * 1024 * 1024; // 50MB +const MAX_ZIP_ENTRIES = 500; +const MAX_MD_FILES = 100; + +const IMAGE_EXTENSIONS = new Set([".jpg", ".jpeg", ".png", ".gif", ".webp"]); +const MIME_BY_EXT: Record = { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".png": "image/png", + ".gif": "image/gif", + ".webp": "image/webp", +}; + +/** + * Parse an import zip file, extracting markdown notes and referenced images. + * Throws ImportError on validation failures. + */ +export function parseImportZip(buffer: ArrayBuffer): ImportResult { + let entries: Record; + try { + entries = unzipSync(new Uint8Array(buffer)); + } catch { + throw new ImportError("Invalid or corrupt zip file."); + } + + const paths = Object.keys(entries); + if (paths.length === 0) { + throw new ImportError("Zip file is empty."); + } + if (paths.length > MAX_ZIP_ENTRIES) { + throw new ImportError( + `Zip contains too many entries (${paths.length}, max ${MAX_ZIP_ENTRIES}).`, + ); + } + + // Check total uncompressed size + let totalSize = 0; + for (const data of Object.values(entries)) { + totalSize += data.length; + if (totalSize > MAX_UNCOMPRESSED_BYTES) { + throw new ImportError("Zip content exceeds 50MB uncompressed limit."); + } + } + + const warnings: string[] = []; + const mdFiles: { name: string; content: string }[] = []; + const imageFiles: Map = new Map(); // lowercase basename -> data + + for (const [path, data] of Object.entries(entries)) { + // Skip directories (entries ending with /) + if (path.endsWith("/")) continue; + + const name = basename(path); + + // Skip hidden files/directories + if ( + name.startsWith(".") || + path.split("/").some((p) => p.startsWith(".")) + ) { + continue; + } + + const ext = name.slice(name.lastIndexOf(".")).toLowerCase(); + + if (ext === ".md" || ext === ".markdown") { + mdFiles.push({ name, content: new TextDecoder().decode(data) }); + } else if (IMAGE_EXTENSIONS.has(ext)) { + imageFiles.set(name.toLowerCase(), data); + } + // All other files silently ignored + } + + if (mdFiles.length === 0) { + throw new ImportError("No markdown files found in zip."); + } + if (mdFiles.length > MAX_MD_FILES) { + throw new ImportError( + `Too many markdown files (${mdFiles.length}, max ${MAX_MD_FILES}).`, + ); + } + + // Collect all image references from all markdown files + const referencedImages = new Set(); + for (const { content } of mdFiles) { + for (const ref of extractLocalImageRefs(content)) { + referencedImages.add(ref); // already lowercased by extractLocalImageRefs + } + } + + // Filter images to only referenced ones, preserving original filename casing + const images: ImportedImage[] = []; + for (const [lowerName, data] of imageFiles) { + if (referencedImages.has(lowerName)) { + const ext = lowerName.slice(lowerName.lastIndexOf(".")); + images.push({ + filename: lowerName, + data, + mimeType: MIME_BY_EXT[ext] ?? "application/octet-stream", + }); + } + } + + // Build notes with slug deduplication + const notes: ImportedNote[] = []; + const usedSlugs = new Set(); + + // Sort so that home.md/Home.md comes first + mdFiles.sort((a, b) => { + const aIsHome = isHomeName(a.name); + const bIsHome = isHomeName(b.name); + if (aIsHome && !bIsHome) return -1; + if (!aIsHome && bIsHome) return 1; + return a.name.localeCompare(b.name); + }); + + for (const { name, content } of mdFiles) { + const title = name.replace(/\.(md|markdown)$/i, ""); + let slug: string; + + if (isHomeName(name) && !usedSlugs.has("home")) { + slug = "home"; + } else { + slug = slugify(title); + } + + if (!slug || !isValidSlug(slug)) { + warnings.push(`Skipped "${name}": could not generate a valid slug.`); + continue; + } + + if (usedSlugs.has(slug)) { + const original = slug; + let counter = 2; + while (usedSlugs.has(slug)) { + slug = `${original}-${counter}`; + counter++; + } + warnings.push(`Renamed "${name}" slug from "${original}" to "${slug}".`); + } + + usedSlugs.add(slug); + notes.push({ filename: name, title, slug, content }); + } + + return { notes, images, warnings }; +} + +function isHomeName(filename: string): boolean { + const name = filename.replace(/\.(md|markdown)$/i, "").toLowerCase(); + return name === "home"; +} diff --git a/tests/lib/import-export/markdown-transform.test.ts b/tests/lib/import-export/markdown-transform.test.ts new file mode 100644 index 0000000..dcbb9ec --- /dev/null +++ b/tests/lib/import-export/markdown-transform.test.ts @@ -0,0 +1,161 @@ +import { describe, expect, test } from "bun:test"; +import { + extractLocalImageRefs, + rewriteForExport, + rewriteForImport, +} from "../../../src/lib/import-export/markdown-transform.ts"; + +describe("rewriteForImport", () => { + const slugMap = new Map([ + ["Getting Started", "getting-started"], + ["My Notes", "my-notes"], + ["Home", "home"], + ]); + const imageMap = new Map([ + ["photo.png", "/blob/did:plc:abc/bafk1"], + ["diagram.jpg", "/blob/did:plc:abc/bafk2"], + ]); + + test("rewrites Obsidian wikilinks to slugs", () => { + const input = "See [[Getting Started]] for info."; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("See [[getting-started]] for info."); + }); + + test("rewrites wikilinks with display text", () => { + const input = "Check [[My Notes|my personal notes]] here."; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("Check [[my-notes|my personal notes]] here."); + }); + + test("leaves unmatched wikilinks as-is", () => { + const input = "See [[Unknown Page]] for info."; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("See [[Unknown Page]] for info."); + }); + + test("rewrites Obsidian image embeds", () => { + const input = "Here is ![[photo.png]] in text."; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("Here is ![photo](/blob/did:plc:abc/bafk1) in text."); + }); + + test("rewrites relative image paths", () => { + const input = "![my diagram](assets/diagram.jpg)"; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("![my diagram](/blob/did:plc:abc/bafk2)"); + }); + + test("leaves absolute URLs untouched", () => { + const input = "![ext](https://example.com/img.png)"; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("![ext](https://example.com/img.png)"); + }); + + test("leaves /blob/ refs untouched", () => { + const input = "![img](/blob/did:plc:xyz/bafkabc)"; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("![img](/blob/did:plc:xyz/bafkabc)"); + }); + + test("leaves unmatched image embeds as-is", () => { + const input = "![[missing.png]]"; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("![[missing.png]]"); + }); + + test("handles case-insensitive image match", () => { + const input = "![[Photo.PNG]]"; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("![Photo](/blob/did:plc:abc/bafk1)"); + }); + + test("handles case-insensitive wikilink match", () => { + const input = "See [[home]] page."; + const result = rewriteForImport(input, slugMap, imageMap); + expect(result).toBe("See [[home]] page."); + }); +}); + +describe("rewriteForExport", () => { + const slugToTitle = new Map([ + ["getting-started", "Getting Started"], + ["my-notes", "My Notes"], + ]); + + test("rewrites slug wikilinks to titles", () => { + const input = "See [[getting-started]] for info."; + const { content } = rewriteForExport(input, slugToTitle); + expect(content).toBe("See [[Getting Started]] for info."); + }); + + test("rewrites wikilinks with display text", () => { + const input = "Check [[my-notes|notes]] here."; + const { content } = rewriteForExport(input, slugToTitle); + expect(content).toBe("Check [[My Notes|notes]] here."); + }); + + test("leaves unmatched wikilinks as-is", () => { + const input = "See [[unknown-slug]] for info."; + const { content } = rewriteForExport(input, slugToTitle); + expect(content).toBe("See [[unknown-slug]] for info."); + }); + + test("rewrites blob refs to attachments and collects refs", () => { + const input = "![photo](/blob/did:plc:abc/bafk123)"; + const { content, blobRefs } = rewriteForExport(input, slugToTitle); + expect(content).toBe("![photo](attachments/bafk123)"); + expect(blobRefs).toEqual([{ did: "did:plc:abc", cid: "bafk123" }]); + }); + + test("deduplicates blob refs", () => { + const input = + "![a](/blob/did:plc:abc/bafk1)\n![b](/blob/did:plc:abc/bafk1)"; + const { blobRefs } = rewriteForExport(input, slugToTitle); + expect(blobRefs).toHaveLength(1); + }); + + test("collects multiple distinct blob refs", () => { + const input = + "![a](/blob/did:plc:abc/bafk1)\n![b](/blob/did:plc:xyz/bafk2)"; + const { blobRefs } = rewriteForExport(input, slugToTitle); + expect(blobRefs).toHaveLength(2); + }); +}); + +describe("extractLocalImageRefs", () => { + test("finds Obsidian embed refs", () => { + const refs = extractLocalImageRefs("text ![[photo.png]] more"); + expect(refs).toEqual(["photo.png"]); + }); + + test("finds standard markdown image refs", () => { + const refs = extractLocalImageRefs("![alt](images/diagram.jpg)"); + expect(refs).toEqual(["diagram.jpg"]); + }); + + test("ignores absolute URLs", () => { + const refs = extractLocalImageRefs("![alt](https://example.com/photo.png)"); + expect(refs).toEqual([]); + }); + + test("ignores /blob/ refs", () => { + const refs = extractLocalImageRefs("![alt](/blob/did:plc:abc/bafk123)"); + expect(refs).toEqual([]); + }); + + test("ignores non-image files", () => { + const refs = extractLocalImageRefs("![[document.pdf]]"); + expect(refs).toEqual([]); + }); + + test("deduplicates refs (case-insensitive)", () => { + const refs = extractLocalImageRefs("![[Photo.PNG]] ![x](photo.png)"); + expect(refs).toHaveLength(1); + }); + + test("extracts basename from nested paths", () => { + const refs = extractLocalImageRefs("![x](deep/nested/path/image.webp)"); + expect(refs).toEqual(["image.webp"]); + }); +}); diff --git a/tests/lib/import-export/zip-parse.test.ts b/tests/lib/import-export/zip-parse.test.ts new file mode 100644 index 0000000..486a01d --- /dev/null +++ b/tests/lib/import-export/zip-parse.test.ts @@ -0,0 +1,194 @@ +import { describe, expect, test } from "bun:test"; +import { zipSync } from "fflate"; +import { parseImportZip } from "../../../src/lib/import-export/zip-parse.ts"; + +function makeZip(files: Record): ArrayBuffer { + const entries: Record = {}; + for (const [name, content] of Object.entries(files)) { + entries[name] = + typeof content === "string" ? new TextEncoder().encode(content) : content; + } + return zipSync(entries).buffer as ArrayBuffer; +} + +describe("parseImportZip", () => { + test("parses valid zip with md files", () => { + const zip = makeZip({ + "Note One.md": "# Hello", + "Note Two.md": "# World", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(2); + expect(result.notes.map((n) => n.title)).toContain("Note One"); + expect(result.notes.map((n) => n.title)).toContain("Note Two"); + }); + + test("derives title from filename (strips .md)", () => { + const zip = makeZip({ "My Great Note.md": "content" }); + const result = parseImportZip(zip); + expect(result.notes[0]?.title).toBe("My Great Note"); + }); + + test("generates valid slugs from titles", () => { + const zip = makeZip({ "My Great Note.md": "content" }); + const result = parseImportZip(zip); + expect(result.notes[0]?.slug).toBe("my-great-note"); + }); + + test("flattens nested directories", () => { + const zip = makeZip({ + "folder/subfolder/deep.md": "content", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + expect(result.notes[0]?.title).toBe("deep"); + }); + + test("deduplicates slugs with -2, -3 suffixes", () => { + const zip = makeZip({ + "Note.md": "first", + "subdir/Note.md": "second", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(2); + const slugs = result.notes.map((n) => n.slug); + expect(slugs).toContain("note"); + expect(slugs).toContain("note-2"); + expect(result.warnings.length).toBeGreaterThan(0); + }); + + test("home.md gets slug 'home'", () => { + const zip = makeZip({ + "Home.md": "Welcome", + "Other.md": "stuff", + }); + const result = parseImportZip(zip); + const homeNote = result.notes.find((n) => n.slug === "home"); + expect(homeNote).toBeDefined(); + expect(homeNote?.title).toBe("Home"); + }); + + test("home.md comes first in notes array", () => { + const zip = makeZip({ + "Zebra.md": "z", + "home.md": "Welcome", + "Alpha.md": "a", + }); + const result = parseImportZip(zip); + expect(result.notes[0]?.slug).toBe("home"); + }); + + test("skips hidden files", () => { + const zip = makeZip({ + ".obsidian/config.json": "{}", + ".DS_Store": "junk", + "Real Note.md": "content", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + expect(result.notes[0]?.title).toBe("Real Note"); + }); + + test("skips files in hidden directories", () => { + const zip = makeZip({ + ".obsidian/plugins/note.md": "hidden", + "visible.md": "ok", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + }); + + test("only keeps referenced images", () => { + const zip = makeZip({ + "note.md": "![[photo.png]]", + "photo.png": new Uint8Array([137, 80, 78, 71]), // PNG header + "unused.jpg": new Uint8Array([255, 216, 255]), // JPEG header + }); + const result = parseImportZip(zip); + expect(result.images).toHaveLength(1); + expect(result.images[0]?.filename).toBe("photo.png"); + }); + + test("ignores non-md text files", () => { + const zip = makeZip({ + "note.md": "content", + "readme.txt": "text", + "data.csv": "1,2,3", + "config.json": "{}", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + }); + + test("handles .markdown extension", () => { + const zip = makeZip({ "note.markdown": "content" }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + expect(result.notes[0]?.title).toBe("note"); + }); + + test("throws on empty zip", () => { + const zip = makeZip({}); + expect(() => parseImportZip(zip)).toThrow("empty"); + }); + + test("throws on corrupt data", () => { + const bad = new ArrayBuffer(100); + expect(() => parseImportZip(bad)).toThrow("Invalid"); + }); + + test("throws on no markdown files", () => { + const zip = makeZip({ "image.png": new Uint8Array([1, 2, 3]) }); + expect(() => parseImportZip(zip)).toThrow("No markdown"); + }); + + test("throws on too many markdown files", () => { + const files: Record = {}; + for (let i = 0; i < 101; i++) { + files[`note-${i}.md`] = `content ${i}`; + } + const zip = makeZip(files); + expect(() => parseImportZip(zip)).toThrow("Too many"); + }); + + test("throws on too many total entries", () => { + const files: Record = {}; + for (let i = 0; i < 501; i++) { + files[`file-${i}.txt`] = "x"; + } + // Need at least 1 md file for the check to reach the entry count check + // Actually entry count is checked before md count + const zip = makeZip(files); + expect(() => parseImportZip(zip)).toThrow("too many entries"); + }); + + test("skips files that produce invalid slugs", () => { + const zip = makeZip({ + "!!!.md": "content", + "valid.md": "ok", + }); + const result = parseImportZip(zip); + expect(result.notes).toHaveLength(1); + expect(result.notes[0]?.slug).toBe("valid"); + expect(result.warnings.length).toBeGreaterThan(0); + }); + + test("skips directory entries", () => { + const zip = makeZip({ + "folder/": "", + "folder/note.md": "content", + }); + // fflate may or may not include directory entries, but our code handles it + const result = parseImportZip(zip); + expect(result.notes.length).toBeGreaterThanOrEqual(1); + }); + + test("image refs are case-insensitive", () => { + const zip = makeZip({ + "note.md": "![[Photo.PNG]]", + "photo.png": new Uint8Array([137, 80, 78, 71]), + }); + const result = parseImportZip(zip); + expect(result.images).toHaveLength(1); + }); +});