From 3341b11a99f9f0de1ce7f88eb57c7af0f5ccbec7 Mon Sep 17 00:00:00 2001 From: Ewan Croft Date: Fri, 1 May 2026 22:15:06 +0100 Subject: [PATCH] feat(opal): rewrite parsers based on export format research --- packages/opal/MAPPING.md | 334 +++++++++++++++++++++++++++++++++ packages/opal/src/index.ts | 3 + packages/opal/src/mastodon.ts | 65 ++++++- packages/opal/src/nostr.ts | 142 ++++++++++++-- packages/opal/src/publisher.ts | 16 +- packages/opal/src/threads.ts | 49 ++++- packages/opal/src/twitter.ts | 117 ++++++++++-- packages/opal/src/types.ts | 105 +++++++++-- 8 files changed, 767 insertions(+), 64 deletions(-) create mode 100644 packages/opal/MAPPING.md diff --git a/packages/opal/MAPPING.md b/packages/opal/MAPPING.md new file mode 100644 index 0000000..e91ffce --- /dev/null +++ b/packages/opal/MAPPING.md @@ -0,0 +1,334 @@ +# Opal: Source Format → Bluesky Post Mapping + +Reference for converting microblog exports from Twitter, Mastodon, Threads, and Nostr into `app.bsky.feed.post` records. + +--- + +## Target: `app.bsky.feed.post` + +| Field | Type | Required | Notes | +|-------|------|----------|-------| +| `text` | string | ✓ | Max 300 grapheme clusters, 3000 UTF-8 bytes | +| `createdAt` | string | ✓ | ISO 8601 datetime | +| `facets` | `app.bsky.richtext.facet[]` | | Rich text annotations (links, mentions, tags) | +| `reply` | `{ root: StrongRef, parent: StrongRef }` | | Both root and parent are `com.atproto.repo.strongRef` (`uri` + `cid`) | +| `embed` | union | | `images`, `external`, `record`, `recordWithMedia` | +| `langs` | `string[]` | | BCP-47 language tags | +| `tags` | `string[]` | | Free-form tags | + +### Facet byte offsets + +Facet `index.byteStart` and `index.byteEnd` are **UTF-8 byte positions**, not character indices. When converting from platforms that use character indices (Twitter) or HTML offsets (Mastodon), recalculate to UTF-8 bytes. + +### Strong refs for replies + +Replies require both `root` and `parent` as `{ uri, cid }`. The `cid` is only available after the referenced post is created — this means **posts must be imported in chronological order within a thread**, and the publisher must track the `uri` → `cid` mapping for previously created posts. + +### Image embeds + +`app.bsky.embed.images` supports up to 4 images. Each requires: +- `image` — a blob (uploaded via `com.atproto.repo.uploadBlob`) +- `alt` — alt text (empty string if unknown) +- `aspectRatio` — optional `{ width, height }` + +### Quote posts + +`app.bsky.embed.record` takes a `{ uri, cid }` strong ref to the quoted post. Only works for AT Protocol records — external quotes become `app.bsky.embed.external` cards instead. + +--- + +## Twitter/X Archive + +### Source file + +`tweets.js` — JavaScript assigning to `window.YTD.tweets.part0` (newer) or `window.YTD.tweet.part0` (older). Each entry is wrapped in `{ tweet: { ... } }`. + +### Timestamp format + +``` +Thu Nov 03 13:21:02 +0000 2022 +``` +Parseable directly by `new Date(ts)`. + +### Field mapping + +| Twitter field | Bluesky field | Transform | +|---------------|---------------|-----------| +| `full_text` | `text` | Strip `t.co` URLs (replace with expanded). Truncate at 300 graphemes. | +| `created_at` | `createdAt` | Parse Twitter format → ISO 8601 | +| `entities.urls[]` | `facets[]` (link) | Replace `t.co` short URL in text with `expanded_url`. Facet byte offsets on the expanded URL. | +| `entities.hashtags[]` | `facets[]` (tag) | Convert character `indices` to UTF-8 byte offsets. `tag` = `text` value (without `#`). | +| `entities.user_mentions[]` | `facets[]` (mention) | **Cannot resolve DID from screen name.** Store as link facet to `https://bsky.app/profile/[handle]` if handle exists, otherwise skip. | +| `entities.media[]` | `embed.images[]` | Upload via `uploadBlob`, max 4 per post. `alt` = empty (no alt text in archive). | +| `extended_entities.media[]` | `embed.images[]` | Same as above — prefer `extended_entities` over `entities` (includes all media, not just first). | +| `in_reply_to_status_id_str` | `reply.parent.uri` | Must map to AT URI of already-imported parent post. Requires import ordering. | +| `conversation_id_str` | `reply.root.uri` | First tweet in conversation = root. Must be imported first. | +| `quoted_status_id_str` | `embed.record` | Only if the quoted tweet was also imported. Otherwise, use `embed.external` with the Twitter URL. | +| `retweeted_status` | — | **Skip.** Retweets are someone else's content. | +| `lang` | `langs[]` | Map Twitter lang code to BCP-47 (mostly the same). | + +### Key gotchas + +- **`t.co` URL replacement**: Twitter replaces all URLs with `t.co` shortlinks. The archive stores both `url` (t.co) and `expanded_url` (original). Replace the `t.co` URL in `full_text` with the `expanded_url` before computing facet byte offsets. +- **Media indices overlap**: Media entities have indices in `full_text` pointing to the `t.co` link. After replacing URLs, these indices shift. Process URL replacements first, then recompute positions. +- **Thread reconstruction**: The archive doesn't give you a thread tree. Reconstruct by grouping on `conversation_id_str` and following `in_reply_to_status_id_str` chains. Posts must be imported oldest-first within each thread. +- **Quote tweets of external content**: If `quoted_status` is not in the archive, the quote is of a tweet by someone else. Use `app.bsky.embed.external` with `quoted_status_permalink.expanded` as the URL. +- **Split archive files**: Large archives may split into `tweet-part1.js`, `tweet-part2.js`, etc. + +### Example tweet → Bluesky record + +```json +// Twitter input +{ + "id_str": "123456", + "created_at": "Thu Nov 03 13:21:02 +0000 2022", + "full_text": "Hello world! https://t.co/abc123 #nostr @someone", + "entities": { + "urls": [{ "url": "https://t.co/abc123", "expanded_url": "https://example.com", "indices": [13, 36] }], + "hashtags": [{ "text": "nostr", "indices": [37, 44] }], + "user_mentions": [{ "screen_name": "someone", "indices": [45, 56] }] + } +} + +// Bluesky output +{ + "$type": "app.bsky.feed.post", + "text": "Hello world! https://example.com #nostr @someone", + "createdAt": "2022-11-03T13:21:02.000Z", + "facets": [ + { "index": { "byteStart": 13, "byteEnd": 31 }, "features": [{ "$type": "app.bsky.richtext.facet#link", "uri": "https://example.com" }] }, + { "index": { "byteStart": 32, "byteEnd": 39 }, "features": [{ "$type": "app.bsky.richtext.facet#tag", "tag": "nostr" }] }, + { "index": { "byteStart": 40, "byteEnd": 49 }, "features": [{ "$type": "app.bsky.richtext.facet#link", "uri": "https://bsky.app/profile/someone.bsky.social" }] } + ] +} +``` + +--- + +## Mastodon + +### Source file + +`outbox.json` — ActivityStreams 2.0 `OrderedCollection` or `OrderedCollectionPage`. Posts are in `orderedItems[]`, each a `Create` activity wrapping a `Note` object. + +### Timestamp format + +ISO 8601: `"2024-01-15T10:30:00Z"` — already compatible with Bluesky. + +### Field mapping + +| Mastodon field | Bluesky field | Transform | +|----------------|---------------|-----------| +| `object.content` | `text` | Strip HTML tags. Decode entities. Preserve `
` → `\n`. Truncate at 300 graphemes. | +| `object.published` | `createdAt` | Already ISO 8601. | +| `object.tag[]` (Hashtag) | `facets[]` (tag) | Find `#tag` in stripped text, compute UTF-8 byte offsets. | +| `object.tag[]` (Mention) | `facets[]` (mention) | **Cannot resolve DID from ActivityPub `href`.** Store as link facet to the profile URL. | +| `object.attachment[]` | `embed.images[]` | Upload via `uploadBlob`. Max 4. Use `attachment.name` as alt text if available. | +| `object.inReplyTo` | `reply.parent.uri` | ActivityPub URI — cannot map to AT URI without resolution. Store as reference; skip reply ref if parent not imported. | +| `object.sensitive` + `object.summary` | — | **No CW/sensitive equivalent in Bluesky.** Prepend `[CW: summary] ` to text if `sensitive` is true. | + +### Key gotchas + +- **HTML content**: Mastodon toots are HTML. Must strip tags, decode entities, and convert `
` to newlines before any facet computation. +- **Hashtag positions shift**: After HTML stripping, the character positions of `#tags` change. Must search for tags in the stripped text, not use any HTML-level offsets. +- **Mentions are `@handle@instance`**: After stripping, mentions appear as `@user@instance.tld` or `@user`. The `tag.href` is the ActivityPub actor URI — cannot directly convert to a Bluesky DID. +- **No content warnings**: Bluesky has no CW/sensitive field. Options: prepend CW text, skip it, or add a `⚠️` prefix. +- **Media types**: Mastodon attachments have `type: "Document"` with `mediaType`. Filter for images (`image/*`). Videos have no Bluesky equivalent (would need `app.bsky.embed.video` which is limited). +- **Paginated outbox**: Large outboxes may use `OrderedCollection` with `first` → `next` pagination. The export ZIP should contain the full flattened outbox, but verify. +- **Boosts**: `Announce` activities in the outbox are boosts (shares of others' posts). Skip these — they're not original content. + +### Example toot → Bluesky record + +```json +// Mastodon input (ActivityPub Note) +{ + "type": "Note", + "content": "

Hello from @user! example.com

#nostr #activitypub

", + "published": "2024-01-15T10:30:00Z", + "tag": [ + { "type": "Mention", "href": "https://mastodon.social/@user", "name": "@user@mastodon.social" }, + { "type": "Hashtag", "name": "#nostr", "href": "https://mastodon.social/tags/nostr" }, + { "type": "Hashtag", "name": "#activitypub", "href": "https://mastodon.social/tags/activitypub" } + ], + "attachment": [ + { "type": "Document", "mediaType": "image/png", "url": "https://files.mastodon.social/media/abc.png", "name": "A screenshot" } + ] +} + +// Bluesky output +{ + "$type": "app.bsky.feed.post", + "text": "Hello from @user! example.com\n\n#nostr #activitypub", + "createdAt": "2024-01-15T10:30:00Z", + "facets": [ + { "index": { "byteStart": 11, "byteEnd": 16 }, "features": [{ "$type": "app.bsky.richtext.facet#link", "uri": "https://mastodon.social/@user" }] }, + { "index": { "byteStart": 18, "byteEnd": 29 }, "features": [{ "$type": "app.bsky.richtext.facet#link", "uri": "https://example.com" }] }, + { "index": { "byteStart": 31, "byteEnd": 37 }, "features": [{ "$type": "app.bsky.richtext.facet#tag", "tag": "nostr" }] }, + { "index": { "byteStart": 38, "byteEnd": 50 }, "features": [{ "$type": "app.bsky.richtext.facet#tag", "tag": "activitypub" }] } + ], + "embed": { + "$type": "app.bsky.embed.images", + "images": [{ "image": { "$type": "blob", "ref": "...", "mimeType": "image/png", "size": 12345 }, "alt": "A screenshot" }] + } +} +``` + +--- + +## Threads + +### Source file + +`posts_1.json` — Instagram-archive-shaped JSON. No official schema published by Meta. + +### Timestamp format + +Unix epoch seconds (integer) in the archive. Convert to ISO 8601. + +### Field mapping + +| Threads field | Bluesky field | Transform | +|---------------|---------------|-----------| +| `title` | `text` | Post caption/text. Truncate at 300 graphemes. | +| `creation_timestamp` | `createdAt` | Unix seconds → ISO 8601 | +| `media[].uri` | `embed.images[]` | Local path in archive. Upload via `uploadBlob`. Max 4. | +| `media[].title` | `embed.images[].alt` | Use media title as alt text if available. | + +### Key gotchas + +- **Undocumented format**: Meta doesn't publish a schema. The structure is inferred from community reverse-engineering. Handle both `title` at root level and `media[].title` as fallback. +- **No reply/quote data**: The export doesn't reliably include reply chains or quote references. These relationships must be inferred or skipped. +- **No hashtags/mentions**: The export doesn't include structured entity data. Could regex-detect `#hashtags` and `@mentions` in text, but no reliable offset data. +- **Timestamp is Unix seconds**: `creation_timestamp` is an integer, not a string. Convert with `new Date(ts * 1000).toISOString()`. +- **Media may be missing**: The archive references local file paths (`media/posts/abcd.jpg`) but the files may not exist in the export. +- **Instagram-shaped**: Threads exports reuse the Instagram/Accounts Center export structure. The `posts_1.json` file may be under `your_instagram_activity/threads/` or `content/` depending on the export version. + +### Example Threads post → Bluesky record + +```json +// Threads input +{ + "title": "Just joined Threads!", + "creation_timestamp": 1700000000, + "media": [ + { "uri": "media/posts/12345.jpg", "title": "Just joined Threads!" } + ] +} + +// Bluesky output +{ + "$type": "app.bsky.feed.post", + "text": "Just joined Threads!", + "createdAt": "2023-11-14T22:13:20.000Z", + "embed": { + "$type": "app.bsky.embed.images", + "images": [{ "image": { "$type": "blob", "ref": "...", "mimeType": "image/jpeg", "size": 12345 }, "alt": "Just joined Threads!" }] + } +} +``` + +--- + +## Nostr + +### Source file + +JSON array of Nostr events (kind 1 = text notes). May also be NDJSON or wrapped in `{ events: [...] }`. + +### Timestamp format + +Unix epoch seconds (integer) in `created_at`. Convert to ISO 8601. + +### Field mapping + +| Nostr field | Bluesky field | Transform | +|-------------|---------------|-----------| +| `content` | `text` | Plain text. Truncate at 300 graphemes. | +| `created_at` | `createdAt` | Unix seconds → ISO 8601 | +| `tags[]` where `[0] === "t"` | `facets[]` (tag) | Find `#tag` in content, compute UTF-8 byte offsets. `tag` = second element. | +| `tags[]` where `[0] === "r"` | `facets[]` (link) | Find URL in content, compute byte offsets. `uri` = second element. | +| `tags[]` where `[0] === "p"` | `facets[]` (mention) | **Cannot resolve DID from npub.** Store as link facet to `nostr:npub1...` or skip. | +| `tags[]` where `[0] === "e"` + marker `"root"` | `reply.root` | Nostr event ID → must map to AT URI of imported post. | +| `tags[]` where `[0] === "e"` + marker `"reply"` | `reply.parent` | Same — requires import ordering and URI mapping. | +| `tags[]` where `[0] === "q"` | `embed.record` | Quoted event — only if also imported. Otherwise `embed.external` with `nostr:nevent1...` URI. | +| `tags[]` where `[0] === "imeta"` | `embed.images[]` | Parse `imeta` tag for `url`, `m`, `alt`, `dim`. Upload via `uploadBlob`. | + +### Key gotchas + +- **`nostr:` URI scheme**: Content may contain `nostr:npub1...` or `nostr:nevent1...` references. These are human-readable but not resolvable to Bluesky DIDs. Convert to link facets. +- **NIP-10 marked `e` tags**: Modern events use `["e", id, relay, marker]` where marker is `"root"` or `"reply"`. Older events use positional `e` tags without markers — the last `e` tag is the parent, the first is the root. +- **`q` tags for quotes**: NIP-18 defines `["q", event_id, relay_url, author_pubkey]` for quote references. These map to `app.bsky.embed.record` if the quoted event was also imported. +- **`imeta` for media**: NIP-92 defines `imeta` tags with space-delimited key/value pairs: `url`, `m` (MIME type), `alt`, `dim` (dimensions). Parse these for image embeds. +- **Kind 6 = reposts**: Skip these — they're reposts of others' content, like Twitter retweets. +- **No content warnings**: Nostr has no CW mechanism. Some clients use a convention of putting CW text in the first line before a `\n\n`, but this is not standardised. + +### Example Nostr event → Bluesky record + +```json +// Nostr input +{ + "id": "abc123", + "pubkey": "def456", + "created_at": 1710000000, + "kind": 1, + "content": "Hello Nostr! #nostr https://example.com", + "tags": [ + ["t", "nostr"], + ["r", "https://example.com"] + ], + "sig": "..." +} + +// Bluesky output +{ + "$type": "app.bsky.feed.post", + "text": "Hello Nostr! #nostr https://example.com", + "createdAt": "2024-03-09T16:00:00.000Z", + "facets": [ + { "index": { "byteStart": 12, "byteEnd": 18 }, "features": [{ "$type": "app.bsky.richtext.facet#tag", "tag": "nostr" }] }, + { "index": { "byteStart": 19, "byteEnd": 37 }, "features": [{ "$type": "app.bsky.richtext.facet#link", "uri": "https://example.com" }] } + ] +} +``` + +--- + +## Cross-platform concerns + +### Reply threading + +All four platforms represent replies differently, and none can directly reference a Bluesky post. The publisher must: + +1. **Import posts in chronological order within each thread** (oldest first). +2. **Maintain a mapping** of `platform:originalId` → `{ uri, cid }` for every successfully created post. +3. **Resolve reply references** using this mapping when creating subsequent posts. +4. **Skip reply refs** where the parent hasn't been imported (orphan replies become standalone posts). + +### Media upload flow + +Images must be uploaded as blobs before the post record is created: + +1. Fetch/download the image from the source URL or local path. +2. Upload via `com.atproto.repo.uploadBlob` — returns `{ blob: { ref, mimeType, size } }`. +3. Include the blob ref in `embed.images[].image`. +4. Max 4 images per post. Max 1MB per blob. + +### Text truncation + +When a post exceeds 300 grapheme clusters: +- Truncate to 299 graphemes + `…` +- Set `truncated: true` on the `MicroblogPost` +- Optionally append a facet link to the original post URL + +### Mention resolution + +No source platform stores Bluesky DIDs. Options: +- **Link facet**: Point to the original profile URL (Twitter, Mastodon) or `nostr:npub1...` (Nostr). User can click through. +- **Skip**: Don't create mention facets for unresolved handles. +- **Best-effort lookup**: Resolve the source handle to a Bluesky DID via `com.atproto.identity.resolveHandle` — but this only works if the person has the same handle on Bluesky. + +### Language detection + +- Twitter provides `lang` (e.g. `"en"`, `"und"`). Map to BCP-47. +- Mastodon has `contentMap` but no explicit lang field per toot. Infer from content or skip. +- Threads and Nostr don't provide language metadata. Use `und` or detect from content. diff --git a/packages/opal/src/index.ts b/packages/opal/src/index.ts index 1ef457a..557b976 100644 --- a/packages/opal/src/index.ts +++ b/packages/opal/src/index.ts @@ -39,13 +39,16 @@ export type { Facet, FacetFeature, ByteSlice, + TwitterArchiveEntry, TwitterTweet, TwitterMediaEntity, + TwitterUrlEntity, MastodonOutboxItem, MastodonStatus, MastodonAttachment, MastodonTag, ThreadsPost, + ThreadsMedia, NostrEvent, } from './types.js'; diff --git a/packages/opal/src/mastodon.ts b/packages/opal/src/mastodon.ts index 3339693..98f2a6a 100644 --- a/packages/opal/src/mastodon.ts +++ b/packages/opal/src/mastodon.ts @@ -1,14 +1,16 @@ /** - * Mastodon outbox/CSV parser. + * Mastodon outbox parser. * * Parses ActivityPub outbox JSON (OrderedCollection) from a Mastodon export. - * CSV export support can be added later. + * Handles Create (Note) and Announce (boost) activities — boosts are skipped. + * Content warnings are preserved as a prefix on the text. */ -import type { MastodonOutboxItem, ConvertResult, MicroblogPost, Facet } from './types.js'; -import { stripHtml, truncateForAtProto, tagFacet } from './utils.js'; +import type { MastodonOutboxItem, MastodonAttachment, ConvertResult, MicroblogPost, Facet } from './types.js'; +import { stripHtml, truncateForAtProto, linkFacet, tagFacet } from './utils.js'; export function convertMastodon(data: unknown): ConvertResult { + // Handle both OrderedCollection wrapper and bare array const outbox = data as { orderedItems?: MastodonOutboxItem[] }; const items = outbox.orderedItems ?? (data as MastodonOutboxItem[]); const posts: MicroblogPost[] = []; @@ -17,6 +19,12 @@ export function convertMastodon(data: unknown): ConvertResult { for (const item of items) { try { + // Skip boosts (Announce activities) — they're someone else's content + if (item.type === 'Announce') { + skipped++; + continue; + } + // Only process Create activities with Note objects if (item.type !== 'Create' || !item.object) { continue; @@ -27,7 +35,14 @@ export function convertMastodon(data: unknown): ConvertResult { continue; } - const text = stripHtml(status.content).trim(); + // Strip HTML and build plain text + let text = stripHtml(status.content).trim(); + + // Prepend content warning if present + if (status.sensitive && status.summary) { + text = `[CW: ${status.summary.trim()}]\n\n${text}`; + } + if (!text) { skipped++; continue; @@ -35,22 +50,53 @@ export function convertMastodon(data: unknown): ConvertResult { const facets: Facet[] = []; - // Convert hashtags + // Convert hashtags and mentions if (status.tag) { for (const tag of status.tag) { if (tag.type === 'Hashtag' && tag.name) { - // Mastodon hashtags include the # prefix const tagName = tag.name.startsWith('#') ? tag.name : `#${tag.name}`; const charIndex = text.indexOf(tagName); if (charIndex !== -1) { facets.push(tagFacet(text, charIndex, charIndex + tagName.length, tagName.slice(1))); } } + + // Mentions — cannot resolve to Bluesky DID, use link facet to profile + if (tag.type === 'Mention' && tag.name && tag.href) { + const mentionText = tag.name.startsWith('@') ? tag.name : `@${tag.name}`; + const charIndex = text.indexOf(mentionText); + if (charIndex !== -1) { + facets.push(linkFacet(text, charIndex, charIndex + mentionText.length, tag.href)); + } + } + } + } + + // Detect bare URLs in text for link facets + const urlRegex = /https?:\/\/[^\s<]+/g; + let match: RegExpExecArray | null; + while ((match = urlRegex.exec(text)) !== null) { + // Skip if this URL is already covered by a tag facet + const start = match.index; + const end = start + match[0].length; + const overlaps = facets.some( + (f) => start < f.index.byteEnd && end > f.index.byteStart, + ); + if (!overlaps) { + facets.push(linkFacet(text, start, end, match[0])); } } const { text: finalText, truncated } = truncateForAtProto(text); + // Filter for image attachments only (videos have no Bluesky equivalent) + const imageAttachments = status.attachment?.filter( + (a: MastodonAttachment) => + a.mediaType?.startsWith('image/') || + a.type === 'Image' || + (a.type === 'Document' && a.url?.match(/\.(jpg|jpeg|png|gif|webp|svg)$/i) !== null), + ); + const post: MicroblogPost = { text: finalText, createdAt: status.published, @@ -58,8 +104,11 @@ export function convertMastodon(data: unknown): ConvertResult { originalId: status.id, facets: facets.length > 0 ? facets : undefined, truncated, - mediaUris: status.attachment?.map((a) => a.url), + mediaUris: imageAttachments?.map((a) => a.url), + mediaAlt: imageAttachments?.map((a) => a.name ?? ''), + mediaMimeTypes: imageAttachments?.map((a) => a.mediaType ?? 'image/jpeg'), replyTo: status.inReplyTo, + contentWarning: status.sensitive && status.summary ? status.summary.trim() : undefined, }; posts.push(post); diff --git a/packages/opal/src/nostr.ts b/packages/opal/src/nostr.ts index 08f3990..2a7edfd 100644 --- a/packages/opal/src/nostr.ts +++ b/packages/opal/src/nostr.ts @@ -1,13 +1,91 @@ /** * Nostr event parser. * - * Parses an array of Nostr events (kind 1 = text notes). - * Events are typically exported from a Nostr client as a JSON array. + * Parses an array of Nostr events. Kind 1 = text notes, kind 6 = reposts + * (skipped). Handles NIP-10 marked e tags, NIP-18 q tags, NIP-12 t tags, + * NIP-27 nostr: mentions, and NIP-92 imeta media tags. */ import type { NostrEvent, ConvertResult, MicroblogPost, Facet } from './types.js'; import { parseUnixTimestamp, truncateForAtProto, linkFacet, tagFacet } from './utils.js'; +interface ParsedETag { + eventId: string; + relayUrl?: string; + marker?: string; // "root", "reply", "mention" + authorPubkey?: string; +} + +interface ParsedImeta { + url: string; + mimeType?: string; + alt?: string; + dimensions?: string; // e.g. "3024x4032" +} + +function parseETags(tags: string[][]): { root?: ParsedETag; reply?: ParsedETag; mentions: ParsedETag[] } { + const eTags: ParsedETag[] = []; + + for (const tag of tags) { + if (tag[0] !== 'e' || tag.length < 2) continue; + eTags.push({ + eventId: tag[1], + relayUrl: tag[2], + marker: tag[3], + authorPubkey: tag[4], + }); + } + + // If there are marked e tags, use those + const marked = eTags.filter((t) => t.marker === 'root' || t.marker === 'reply'); + if (marked.length > 0) { + return { + root: marked.find((t) => t.marker === 'root'), + reply: marked.find((t) => t.marker === 'reply'), + mentions: eTags.filter((t) => t.marker === 'mention'), + }; + } + + // Fallback: positional e tags (deprecated but still common) + // First e tag = root, last e tag = parent reply + if (eTags.length === 1) { + return { root: eTags[0], reply: eTags[0], mentions: [] }; + } + if (eTags.length >= 2) { + return { root: eTags[0], reply: eTags[eTags.length - 1], mentions: [] }; + } + + return { mentions: [] }; +} + +function parseImetaTags(tags: string[][]): ParsedImeta[] { + const images: ParsedImeta[] = []; + + for (const tag of tags) { + if (tag[0] !== 'imeta') continue; + + const imeta: ParsedImeta = { url: '' }; + for (let i = 1; i < tag.length; i++) { + const part = tag[i]; + const spaceIdx = part.indexOf(' '); + if (spaceIdx === -1) continue; + const key = part.slice(0, spaceIdx); + const value = part.slice(spaceIdx + 1); + + if (key === 'url') imeta.url = value; + else if (key === 'm') imeta.mimeType = value; + else if (key === 'alt') imeta.alt = value; + else if (key === 'dim') imeta.dimensions = value; + } + + if (imeta.url) { + images.push(imeta); + } + } + + return images; +} + export function convertNostr(data: unknown): ConvertResult { const events = Array.isArray(data) ? (data as NostrEvent[]) : []; const posts: MicroblogPost[] = []; @@ -16,7 +94,11 @@ export function convertNostr(data: unknown): ConvertResult { for (const event of events) { try { - // Only process text notes (kind 1) + // Only process text notes (kind 1). Skip kind 6 (reposts). + if (event.kind === 6) { + skipped++; + continue; + } if (event.kind !== 1) { continue; } @@ -29,22 +111,19 @@ export function convertNostr(data: unknown): ConvertResult { const facets: Facet[] = []; - // Parse tags for reply and quote references - let replyTo: string | undefined; - let quoteUri: string | undefined; + // Parse e tags for reply threading + const { root, reply } = parseETags(event.tags); + // Parse q tags for quote references (NIP-18) + let quoteUri: string | undefined; for (const tag of event.tags) { - // ["e", "", "", ""] - if (tag[0] === 'e' && tag.length >= 4) { - const marker = tag[3]; - if (marker === 'reply' || marker === 'root') { - replyTo = `nostr:${tag[1]}`; - } else if (marker === 'mention') { - quoteUri = `nostr:${tag[1]}`; - } + if (tag[0] === 'q' && tag[1]) { + quoteUri = `nostr:${tag[1]}`; } + } - // ["t", ""] + // Parse t tags for hashtags (NIP-12) + for (const tag of event.tags) { if (tag[0] === 't' && tag[1]) { const tagName = tag[1].startsWith('#') ? tag[1] : `#${tag[1]}`; const charIndex = text.indexOf(tagName); @@ -52,8 +131,10 @@ export function convertNostr(data: unknown): ConvertResult { facets.push(tagFacet(text, charIndex, charIndex + tagName.length, tag[1])); } } + } - // ["r", ""] + // Parse r tags for URLs + for (const tag of event.tags) { if (tag[0] === 'r' && tag[1]) { const url = tag[1]; const charIndex = text.indexOf(url); @@ -63,6 +144,27 @@ export function convertNostr(data: unknown): ConvertResult { } } + // Parse p tags for mentions — cannot resolve to Bluesky DID + // Convert nostr:npub references in text to link facets + for (const tag of event.tags) { + if (tag[0] === 'p' && tag[1]) { + // Find nostr:npub1... or nostr:nprofile1... references in text + const nostrRefRegex = /nostr:(nprofile1[a-zA-Z0-9]+|npub1[a-zA-Z0-9]+)/g; + let match: RegExpExecArray | null; + while ((match = nostrRefRegex.exec(text)) !== null) { + facets.push(linkFacet( + text, + match.index, + match.index + match[0].length, + `nostr:${tag[1]}`, + )); + } + } + } + + // Parse imeta tags for media (NIP-92) + const imetaImages = parseImetaTags(event.tags); + const { text: finalText, truncated } = truncateForAtProto(text); const post: MicroblogPost = { @@ -73,8 +175,14 @@ export function convertNostr(data: unknown): ConvertResult { originalUrl: `nostr:nevent1${event.id}`, facets: facets.length > 0 ? facets : undefined, truncated, - replyTo, + // Reply threading + replyTo: reply ? `nostr:${reply.eventId}` : undefined, + threadRoot: root ? `nostr:${root.eventId}` : undefined, quoteUri, + // Media from imeta tags + mediaUris: imetaImages.length > 0 ? imetaImages.map((i) => i.url) : undefined, + mediaAlt: imetaImages.length > 0 ? imetaImages.map((i) => i.alt ?? '') : undefined, + mediaMimeTypes: imetaImages.length > 0 ? imetaImages.map((i) => i.mimeType ?? 'image/jpeg') : undefined, }; posts.push(post); diff --git a/packages/opal/src/publisher.ts b/packages/opal/src/publisher.ts index c35379e..7c2eb9b 100644 --- a/packages/opal/src/publisher.ts +++ b/packages/opal/src/publisher.ts @@ -91,14 +91,26 @@ function toPostRecord(post: MicroblogPost): Record { })); } - // Reply reference — if the reply target is a Bluesky post URI + // Reply reference — requires both root and parent as strongRefs + // The publisher must track previously created posts' URIs and CIDs if (post.replyTo?.startsWith('at://')) { + const rootUri = post.threadRoot?.startsWith('at://') ? post.threadRoot : post.replyTo; record.reply = { - root: { uri: post.replyTo, cid: '' }, // CID resolved at publish time + root: { uri: rootUri, cid: '' }, // CID filled in by publisher from tracking map parent: { uri: post.replyTo, cid: '' }, }; } + // Language tags + if (post.langs && post.langs.length > 0) { + record.langs = post.langs; + } + + // Tags (not hashtags — those are facets. This is for free-form post tags) + if (post.contentWarning) { + record.tags = [`cw:${post.contentWarning}`]; + } + return record; } diff --git a/packages/opal/src/threads.ts b/packages/opal/src/threads.ts index 8df6c62..06fdcce 100644 --- a/packages/opal/src/threads.ts +++ b/packages/opal/src/threads.ts @@ -1,12 +1,16 @@ /** * Threads export parser. * - * Meta's data export for Threads is limited. This parser handles - * whatever JSON structure is available. + * Meta doesn't publish a schema for Threads exports. The format is + * Instagram-archive-shaped: posts use `title` for text and + * `creation_timestamp` (Unix seconds) for timestamps. + * + * This parser handles both the documented Instagram-style format and + * any alternative field names that may appear. */ import type { ThreadsPost, ConvertResult, MicroblogPost } from './types.js'; -import { truncateForAtProto } from './utils.js'; +import { truncateForAtProto, parseUnixTimestamp } from './utils.js'; export function convertThreads(data: unknown): ConvertResult { const threadPosts = Array.isArray(data) ? (data as ThreadsPost[]) : []; @@ -16,28 +20,59 @@ export function convertThreads(data: unknown): ConvertResult { for (const post of threadPosts) { try { - const text = post.text?.trim(); + // Threads uses `title` for post text, fallback to `text` + const text = (post.title ?? post.text)?.trim(); if (!text) { skipped++; continue; } + // Threads uses `creation_timestamp` (Unix seconds), fallback to `timestamp` (ISO 8601) + let createdAt: string; + if (post.creation_timestamp != null) { + createdAt = parseUnixTimestamp(post.creation_timestamp); + } else if (post.timestamp) { + createdAt = post.timestamp; // Already ISO 8601 + } else { + skipped++; + continue; + } + const { text: finalText, truncated } = truncateForAtProto(text); + // Regex-detect hashtags in text (Threads export doesn't include structured entity data) + const hashtagRegex = /#(\w+)/g; + const facets: MicroblogPost['facets'] = []; + let match: RegExpExecArray | null; + while ((match = hashtagRegex.exec(finalText)) !== null) { + const start = match.index; + const end = start + match[0].length; + const { byteStart, byteEnd } = { + byteStart: new TextEncoder().encode([...finalText].slice(0, start).join('')).byteLength, + byteEnd: new TextEncoder().encode([...finalText].slice(0, end).join('')).byteLength, + }; + facets!.push({ + index: { byteStart, byteEnd }, + features: [{ $type: 'app.bsky.richtext.facet#tag', tag: match[1] }], + }); + } + const result: MicroblogPost = { text: finalText, - createdAt: post.timestamp, + createdAt, platform: 'threads', - originalId: post.id, + originalId: post.id ?? `threads-${post.creation_timestamp}`, truncated, mediaUris: post.media?.map((m) => m.uri), + mediaAlt: post.media?.map((m) => m.title ?? ''), replyTo: post.parent_id ? `threads:${post.parent_id}` : undefined, quoteUri: post.quote_id ? `threads:${post.quote_id}` : undefined, + facets: facets && facets.length > 0 ? facets : undefined, }; posts.push(result); } catch (err) { - errors.push(`Threads post ${post.id}: ${(err as Error).message}`); + errors.push(`Threads post: ${(err as Error).message}`); skipped++; } } diff --git a/packages/opal/src/twitter.ts b/packages/opal/src/twitter.ts index f4a42c2..227a54a 100644 --- a/packages/opal/src/twitter.ts +++ b/packages/opal/src/twitter.ts @@ -4,72 +4,165 @@ * Parses the `tweets.js` file from a Twitter data export. * The file assigns to `window.YTD.tweets.part0` — the convert dispatcher * already extracts the JSON array before passing it here. + * + * Each entry is wrapped: { tweet: { ... } } */ -import type { TwitterTweet, ConvertResult, MicroblogPost, Facet } from './types.js'; +import type { TwitterArchiveEntry, TwitterTweet, ConvertResult, MicroblogPost, Facet } from './types.js'; import { parseTwitterTimestamp, truncateForAtProto, linkFacet, tagFacet, twitterUrl } from './utils.js'; +/** + * Replace t.co short URLs in tweet text with their expanded forms. + * Returns the cleaned text and a map of original indices → new positions. + */ +function replaceTcoUrls( + text: string, + urls: { url: string; expanded_url: string; indices: [number, number] }[] | undefined, + media: { url: string; indices: [number, number] }[] | undefined, +): string { + if (!urls && !media) return text; + + // Collect all t.co replacements (URLs + media shortlinks) + const replacements: { start: number; end: number; replacement: string }[] = []; + + if (urls) { + for (const url of urls) { + replacements.push({ + start: url.indices[0], + end: url.indices[1], + replacement: url.expanded_url, + }); + } + } + + // Media t.co links are replaced with the display text (not the URL) + if (media) { + for (const m of media) { + replacements.push({ + start: m.indices[0], + end: m.indices[1], + replacement: '', // Remove the t.co media link — media is handled separately + }); + } + } + + // Sort by position (descending) to replace from end to start + replacements.sort((a, b) => b.start - a.start); + + let result = text; + for (const r of replacements) { + result = result.slice(0, r.start) + r.replacement + result.slice(r.end); + } + + return result; +} + export function convertTwitter(data: unknown): ConvertResult { - const tweets = data as TwitterTweet[]; + const entries = data as TwitterArchiveEntry[]; const posts: MicroblogPost[] = []; const errors: string[] = []; let skipped = 0; - for (const tweet of tweets) { + for (const entry of entries) { try { + // Handle both wrapped ({ tweet: {...} }) and bare tweet objects + const tweet: TwitterTweet = entry.tweet ?? (entry as unknown as TwitterTweet); + // Skip retweets — they're someone else's content if (tweet.retweeted_status) { skipped++; continue; } - const text = tweet.full_text?.trim(); - if (!text) { + const rawText = tweet.full_text?.trim(); + if (!rawText) { skipped++; continue; } + // Replace t.co URLs with expanded URLs in the text + const allMedia = tweet.extended_entities?.media ?? tweet.entities?.media; + const text = replaceTcoUrls(rawText, tweet.entities?.urls, allMedia); + const facets: Facet[] = []; // Convert Twitter entities to ATProto facets + // NOTE: After t.co replacement, character positions have shifted. + // We need to find the expanded URLs and hashtags in the new text. if (tweet.entities) { - // URLs — replace t.co links with expanded URLs + // URLs — find expanded URLs in the replaced text if (tweet.entities.urls) { for (const url of tweet.entities.urls) { - facets.push(linkFacet(text, url.indices[0], url.indices[1], url.expanded_url)); + const charIndex = text.indexOf(url.expanded_url); + if (charIndex !== -1) { + facets.push(linkFacet(text, charIndex, charIndex + url.expanded_url.length, url.expanded_url)); + } } } - // Hashtags + // Hashtags — find in the cleaned text if (tweet.entities.hashtags) { for (const hashtag of tweet.entities.hashtags) { - facets.push(tagFacet(text, hashtag.indices[0], hashtag.indices[1] + 1, hashtag.text)); + const tagWithHash = `#${hashtag.text}`; + const charIndex = text.indexOf(tagWithHash); + if (charIndex !== -1) { + facets.push(tagFacet(text, charIndex, charIndex + tagWithHash.length, hashtag.text)); + } + } + } + + // User mentions — cannot resolve to Bluesky DID, use link facet + if (tweet.entities.user_mentions) { + for (const mention of tweet.entities.user_mentions) { + const mentionText = `@${mention.screen_name}`; + const charIndex = text.indexOf(mentionText); + if (charIndex !== -1) { + facets.push(linkFacet( + text, + charIndex, + charIndex + mentionText.length, + `https://bsky.app/profile/${mention.screen_name}.bsky.social`, + )); + } } } } const { text: finalText, truncated } = truncateForAtProto(text); + // Use extended_entities for media (includes all media, not just first) + const mediaEntities = tweet.extended_entities?.media ?? tweet.entities?.media; + const imageMedia = mediaEntities?.filter((m) => m.type === 'photo'); + const post: MicroblogPost = { text: finalText, createdAt: parseTwitterTimestamp(tweet.created_at), platform: 'twitter', originalId: tweet.id_str, - originalUrl: twitterUrl('_', tweet.id_str), // username filled in by caller if known + originalUrl: twitterUrl('_', tweet.id_str), facets: facets.length > 0 ? facets : undefined, truncated, - mediaUris: tweet.entities?.media?.map((m) => m.media_url_https), + mediaUris: imageMedia?.map((m) => m.media_url_https), + mediaAlt: imageMedia?.map(() => ''), // Twitter archives don't include alt text + mediaMimeTypes: imageMedia?.map(() => 'image/jpeg'), // Twitter media is typically JPEG + // Reply threading replyTo: tweet.in_reply_to_status_id_str ? `twitter:status:${tweet.in_reply_to_status_id_str}` : undefined, + threadRoot: tweet.conversation_id_str && tweet.conversation_id_str !== tweet.id_str + ? `twitter:status:${tweet.conversation_id_str}` + : undefined, + // Quote tweets quoteUri: tweet.quoted_status_id_str ? twitterUrl('_', tweet.quoted_status_id_str) : undefined, + // Language + langs: tweet.lang && tweet.lang !== 'und' ? [tweet.lang] : undefined, }; posts.push(post); } catch (err) { - errors.push(`Twitter tweet ${tweet.id_str}: ${(err as Error).message}`); + errors.push(`Twitter tweet: ${(err as Error).message}`); skipped++; } } diff --git a/packages/opal/src/types.ts b/packages/opal/src/types.ts index d74098e..b4930d6 100644 --- a/packages/opal/src/types.ts +++ b/packages/opal/src/types.ts @@ -16,9 +16,15 @@ export interface MicroblogPost { text: string; /** ISO 8601 timestamp of the original post */ createdAt: string; - /** Local file paths or URLs for media attachments */ + /** URLs or local paths for media attachments */ mediaUris?: string[]; - /** URI of parent post (for threading) */ + /** Alt text for media attachments (parallel to mediaUris) */ + mediaAlt?: string[]; + /** MIME types for media attachments (parallel to mediaUris) */ + mediaMimeTypes?: string[]; + /** URI of the root post in a thread (for Bluesky reply.root) */ + threadRoot?: string; + /** URI of parent post (for Bluesky reply.parent) */ replyTo?: string; /** URI of quoted post */ quoteUri?: string; @@ -32,6 +38,10 @@ export interface MicroblogPost { facets?: Facet[]; /** Whether this post was truncated to fit the 300-grapheme limit */ truncated?: boolean; + /** BCP-47 language tag(s) */ + langs?: string[]; + /** Content warning text (Mastodon CWs — prepend to text) */ + contentWarning?: string; } // ─── ATProto facet types ──────────────────────────────────────────────────── @@ -77,75 +87,134 @@ export interface ConvertOptions { // ─── Platform-specific input types ────────────────────────────────────────── -/** Twitter archive tweet object (from tweets.js) */ +/** + * Twitter archive entry wrapper. + * Each entry in the tweets.js array is wrapped: { tweet: { ... } } + */ +export interface TwitterArchiveEntry { + tweet: TwitterTweet; +} + +/** Twitter archive tweet object */ export interface TwitterTweet { id_str: string; full_text: string; created_at: string; // "Tue Jun 15 20:32:00 +0000 2021" + lang?: string; // e.g. "en", "und" in_reply_to_status_id_str?: string; in_reply_to_user_id_str?: string; + in_reply_to_screen_name?: string; + conversation_id_str?: string; + is_quote_status?: boolean; quoted_status_id_str?: string; + quoted_status_permalink?: { + url: string; + expanded: string; + display: string; + }; + retweeted_status?: TwitterTweet; entities?: { media?: TwitterMediaEntity[]; - urls?: { url: string; expanded_url: string; indices: [number, number] }[]; + urls?: TwitterUrlEntity[]; hashtags?: { text: string; indices: [number, number] }[]; - user_mentions?: { screen_name: string; id_str: string; indices: [number, number] }[]; + user_mentions?: { screen_name: string; name: string; id_str: string; indices: [number, number] }[]; + symbols?: unknown[]; + }; + extended_entities?: { + media?: TwitterMediaEntity[]; }; - retweeted_status?: TwitterTweet; } export interface TwitterMediaEntity { + id_str: string; media_url_https: string; type: 'photo' | 'video' | 'animated_gif'; - url: string; + url: string; // t.co shortlink + display_url: string; + expanded_url: string; + indices: [number, number]; + sizes?: Record; + video_info?: { + variants: { bitrate?: number; content_type: string; url: string }[]; + }; +} + +export interface TwitterUrlEntity { + url: string; // t.co shortlink + expanded_url: string; + display_url: string; indices: [number, number]; } -/** Mastodon outbox item */ +/** Mastodon outbox item (Create activity wrapping a Note) */ export interface MastodonOutboxItem { - type: string; + type: string; // "Create", "Announce", etc. object: MastodonStatus; } export interface MastodonStatus { - type: string; + type: string; // "Note" id: string; content: string; // HTML published: string; // ISO 8601 inReplyTo?: string; + sensitive?: boolean; + summary?: string; // Content warning text attachment?: MastodonAttachment[]; tag?: MastodonTag[]; } export interface MastodonAttachment { - type: 'Document'; + type: string; // "Document", "Image", etc. mediaType?: string; url: string; - name?: string; + name?: string; // alt text + width?: number; + height?: number; } export interface MastodonTag { - type: 'Hashtag' | 'Mention'; + type: string; // "Hashtag", "Mention" name?: string; href?: string; } -/** Threads export post (limited schema) */ +/** + * Threads export post. + * Meta doesn't publish a schema — this is inferred from community reverse-engineering. + * The actual export uses `title` for text and `creation_timestamp` (Unix seconds). + */ export interface ThreadsPost { - id: string; + title?: string; // Post text/caption + creation_timestamp?: number; // Unix seconds + // Legacy/alternative fields (some exports use these instead) text?: string; - timestamp: string; // ISO 8601 - media?: { uri: string; type: string }[]; + timestamp?: string; + id?: string; + media?: ThreadsMedia[]; + // These are rarely present in the export parent_id?: string; quote_id?: string; } +export interface ThreadsMedia { + uri: string; // Local path in archive, e.g. "media/posts/12345.jpg" + title?: string; // Caption/alt text + creation_timestamp?: number; + media_metadata?: { + photo_metadata?: { + exif_data?: Array<{ latitude?: number; longitude?: number }>; + }; + }; +} + /** Nostr event (kind 1 = text note) */ export interface NostrEvent { id: string; kind: number; content: string; - created_at: number; // Unix timestamp + created_at: number; // Unix timestamp (seconds) tags: string[][]; pubkey: string; + sig?: string; } -- 2.51.2