diff --git a/internal/materialize/richtext.go b/internal/materialize/richtext.go
index 3f6fa8a..296558e 100644
--- a/internal/materialize/richtext.go
+++ b/internal/materialize/richtext.go
@@ -13,6 +13,7 @@ import (
"github.com/yuin/goldmark/ast"
"github.com/yuin/goldmark/extension"
east "github.com/yuin/goldmark/extension/ast"
+ "github.com/yuin/goldmark/parser"
"github.com/yuin/goldmark/text"
"github.com/yuin/goldmark/util"
)
@@ -24,10 +25,12 @@ import (
// tables, and linkify — Lemmy's own dialect) and emits stripped text plus
// facets:
//
-// - heading/blockquote/codeBlock ranges span whole lines, excluding the
-// trailing newline; nested quotes become DISJOINT ranges with increasing
-// level (quote-in-quote containment is forbidden; cross-type containment
-// is fine), clamped at level 6
+// - heading/blockquote/codeBlock/spoiler ranges span whole lines, excluding
+// the trailing newline; nested quotes become DISJOINT ranges with
+// increasing level (quote-in-quote containment is forbidden; cross-type
+// containment is fine), clamped at level 6
+// - Lemmy's non-CommonMark `::: spoiler Title` containers parse via
+// spoiler.go and become a spoiler facet whose reason is the title
// - inline bold/italic/strikethrough/code/link map directly
// - lists have no facet by design: markers are normalized into the text
// ("• ", "1. ") and degrade as plain lines
@@ -47,6 +50,10 @@ const (
maxQuoteLevel = 6
maxHeadingLevel = 6
+ // The lexicon's caps on a spoiler facet's optional `reason`.
+ maxSpoilerReasonGraphemes = 32
+ maxSpoilerReasonBytes = 128
+
// parseSlack is how much markdown beyond the byte cap still gets parsed:
// markup renders shorter than its source (markers, link URLs), so a small
// slack keeps legitimate near-cap bodies intact while bounding
@@ -65,8 +72,16 @@ const (
var richTextMarkdown = goldmark.New(
goldmark.WithExtensions(extension.Strikethrough, extension.Table, extension.Linkify),
+ goldmark.WithParserOptions(parser.WithBlockParsers(
+ util.Prioritized(spoilerBlockParser{}, spoilerParserPriority),
+ )),
)
+// bareURLRe finds an http(s) URL in already-rendered plaintext. The leading
+// word boundary keeps "xhttps://…" from matching; the tail is trimmed by
+// trimURLTail, which the character class deliberately leaves to do its job.
+var bareURLRe = regexp.MustCompile(`(?i)\bhttps?://[^\s<>]+`)
+
// brTagRe matches every
variant (
,
,
):
// federated HTML uses them all, and a missed one concatenates words.
var brTagRe = regexp.MustCompile(`(?i)^
]*>$`)
@@ -152,6 +167,18 @@ func codeBlockFeature(language string) map[string]any {
return f
}
+// spoilerFeature carries the container's title as the lexicon's optional
+// `reason`. Unlike codeBlockFeature's language, an over-cap reason is
+// TRUNCATED rather than dropped: a clipped label still names what is hidden,
+// where a clipped language tag is meaningless.
+func spoilerFeature(reason string) map[string]any {
+ f := featureOf("spoiler")
+ if r := truncateText(strings.TrimSpace(reason), maxSpoilerReasonGraphemes, maxSpoilerReasonBytes); r != "" {
+ f["reason"] = r
+ }
+ return f
+}
+
// linkFeature gates every bridge-authored clickable uri behind the http(s)
// scheme check — the lexicon's format:"uri" would happily carry javascript:
// into clients. ok=false means keep the text, emit no facet.
@@ -233,7 +260,97 @@ func richTextFromMarkdown(md string) (string, []facetSpan) {
c := &richTextConverter{source: source}
c.renderBlocks(root, 0)
rendered := c.b.String()
- return rendered, splitContainingQuotes(rendered, c.facets)
+ spans := splitContainingQuotes(rendered, c.facets)
+ return rendered, append(spans, bareURLSpans(rendered, spans)...)
+}
+
+// bareURLSpans linkifies http(s) URLs that goldmark's Linkify extension never
+// saw. Linkify runs over the SOURCE bytes during inline parsing, but backslash
+// escapes are only resolved at RENDER time (unescapeMarkdown) — so a federated
+// app that escapes punctuation on the way out (PieFed writes
+// `https\://example.com`) yields plaintext that reads as a URL and carries no
+// facet at all. Scanning the finished plaintext catches those, and any other
+// linkify miss, at the point where offsets are already final.
+//
+// Ranges already spoken for are skipped: an existing link (Linkify worked, or
+// the author wrote `[url-ish text](other-url)` — whose text must keep pointing
+// at the destination the author chose), and code/codeBlock, where literal text
+// must never become clickable.
+func bareURLSpans(rendered string, existing []facetSpan) []facetSpan {
+ blocked := blockedLinkRanges(existing)
+ var out []facetSpan
+ // Regexp matches arrive in ascending order and blocked is disjoint and
+ // sorted, so one forward cursor decides every overlap — a body that is
+ // thousands of code spans around thousands of URLs stays linear.
+ next := 0
+ for _, m := range bareURLRe.FindAllStringIndex(rendered, -1) {
+ start := m[0]
+ end := start + len(trimURLTail(rendered[start:m[1]]))
+ if start >= end {
+ continue
+ }
+ for next < len(blocked) && blocked[next][1] <= start {
+ next++
+ }
+ if next < len(blocked) && blocked[next][0] < end {
+ continue
+ }
+ if feature, ok := linkFeature(rendered[start:end]); ok {
+ out = append(out, facetSpan{start: start, end: end, feature: feature})
+ }
+ }
+ return out
+}
+
+// blockedLinkRanges collects the ranges bareURLSpans must not touch, merged
+// into disjoint sorted intervals. Link and code ranges genuinely nest
+// (`[`x`](url)`), so merging — not just sorting — is what makes a single
+// forward cursor correct.
+func blockedLinkRanges(spans []facetSpan) [][2]int {
+ var ranges [][2]int
+ for _, s := range spans {
+ switch s.feature["$type"] {
+ case facetTypePrefix + "link", facetTypePrefix + "code", facetTypePrefix + "codeBlock":
+ ranges = append(ranges, [2]int{s.start, s.end})
+ }
+ }
+ if len(ranges) == 0 {
+ return nil
+ }
+ sort.Slice(ranges, func(i, j int) bool { return ranges[i][0] < ranges[j][0] })
+ merged := ranges[:1]
+ for _, r := range ranges[1:] {
+ last := &merged[len(merged)-1]
+ if r[0] <= last[1] {
+ last[1] = max(last[1], r[1])
+ } else {
+ merged = append(merged, r)
+ }
+ }
+ return merged
+}
+
+// trimURLTail applies GFM's extended-autolink tail rules to a candidate match:
+// trailing punctuation belongs to the sentence, not the link, and a closing
+// bracket counts only when the URL itself opened one (Wikipedia-style paths).
+func trimURLTail(u string) string {
+ for u != "" {
+ switch u[len(u)-1] {
+ case '?', '!', '.', ',', ':', ';', '*', '_', '~', '\'', '"':
+ case ')':
+ if strings.Count(u, ")") <= strings.Count(u, "(") {
+ return u
+ }
+ case ']':
+ if strings.Count(u, "]") <= strings.Count(u, "[") {
+ return u
+ }
+ default:
+ return u
+ }
+ u = u[:len(u)-1]
+ }
+ return u
}
// requestSep asks for a separator before the next written content. Competing
@@ -337,6 +454,8 @@ func (c *richTextConverter) renderBlock(n ast.Node, quoteLevel int) {
c.writeText("———")
case *ast.HTMLBlock:
c.renderHTMLBlock(t)
+ case *spoilerNode:
+ c.renderSpoiler(t, quoteLevel)
default:
if n.Kind() == east.KindTable {
c.renderTable(n)
@@ -415,6 +534,17 @@ func (c *richTextConverter) renderCodeBlock(n ast.Node, language string) {
c.addFacet(start, c.b.Len(), codeBlockFeature(language))
}
+// renderSpoiler emits the container's contents (fence lines already consumed
+// by the parser) under a spoiler facet carrying the title as `reason`. The
+// title itself stays out of the plaintext: duplicating it there would read as
+// a stray line for any client that ignores the facet, and `reason` is where
+// the lexicon puts it.
+func (c *richTextConverter) renderSpoiler(s *spoilerNode, quoteLevel int) {
+ start := c.blockFacetStart()
+ c.renderBlocks(s, quoteLevel)
+ c.addFacet(start, c.b.Len(), spoilerFeature(s.Reason))
+}
+
// renderList normalizes list markers into the canonical text — "• " bullets
// and "N. " ordinals, two spaces of indent per nesting level, one item per
// line. No facet by design: the lexicon deliberately has no list feature
diff --git a/internal/materialize/richtext_test.go b/internal/materialize/richtext_test.go
index c34b6ae..0826292 100644
--- a/internal/materialize/richtext_test.go
+++ b/internal/materialize/richtext_test.go
@@ -640,3 +640,211 @@ func TestMaterializeCommentStoresRichText(t *testing.T) {
facetTypePrefix + "bold",
}, types)
}
+
+// --- Lemmy spoiler containers (spoiler.go) ---
+
+// TestRichTextSpoilerContainer: Lemmy's non-CommonMark `::: spoiler` block
+// becomes a spoiler facet whose reason is the title. The fence lines must not
+// survive into the canonical plaintext — before spoiler.go they leaked as
+// literal ":::" lines and the hidden content rendered revealed.
+func TestRichTextSpoilerContainer(t *testing.T) {
+ md := "before\n\n::: spoiler Bonus Panel\nthe **hidden** thing\n:::\n\nafter"
+ content, facets := bridgedRichText(md, 10000, 100000)
+
+ assert.Equal(t, "before\n\nthe hidden thing\n\nafter", content)
+ assert.NotContains(t, content, ":::")
+ assert.NotContains(t, content, "Bonus Panel", "the title belongs in reason, not the text")
+
+ spoiler := requireOneFacet(t, facets, "spoiler")
+ assert.Equal(t, "the hidden thing", facetText(t, content, spoiler))
+ assert.Equal(t, "Bonus Panel", facetFeatures(t, spoiler)[0]["reason"])
+ requireWholeLines(t, content, spoiler)
+
+ // Inline markup inside the container still parses.
+ assert.Equal(t, "hidden", facetText(t, content, requireOneFacet(t, facets, "bold")))
+}
+
+func TestRichTextSpoilerWithoutTitle(t *testing.T) {
+ content, facets := bridgedRichText("::: spoiler\nhidden\n:::", 10000, 100000)
+ assert.Equal(t, "hidden", content)
+ feature := facetFeatures(t, requireOneFacet(t, facets, "spoiler"))[0]
+ _, hasReason := feature["reason"]
+ assert.False(t, hasReason, "a titleless container must omit reason, not send an empty one")
+}
+
+// TestRichTextSpoilerReasonTruncated: unlike a codeBlock language (dropped
+// when over cap), a clipped reason still names what is hidden — so it is
+// truncated to the lexicon's 32 graphemes rather than discarded.
+func TestRichTextSpoilerReasonTruncated(t *testing.T) {
+ title := strings.Repeat("ő", 60)
+ content, facets := bridgedRichText("::: spoiler "+title+"\nhidden\n:::", 10000, 100000)
+ assert.Equal(t, "hidden", content)
+
+ reason, ok := facetFeatures(t, requireOneFacet(t, facets, "spoiler"))[0]["reason"].(string)
+ require.True(t, ok, "an over-cap reason must be truncated, not dropped")
+ assert.LessOrEqual(t, graphemeCount(reason), maxSpoilerReasonGraphemes)
+ assert.LessOrEqual(t, len(reason), maxSpoilerReasonBytes)
+}
+
+// TestRichTextSpoilerUnclosed: markdown-it closes an open container at end of
+// input; so must we, or a missing fence would swallow the facet entirely.
+func TestRichTextSpoilerUnclosed(t *testing.T) {
+ content, facets := bridgedRichText("::: spoiler Ending\nhidden", 10000, 100000)
+ assert.Equal(t, "hidden", content)
+ assert.Equal(t, "hidden", facetText(t, content, requireOneFacet(t, facets, "spoiler")))
+}
+
+// TestRichTextSpoilerSiblingsStayDisjoint mirrors the real Mr Lovenstein post:
+// three sibling containers, each its own range.
+func TestRichTextSpoilerSiblingsStayDisjoint(t *testing.T) {
+ md := "::: spoiler One\nfirst\n:::\n\n::: spoiler Two\nsecond\n:::"
+ content, facets := bridgedRichText(md, 10000, 100000)
+ assert.Equal(t, "first\n\nsecond", content)
+
+ spoilers := findFacets(t, facets, "spoiler")
+ require.Len(t, spoilers, 2)
+ assert.Equal(t, "first", facetText(t, content, spoilers[0]))
+ assert.Equal(t, "One", facetFeatures(t, spoilers[0])[0]["reason"])
+ assert.Equal(t, "second", facetText(t, content, spoilers[1]))
+ assert.Equal(t, "Two", facetFeatures(t, spoilers[1])[0]["reason"])
+
+ _, firstEnd := facetIndex(t, spoilers[0])
+ secondStart, _ := facetIndex(t, spoilers[1])
+ assert.LessOrEqual(t, firstEnd, secondStart, "sibling spoilers must be disjoint")
+}
+
+// TestRichTextNonSpoilerContainerStaysText: only `spoiler` is a Lemmy
+// container. Anything else — and a name that merely starts with "spoiler" —
+// keeps today's degradation to plain text rather than inventing a facet.
+func TestRichTextNonSpoilerContainerStaysText(t *testing.T) {
+ for _, md := range []string{
+ "::: warning Careful\nbody\n:::",
+ ":::spoilered Nope\nbody\n:::",
+ } {
+ content, facets := bridgedRichText(md, 10000, 100000)
+ assert.Contains(t, content, "body")
+ assert.Empty(t, findFacets(t, facets, "spoiler"), "input %q must not open a spoiler", md)
+ }
+}
+
+// --- bare-URL linkification (bareURLSpans) ---
+
+// TestRichTextEscapedURLStillLinks is the PieFed case: it escapes punctuation
+// on the way out, writing `https\://…`. Linkify runs over the SOURCE bytes and
+// cannot see through the backslash, while unescapeMarkdown resolves it at
+// render time — so without the plaintext post-pass the stored text reads as a
+// URL and carries no facet at all.
+func TestRichTextEscapedURLStillLinks(t *testing.T) {
+ content, facets := bridgedRichText(`see https\://piefed.social/post/1258520 now`, 10000, 100000)
+ assert.Equal(t, "see https://piefed.social/post/1258520 now", content)
+
+ link := requireOneFacet(t, facets, "link")
+ assert.Equal(t, "https://piefed.social/post/1258520", facetText(t, content, link))
+ assert.Equal(t, "https://piefed.social/post/1258520", facetFeatures(t, link)[0]["uri"])
+}
+
+// TestRichTextAutoLinkNotDoubled: a URL Linkify already caught must not also
+// pick up a post-pass facet — two identical ranges would merge into one facet
+// carrying the same feature twice.
+func TestRichTextAutoLinkNotDoubled(t *testing.T) {
+ content, facets := bridgedRichText("go to https://example.com/p now", 10000, 100000)
+ require.Len(t, findFacets(t, facets, "link"), 1)
+ assert.Equal(t, "https://example.com/p", facetText(t, content, facets[0]))
+}
+
+// TestRichTextURLInCodeNotLinkified: literal text must never become
+// clickable, in either code form.
+func TestRichTextURLInCodeNotLinkified(t *testing.T) {
+ for _, md := range []string{
+ "`https://example.com/p`",
+ "```\nhttps://example.com/p\n```",
+ " https://example.com/p",
+ } {
+ content, facets := bridgedRichText(md, 10000, 100000)
+ assert.Contains(t, content, "https://example.com/p")
+ assert.Empty(t, findFacets(t, facets, "link"), "code must not linkify: %q", md)
+ }
+}
+
+// TestRichTextURLTextKeepsAuthorDestination: when the link TEXT is itself
+// URL-ish, the author's chosen destination wins — the post-pass must not
+// re-point it at the text.
+func TestRichTextURLTextKeepsAuthorDestination(t *testing.T) {
+ content, facets := bridgedRichText("[https://decoy.example](https://real.example/p)", 10000, 100000)
+ assert.Equal(t, "https://decoy.example", content)
+
+ links := findFacets(t, facets, "link")
+ require.Len(t, links, 1)
+ assert.Equal(t, "https://real.example/p", facetFeatures(t, links[0])[0]["uri"])
+}
+
+// TestRichTextBareURLTailTrimmed applies GFM's extended-autolink tail rules:
+// sentence punctuation is not part of the link, and a closing bracket counts
+// only when the URL opened one.
+func TestRichTextBareURLTailTrimmed(t *testing.T) {
+ for _, tc := range []struct{ md, want string }{
+ {`see https\://example.com/p.`, "https://example.com/p"},
+ {`see https\://example.com/p, then`, "https://example.com/p"},
+ {`(see https\://example.com/p)`, "https://example.com/p"},
+ {`see https\://en.wikipedia.org/wiki/Foo_(bar)`, "https://en.wikipedia.org/wiki/Foo_(bar)"},
+ } {
+ content, facets := bridgedRichText(tc.md, 10000, 100000)
+ link := requireOneFacet(t, facets, "link")
+ assert.Equal(t, tc.want, facetText(t, content, link), "input %q", tc.md)
+ assert.Equal(t, tc.want, facetFeatures(t, link)[0]["uri"], "input %q", tc.md)
+ }
+}
+
+// TestRichTextDeepSpoilerNestingBounded mirrors the deep-quote guard: an
+// adversarial container nest must terminate against maxRenderDepth rather
+// than exhaust the stack, and every surviving facet must stay in bounds.
+func TestRichTextDeepSpoilerNestingBounded(t *testing.T) {
+ md := strings.Repeat("::: spoiler x\n", 5000) + "deep"
+ content, facets := bridgedRichText(md, 10000, 100000)
+ for _, f := range facets {
+ facetText(t, content, f) // asserts bounds
+ }
+}
+
+// TestRichTextManyURLsAroundCode pins the linear overlap walk: interleaved
+// code spans and bare URLs must not go quadratic, and each URL outside code
+// still gets exactly one facet.
+func TestRichTextManyURLsAroundCode(t *testing.T) {
+ md := strings.Repeat("`https://code.example/x` https\\://bare.example/y\n\n", 500)
+ content, facets := bridgedRichText(md, 10000, 100000)
+ for _, f := range facets {
+ facetText(t, content, f) // asserts bounds
+ }
+ for _, link := range findFacets(t, facets, "link") {
+ assert.Equal(t, "https://bare.example/y", facetText(t, content, link),
+ "only the bare URL may linkify")
+ }
+}
+
+// TestMaterializePostStoresSpoilerFacet proves the new feature survives the
+// real wiring, including lexicon validation against the vendored catalog —
+// a #spoiler carrying a `reason` has to be a shape Coves' validator accepts,
+// not just one the converter is willing to emit.
+func TestMaterializePostStoresSpoilerFacet(t *testing.T) {
+ h := newHarness(t)
+ h.serveLemmyWorldFixtures()
+ page := loadFixtureObject(t, "page_lemmy_world.json")
+ page.Source = &ap.Source{
+ Content: "Mr Lovenstein\n\n::: spoiler Transcript\nCaveman 1: Same\n:::",
+ MediaType: "text/markdown",
+ }
+
+ _, err := h.m.MaterializePost(context.Background(), page)
+ require.NoError(t, err)
+
+ record := h.recordFor(t, pageID)
+ assert.Equal(t, "Mr Lovenstein\n\nCaveman 1: Same", record["content"])
+ facets, ok := record["facets"].([]any)
+ require.True(t, ok, "record must carry facets")
+
+ // Read back through the repo the indices are CBOR int64, not the int the
+ // facetIndex helper expects — so assert on the feature, whose byte range
+ // the converter-level tests already pin.
+ spoiler := requireOneFacet(t, facets, "spoiler")
+ assert.Equal(t, "Transcript", facetFeatures(t, spoiler)[0]["reason"])
+}
diff --git a/internal/materialize/spoiler.go b/internal/materialize/spoiler.go
new file mode 100644
index 0000000..4ee4525
--- /dev/null
+++ b/internal/materialize/spoiler.go
@@ -0,0 +1,101 @@
+package materialize
+
+import (
+ "regexp"
+
+ "github.com/yuin/goldmark/ast"
+ "github.com/yuin/goldmark/parser"
+ "github.com/yuin/goldmark/text"
+ "github.com/yuin/goldmark/util"
+)
+
+// Lemmy (and PieFed) spell spoilers with the markdown-it-container syntax,
+// which is not CommonMark and which goldmark therefore leaves as paragraph
+// text:
+//
+// ::: spoiler Bonus Panel
+// 
+// :::
+//
+// Lemmy renders that to Bonus Panel
… ,
+// and social.coves.richtext.facet has a matching #spoiler feature whose
+// optional `reason` is exactly the container title. Without a parser the
+// marker lines leak into the canonical plaintext AND the hidden content
+// renders revealed — so this file teaches goldmark the container as a real
+// block, and richtext.go maps it onto the facet.
+//
+// The fence is three or more colons; a container closes on the first
+// all-colon line at least as long as its opener, or at end of input. Nesting
+// falls out of goldmark's innermost-first close order rather than
+// markdown-it's longer-fence rule — Lemmy authors do not nest spoilers, and
+// the degradation (inner closes first) keeps every range well-formed.
+const spoilerParserPriority = 750 // between FencedCodeBlock (700) and Blockquote (800)
+
+// spoilerOpenRe matches an opening fence, capturing the colon run and the
+// optional title. `\b` after the container name keeps ":::spoilered" from
+// opening a spoiler; a bare "::: spoiler" (no title) is valid and titleless.
+var spoilerOpenRe = regexp.MustCompile(`^(:{3,})[ \t]*spoiler\b[ \t]*(.*?)[ \t]*$`)
+
+var kindSpoiler = ast.NewNodeKind("LemmySpoiler")
+
+// spoilerNode is a parsed `::: spoiler` container. fence records the opening
+// colon run so the matching close can require at least as many.
+type spoilerNode struct {
+ ast.BaseBlock
+ Reason string
+ fence int
+}
+
+func (n *spoilerNode) Kind() ast.NodeKind { return kindSpoiler }
+
+func (n *spoilerNode) Dump(source []byte, level int) {
+ ast.DumpHelper(n, source, level, map[string]string{"Reason": n.Reason}, nil)
+}
+
+// spoilerBlockParser parses the container as a child-bearing block, so its
+// contents keep their normal markdown structure (images, emphasis, lists).
+type spoilerBlockParser struct{}
+
+func (spoilerBlockParser) Trigger() []byte { return []byte{':'} }
+
+func (spoilerBlockParser) Open(parent ast.Node, reader text.Reader, pc parser.Context) (ast.Node, parser.State) {
+ line, _ := reader.PeekLine()
+ pos := pc.BlockOffset()
+ if pos < 0 {
+ return nil, parser.NoChildren
+ }
+ m := spoilerOpenRe.FindSubmatch(util.TrimRightSpace(line[pos:]))
+ if m == nil {
+ return nil, parser.NoChildren
+ }
+ // The fence line is metadata, not content: consume it whole so the title
+ // never reaches the plaintext (it becomes the facet's `reason`).
+ reader.AdvanceToEOL()
+ return &spoilerNode{Reason: string(m[2]), fence: len(m[1])}, parser.HasChildren
+}
+
+func (spoilerBlockParser) Continue(node ast.Node, reader text.Reader, pc parser.Context) parser.State {
+ spoiler, ok := node.(*spoilerNode)
+ if !ok {
+ return parser.Close
+ }
+ line, _ := reader.PeekLine()
+ // Mirror the fenced-code rule: a 4-space indent makes the line content,
+ // not a fence.
+ if w, pos := util.IndentWidth(line, reader.LineOffset()); w < 4 {
+ i := pos
+ for ; i < len(line) && line[i] == ':'; i++ {
+ }
+ if i-pos >= spoiler.fence && util.IsBlank(line[i:]) {
+ reader.AdvanceToEOL()
+ return parser.Close
+ }
+ }
+ return parser.Continue | parser.HasChildren
+}
+
+func (spoilerBlockParser) Close(node ast.Node, reader text.Reader, pc parser.Context) {}
+
+func (spoilerBlockParser) CanInterruptParagraph() bool { return true }
+
+func (spoilerBlockParser) CanAcceptIndentedLine() bool { return false }