diff --git a/email/render.go b/email/render.go index ba2bf5e..f5b0180 100644 --- a/email/render.go +++ b/email/render.go @@ -4,6 +4,8 @@ import ( "bytes" "embed" htmltemplate "html/template" + "regexp" + "strings" texttemplate "text/template" "time" @@ -36,7 +38,8 @@ type FeedItem struct { type templateFeedItem struct { Title string Link string - Content string // Original content for text template + Content string // Original content (unused, kept for compatibility) + PlainContent string // HTML-stripped content for text template SanitizedContent htmltemplate.HTML // Sanitized HTML for HTML template Published time.Time } @@ -48,9 +51,35 @@ type templateFeedGroup struct { Items []templateFeedItem } +// emailUnsafeTags are HTML5 semantic tags not supported by most email clients (Gmail, Outlook, etc.) +var emailUnsafeTags = regexp.MustCompile(`]*)?>`) + // sanitizeHTML sanitizes HTML content, allowing safe tags while stripping styles and unsafe elements func sanitizeHTML(html string) string { - return policy.Sanitize(html) + sanitized := policy.Sanitize(html) + // Strip HTML5 semantic tags that email clients don't support + return emailUnsafeTags.ReplaceAllString(sanitized, "") +} + +// htmlTagRegex matches HTML tags for stripping +var htmlTagRegex = regexp.MustCompile(`<[^>]*>`) + +// stripHTML removes all HTML tags and decodes entities for plain text output +func stripHTML(html string) string { + // First sanitize to ensure we're working with clean HTML + sanitized := policy.Sanitize(html) + // Strip all remaining HTML tags + text := htmlTagRegex.ReplaceAllString(sanitized, "") + // Decode common HTML entities + text = strings.ReplaceAll(text, "&", "&") + text = strings.ReplaceAll(text, "<", "<") + text = strings.ReplaceAll(text, ">", ">") + text = strings.ReplaceAll(text, """, "\"") + text = strings.ReplaceAll(text, "'", "'") + text = strings.ReplaceAll(text, " ", " ") + // Collapse multiple whitespace/newlines + text = regexp.MustCompile(`\s+`).ReplaceAllString(text, " ") + return strings.TrimSpace(text) } var ( @@ -86,6 +115,7 @@ func RenderDigest(data *DigestData, inline bool, daysUntilExpiry int, showUrgent Title: item.Title, Link: item.Link, Content: item.Content, + PlainContent: stripHTML(item.Content), SanitizedContent: htmltemplate.HTML(sanitizeHTML(item.Content)), // #nosec G203 -- Content is sanitized by bluemonday before conversion Published: item.Published, } @@ -116,15 +146,19 @@ func RenderDigest(data *DigestData, inline bool, daysUntilExpiry int, showUrgent ShowWarningBanner: showWarningBanner, } - // Prepare template data for text template (with original content) + // Prepare template data for text template (with plain text content) textTmplData := struct { - *DigestData + ConfigName string + TotalItems int + FeedGroups []templateFeedGroup Inline bool DaysUntilExpiry int ShowUrgentBanner bool ShowWarningBanner bool }{ - DigestData: data, + ConfigName: data.ConfigName, + TotalItems: data.TotalItems, + FeedGroups: sanitizedGroups, Inline: inline, DaysUntilExpiry: daysUntilExpiry, ShowUrgentBanner: showUrgentBanner, diff --git a/email/render_test.go b/email/render_test.go index 003c31b..5312d53 100644 --- a/email/render_test.go +++ b/email/render_test.go @@ -93,3 +93,44 @@ func TestRenderDigest_UnsafeHTMLStripped(t *testing.T) { t.Error("Safe HTML content was incorrectly removed") } } + +func TestRenderDigest_TextOutputNoHTMLTags(t *testing.T) { + // Create test data with HTML content + data := &DigestData{ + ConfigName: "Test Config", + TotalItems: 1, + FeedGroups: []FeedGroup{ + { + FeedName: "Test Feed", + FeedURL: "https://example.com/feed", + Items: []FeedItem{ + { + Title: "Test Article", + Link: "https://example.com/article", + Content: "

This is a test article with a link.

", + Published: time.Now(), + }, + }, + }, + }, + } + + // Render with inline mode enabled + _, textOutput, err := RenderDigest(data, true, 30, false, false) + if err != nil { + t.Fatalf("RenderDigest failed: %v", err) + } + + // Verify text output does NOT contain HTML tags + htmlTags := []string{"
", "

", "", "", "

", "", ""} + for _, tag := range htmlTags { + if strings.Contains(textOutput, tag) { + t.Errorf("Text output contains HTML tag %q - should be stripped", tag) + } + } + + // Verify the actual text content is present + if !strings.Contains(textOutput, "This is a test article with a link") { + t.Error("Text content was not preserved after HTML stripping") + } +} diff --git a/email/templates/digest.txt b/email/templates/digest.txt index 1d976eb..98ec2ac 100644 --- a/email/templates/digest.txt +++ b/email/templates/digest.txt @@ -14,9 +14,9 @@ Summary {{range .Items}} {{.Title}} {{.Link}} -{{if and $.Inline .Content}} +{{if and $.Inline .PlainContent}} -{{.Content}} +{{.PlainContent}} {{end}} {{end}} {{end}}