From 2df41d6d8d198a0798616efef77563baf61669a6 Mon Sep 17 00:00:00 2001 From: Michael Stahnke Date: Thu, 19 Feb 2026 16:18:05 -0600 Subject: [PATCH] feat: add Reddit video embed support with proper sizing Replace OEmbed blockquote with redditmedia.com iframe for Reddit video posts. Resolve /s/ share link redirects, fetch uncropped images and dimensions from the Reddit JSON API, and extract video metadata from OG tags. Size the embed iframe using actual video dimensions with aspect ratio clamping for portrait content. Respect caching config for Cache-Control headers on preview responses. --- internal/handler/preview.go | 27 ++- internal/handler/preview_reddit.go | 286 +++++++++++++++++++++++++--- internal/templates/views/index.html | 96 +++++++++- 3 files changed, 380 insertions(+), 29 deletions(-) diff --git a/internal/handler/preview.go b/internal/handler/preview.go index 44b5d80..5de35d3 100644 --- a/internal/handler/preview.go +++ b/internal/handler/preview.go @@ -10,8 +10,9 @@ import ( "strings" "time" - "golang.org/x/net/html" "tumble/internal/data" + + "golang.org/x/net/html" ) const ( @@ -121,7 +122,11 @@ func (h *Handler) TryServeCachedOGPreview(w http.ResponseWriter, r *http.Request // Cache hit - serve response w.Header().Set("Content-Type", "application/json") - w.Header().Set("Cache-Control", "public, max-age=86400") + if !h.Config.Caching.Enabled { + w.Header().Set("Cache-Control", "no-cache, no-store, must-revalidate") + } else { + w.Header().Set("Cache-Control", "public, max-age=86400") + } json.NewEncoder(w).Encode(meta) return true } @@ -284,8 +289,12 @@ func (h *Handler) cacheAndRespond(w http.ResponseWriter, r *http.Request, urlPar h.Store.InsertLinkPreview(r.Context(), urlParam, data) } } - // Client-side Caching Header (24h) - w.Header().Set("Cache-Control", "public, max-age=86400") + if !h.Config.Caching.Enabled { + w.Header().Set("Cache-Control", "no-cache, no-store, must-revalidate") + } else { + // Client-side Caching Header (24h) + w.Header().Set("Cache-Control", "public, max-age=86400") + } json.NewEncoder(w).Encode(meta) } @@ -481,6 +490,16 @@ func (h *Handler) scrapeOpenGraph(targetURL, userAgent string) (map[string]strin metadata["image"] = content } else if property == "og:site_name" { metadata["provider_name"] = content + } else if property == "og:video" || property == "og:video:url" { + metadata["video"] = content + } else if property == "og:video:secure_url" { + metadata["video_secure_url"] = content + } else if property == "og:video:width" { + metadata["video_width"] = content + } else if property == "og:video:height" { + metadata["video_height"] = content + } else if property == "og:type" { + metadata["og_type"] = content } else if name == "twitter:image" { metadata["twitter_image"] = content } else if name == "twitter:title" { diff --git a/internal/handler/preview_reddit.go b/internal/handler/preview_reddit.go index 32008d6..42a6230 100644 --- a/internal/handler/preview_reddit.go +++ b/internal/handler/preview_reddit.go @@ -3,48 +3,98 @@ package handler import ( "encoding/json" "fmt" + "io" "net/http" "net/url" + "regexp" + "strconv" + "strings" ) -// GetRedditPreview implements the hybrid Scrape + OEmbed approach +// GetRedditPreview implements the hybrid Scrape + OEmbed approach. +// For video posts, it builds a proper video embed using Reddit's media +// embed player instead of the OEmbed blockquote. func (h *Handler) GetRedditPreview(targetURL string) (map[string]string, error) { + // 0. Resolve share link redirects (/s/ URLs redirect to canonical post URL) + resolvedURL := targetURL + if strings.Contains(targetURL, "/s/") { + if resolved, err := h.resolveRedirect(targetURL); err == nil && resolved != "" { + resolvedURL = resolved + } + } + // 1. Scrape with Slackbot UA for rich metadata (Title, Image, Description, Icon) slackbotUA := "Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)" - meta, err := h.scrapeOpenGraph(targetURL, slackbotUA) + meta, err := h.scrapeOpenGraph(resolvedURL, slackbotUA) if err != nil { - // If scrape fails, we might still try OEmbed, but usually if scrape fails, OEmbed might also fail/be lesser. - // Let's rely on fallback. But for now, let's proceed to OEmbed only if scrape worked partially? - // Or if scrape failed completely, just try OEmbed as last resort. - // For simplicity, let's initialize map if nil if meta == nil { meta = make(map[string]string) } } - // 2. Fetch OEmbed for the Embed HTML (Video Player) - // We manually fetch from Reddit's OEmbed endpoint to bypass the provider check in global tryOEmbed. - oembedEndpoint := "https://www.reddit.com/oembed" - reqURL := fmt.Sprintf("%s?url=%s&format=json", oembedEndpoint, url.QueryEscape(targetURL)) + // 2. Check for video post via OG tags first, then Reddit JSON API + postID := extractRedditPostID(resolvedURL) + if postID != "" { + // Enhance metadata with uncropped image from Reddit JSON API + // This is critical for vertical videos where og:image is cropped to 16:9 + if imgURL, w, h, err := h.fetchRedditJSONDetails(postID); err == nil && imgURL != "" { + meta["image"] = imgURL + meta["embed_width"] = fmt.Sprintf("%d", w) + meta["embed_height"] = fmt.Sprintf("%d", h) + } - oembedMeta, err2 := h.manualFetchOEmbed(reqURL) - if err2 == nil { - // Merge OEmbed HTML into Scraped Meta - if html, ok := oembedMeta["html"]; ok { - meta["embed_html"] = html + isVideo := isRedditVideoPost(meta) + var videoInfo *redditVideoInfo + if !isVideo { + videoInfo = h.getRedditVideoInfo(postID) + isVideo = videoInfo != nil } - // If Scrape failed to get type, use OEmbed type - if _, ok := meta["type"]; !ok { - if t, ok := oembedMeta["type"]; ok { - meta["type"] = t + if isVideo { + embedURL := fmt.Sprintf("https://www.redditmedia.com/mediaembed/%s", postID) + meta["embed_html"] = fmt.Sprintf( + ``, + embedURL, + ) + meta["type"] = "video" + // Note: fetchRedditJSONDetails already sets embed_width/height if available, + // but we keep the fallback to videoInfo if needed (though JSON is preferred). + if _, ok := meta["embed_width"]; !ok { + if videoInfo != nil && videoInfo.Width > 0 && videoInfo.Height > 0 { + meta["embed_width"] = fmt.Sprintf("%d", videoInfo.Width) + meta["embed_height"] = fmt.Sprintf("%d", videoInfo.Height) + } } } - // Force type to video/rich if we have html - if meta["embed_html"] != "" { - meta["type"] = "rich" + } + + // 3. If no video embed, fall back to OEmbed for blockquote embed + if meta["embed_html"] == "" { + oembedEndpoint := "https://www.reddit.com/oembed" + reqURL := fmt.Sprintf("%s?url=%s&format=json", oembedEndpoint, url.QueryEscape(resolvedURL)) + + oembedMeta, err2 := h.manualFetchOEmbed(reqURL) + if err2 == nil { + if html, ok := oembedMeta["html"]; ok { + meta["embed_html"] = html + } + if _, ok := meta["type"]; !ok { + if t, ok := oembedMeta["type"]; ok { + meta["type"] = t + } + } + if meta["embed_html"] != "" { + meta["type"] = "rich" + } } } + // Clean up internal-only OG keys before responding + delete(meta, "video") + delete(meta, "video_secure_url") + delete(meta, "video_width") + delete(meta, "video_height") + delete(meta, "og_type") + // Ensure provider_name is always set for Reddit URLs if meta["provider_name"] == "" { meta["provider_name"] = "Reddit" @@ -53,6 +103,108 @@ func (h *Handler) GetRedditPreview(targetURL string) (map[string]string, error) return meta, nil } +// resolveRedirect follows HTTP redirects and returns the final URL. +func (h *Handler) resolveRedirect(targetURL string) (string, error) { + req, err := http.NewRequest("HEAD", targetURL, nil) + if err != nil { + return "", err + } + req.Header.Set("User-Agent", "Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return "", err + } + defer resp.Body.Close() + + return resp.Request.URL.String(), nil +} + +// isRedditVideoPost checks OG metadata for indicators that the post contains video. +func isRedditVideoPost(meta map[string]string) bool { + if v := meta["video"]; v != "" { + return true + } + if v := meta["video_secure_url"]; v != "" { + return true + } + if t := meta["og_type"]; strings.Contains(t, "video") { + return true + } + return false +} + +type redditVideoInfo struct { + Width int // embed player's data-video-width (NOT the raw video dimensions) + Height int // embed player's data-video-height +} + +var ( + dataVideoWidthRe = regexp.MustCompile(`data-video-width="(\d+)"`) + dataVideoHeightRe = regexp.MustCompile(`data-video-height="(\d+)"`) +) + +// getRedditVideoInfo fetches the redditmedia embed page to detect video posts +// and extract the embed player's actual display dimensions. +// Returns nil if the post has no video embed. +func (h *Handler) getRedditVideoInfo(postID string) *redditVideoInfo { + embedURL := fmt.Sprintf("https://www.redditmedia.com/mediaembed/%s", postID) + + req, err := http.NewRequest("GET", embedURL, nil) + if err != nil { + return nil + } + req.Header.Set("User-Agent", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return nil + } + defer resp.Body.Close() + + if resp.StatusCode >= 400 { + return nil + } + + body, err := io.ReadAll(resp.Body) + if err != nil { + return nil + } + html := string(body) + + // Only treat as video if the embed page contains a video player + if !strings.Contains(html, "video-player") { + return nil + } + + info := &redditVideoInfo{} + if m := dataVideoWidthRe.FindStringSubmatch(html); len(m) >= 2 { + info.Width, _ = strconv.Atoi(m[1]) + } + if m := dataVideoHeightRe.FindStringSubmatch(html); len(m) >= 2 { + info.Height, _ = strconv.Atoi(m[1]) + } + return info +} + +var redditPostIDRe = regexp.MustCompile(`/comments/([a-z0-9]+)`) + +// extractRedditPostID extracts the post ID from a canonical Reddit URL. +// URL format: https://www.reddit.com/r/{subreddit}/comments/{post_id}/{slug}/ +func extractRedditPostID(rawURL string) string { + matches := redditPostIDRe.FindStringSubmatch(rawURL) + if len(matches) >= 2 { + return matches[1] + } + return "" +} + // manualFetchOEmbed fetches OEmbed data from a fully constructed URL func (h *Handler) manualFetchOEmbed(reqURL string) (map[string]string, error) { req, err := http.NewRequest("GET", reqURL, nil) @@ -91,3 +243,93 @@ func (h *Handler) manualFetchOEmbed(reqURL string) (map[string]string, error) { return m, nil } + +// fetchRedditJSONDetails fetches the uncropped image and dimensions from the Reddit JSON API. +// This is critical because the standard og:image is often cropped to 16:9, breaking vertical video embeds. +func (h *Handler) fetchRedditJSONDetails(postID string) (imageURL string, width, height int, err error) { + fmt.Printf("[Debug] fetchRedditJSONDetails called for %s\n", postID) + jsonURL := fmt.Sprintf("https://www.reddit.com/comments/%s.json", postID) + req, err := http.NewRequest("GET", jsonURL, nil) + if err != nil { + return "", 0, 0, err + } + // Use a standard browser UA to avoid bot detection/rate limiting + req.Header.Set("User-Agent", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return "", 0, 0, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return "", 0, 0, fmt.Errorf("reddit json status: %d", resp.StatusCode) + } + + // Define a minimal struct to parse the deep JSON structure + var response []struct { + Data struct { + Children []struct { + Data struct { + Preview struct { + Images []struct { + Source struct { + URL string `json:"url"` + Width int `json:"width"` + Height int `json:"height"` + } `json:"source"` + } `json:"images"` + } `json:"preview"` + Media struct { + RedditVideo struct { + Width int `json:"width"` + Height int `json:"height"` + DashURL string `json:"dash_url"` + HLSURL string `json:"hls_url"` + FallbackURL string `json:"fallback_url"` + } `json:"reddit_video"` + } `json:"media"` + } `json:"data"` + } `json:"children"` + } `json:"data"` + } + + if err := json.NewDecoder(resp.Body).Decode(&response); err != nil { + return "", 0, 0, err + } + + if len(response) > 0 && len(response[0].Data.Children) > 0 { + post := response[0].Data.Children[0].Data + + // Priority 1: Check deep media info (most accurate for videos) + if post.Media.RedditVideo.Width > 0 && post.Media.RedditVideo.Height > 0 { + // Use the fallback URL (mp4) or HLS if needed + url := post.Media.RedditVideo.FallbackURL + if url == "" { + url = post.Media.RedditVideo.HLSURL + } + if url == "" { + url = post.Media.RedditVideo.DashURL + } + if url == "" { + // We need *some* string for the caller to accept the dimensions + // Use the post ID as a placeholder if nothing else + url = fmt.Sprintf("https://v.redd.it/%s", postID) + } + return url, post.Media.RedditVideo.Width, post.Media.RedditVideo.Height, nil + } + + // Priority 2: Check preview images + if len(post.Preview.Images) > 0 { + source := post.Preview.Images[0].Source + // Reddit JSON URLs often contain & encoding + url := strings.ReplaceAll(source.URL, "&", "&") + return url, source.Width, source.Height, nil + } + } + + return "", 0, 0, fmt.Errorf("no preview image found") +} diff --git a/internal/templates/views/index.html b/internal/templates/views/index.html index adc078b..1a6e0be 100644 --- a/internal/templates/views/index.html +++ b/internal/templates/views/index.html @@ -169,7 +169,8 @@ } // Create preview endpoint URL - var previewUrl = "/api/v1/preview?url=" + encodeURIComponent(url); + // Append timestamp to bypass previously aggressive browser-cached responses + var previewUrl = "/api/v1/preview?url=" + encodeURIComponent(url) + "&t=" + new Date().getTime(); // Load preview asynchronously var xhr = new XMLHttpRequest(); @@ -343,6 +344,8 @@ var embedHtmlAttr = ''; if (data.embed_html) { embedHtmlAttr = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; + if (data.embed_width) embedHtmlAttr += ' data-embed-width="' + data.embed_width + '"'; + if (data.embed_height) embedHtmlAttr += ' data-embed-height="' + data.embed_height + '"'; } // Determine aspect ratio @@ -368,8 +371,10 @@ ""; } else if (!imageUrl && data.embed_html && (data.type === "video" || data.type === "rich")) { // Embed exists but no thumbnail image — create container for auto-expand - var embedHtmlAttr = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; - previewHTML += '
'; + var embedHtmlAttr2 = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; + if (data.embed_width) embedHtmlAttr2 += ' data-embed-width="' + data.embed_width + '"'; + if (data.embed_height) embedHtmlAttr2 += ' data-embed-height="' + data.embed_height + '"'; + previewHTML += '
'; } // Text Content @@ -420,6 +425,91 @@ if (embedHtmlEncoded) { var embedHtml = decodeURIComponent(embedHtmlEncoded); + // Reddit video embeds (redditmedia.com/mediaembed/) use src directly + // to avoid iframe-in-iframe and ensure the video player works correctly + var redditVideoMatch = embedHtml.match(/src="(https:\/\/www\.redditmedia\.com\/mediaembed\/[^"]+)"/); + if (redditVideoMatch) { + var iframe = document.createElement('iframe'); + iframe.sandbox = 'allow-scripts allow-same-origin allow-popups allow-presentation'; + iframe.src = redditVideoMatch[1]; + iframe.style.width = '100%'; + iframe.style.border = 'none'; + iframe.style.display = 'block'; + iframe.allowFullscreen = true; + container.setAttribute('data-embed-type', 'reddit'); + + // Size iframe to match the embed player's aspect ratio. + // Dimensions come from the embed page's data-video-width/height + // (the player's display ratio, not the raw video dimensions). + // Controls overlay on the video so no extra padding is needed. + var ew = parseInt(container.getAttribute('data-embed-width')); + var eh = parseInt(container.getAttribute('data-embed-height')); + console.log("[Debug] replaceWithVideo: ew=" + ew + ", eh=" + eh + ", html=" + container.outerHTML); + window.__redditDebug = window.__redditDebug || []; + window.__redditDebug.push({ew: ew, eh: eh, html: container.outerHTML}); + + // Capture image dimensions before clearing container + var img = container.querySelector('img'); + var naturalRatio = (img && img.naturalWidth) ? (img.naturalHeight / img.naturalWidth) : 0; + + // Essential for CSS to allow height > 600px (see screen.css .og-image[data-embed-type="reddit"]) + container.setAttribute('data-embed-type', 'reddit'); + + // If image dimensions aren't available yet, verify them via a new Image object + if (!naturalRatio && img && img.src) { + var tester = new Image(); + tester.onload = function() { + if (this.naturalWidth > 0) { + naturalRatio = this.naturalHeight / this.naturalWidth; + iframe.setAttribute('data-debug-loaded', 'true'); + iframe.setAttribute('data-debug-ratio', naturalRatio); + iframe.setAttribute('data-debug-img-src', img.src); + sizeRedditEmbed(); // Recalculate with new ratio + } + }; + tester.src = img.src; + } + + function sizeRedditEmbed() { + var cw = container.offsetWidth || 640; + iframe.setAttribute('data-debug-ew', ew); + iframe.setAttribute('data-debug-eh', eh); + + // Priority 1: Use explicit embed dimensions if they exist + if (ew && eh) { + var calculatedHeight = Math.round(cw * eh / ew); + // Reddit's mediaembed player doesn't fill portrait iframes properly (it letterboxes). + // Clamp the aspect ratio to a maximum of 1:1 (square) to avoid huge whitespace below portrait videos. + if (calculatedHeight > cw) { + iframe.style.height = cw + 'px'; + } else { + iframe.style.height = calculatedHeight + 'px'; + } + } + // Priority 2: Use natural image aspect ratio if valid + else if (naturalRatio) { + iframe.style.height = Math.round(cw * naturalRatio) + 'px'; + } + // Priority 3: Default to 16:9 + else { + iframe.style.height = Math.round(cw * 9 / 16) + 'px'; + } + iframe.setAttribute('data-debug-final-height', iframe.style.height); + } + sizeRedditEmbed(); + + if (window.ResizeObserver) { + new ResizeObserver(sizeRedditEmbed).observe(container); + } + + container.innerHTML = ''; + container.appendChild(iframe); + container.onclick = null; + container.classList.remove('is-video'); + container.style.cursor = 'default'; + return; + } + // Security: Use sandboxed iframe to isolate external embed content // This prevents embedded content from accessing the parent page var iframe = document.createElement('iframe'); -- 2.51.2