diff --git a/internal/handler/preview.go b/internal/handler/preview.go index 44b5d80..5de35d3 100644 --- a/internal/handler/preview.go +++ b/internal/handler/preview.go @@ -10,8 +10,9 @@ import ( "strings" "time" - "golang.org/x/net/html" "tumble/internal/data" + + "golang.org/x/net/html" ) const ( @@ -121,7 +122,11 @@ func (h *Handler) TryServeCachedOGPreview(w http.ResponseWriter, r *http.Request // Cache hit - serve response w.Header().Set("Content-Type", "application/json") - w.Header().Set("Cache-Control", "public, max-age=86400") + if !h.Config.Caching.Enabled { + w.Header().Set("Cache-Control", "no-cache, no-store, must-revalidate") + } else { + w.Header().Set("Cache-Control", "public, max-age=86400") + } json.NewEncoder(w).Encode(meta) return true } @@ -284,8 +289,12 @@ func (h *Handler) cacheAndRespond(w http.ResponseWriter, r *http.Request, urlPar h.Store.InsertLinkPreview(r.Context(), urlParam, data) } } - // Client-side Caching Header (24h) - w.Header().Set("Cache-Control", "public, max-age=86400") + if !h.Config.Caching.Enabled { + w.Header().Set("Cache-Control", "no-cache, no-store, must-revalidate") + } else { + // Client-side Caching Header (24h) + w.Header().Set("Cache-Control", "public, max-age=86400") + } json.NewEncoder(w).Encode(meta) } @@ -481,6 +490,16 @@ func (h *Handler) scrapeOpenGraph(targetURL, userAgent string) (map[string]strin metadata["image"] = content } else if property == "og:site_name" { metadata["provider_name"] = content + } else if property == "og:video" || property == "og:video:url" { + metadata["video"] = content + } else if property == "og:video:secure_url" { + metadata["video_secure_url"] = content + } else if property == "og:video:width" { + metadata["video_width"] = content + } else if property == "og:video:height" { + metadata["video_height"] = content + } else if property == "og:type" { + metadata["og_type"] = content } else if name == "twitter:image" { metadata["twitter_image"] = content } else if name == "twitter:title" { diff --git a/internal/handler/preview_reddit.go b/internal/handler/preview_reddit.go index 32008d6..42a6230 100644 --- a/internal/handler/preview_reddit.go +++ b/internal/handler/preview_reddit.go @@ -3,48 +3,98 @@ package handler import ( "encoding/json" "fmt" + "io" "net/http" "net/url" + "regexp" + "strconv" + "strings" ) -// GetRedditPreview implements the hybrid Scrape + OEmbed approach +// GetRedditPreview implements the hybrid Scrape + OEmbed approach. +// For video posts, it builds a proper video embed using Reddit's media +// embed player instead of the OEmbed blockquote. func (h *Handler) GetRedditPreview(targetURL string) (map[string]string, error) { + // 0. Resolve share link redirects (/s/ URLs redirect to canonical post URL) + resolvedURL := targetURL + if strings.Contains(targetURL, "/s/") { + if resolved, err := h.resolveRedirect(targetURL); err == nil && resolved != "" { + resolvedURL = resolved + } + } + // 1. Scrape with Slackbot UA for rich metadata (Title, Image, Description, Icon) slackbotUA := "Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)" - meta, err := h.scrapeOpenGraph(targetURL, slackbotUA) + meta, err := h.scrapeOpenGraph(resolvedURL, slackbotUA) if err != nil { - // If scrape fails, we might still try OEmbed, but usually if scrape fails, OEmbed might also fail/be lesser. - // Let's rely on fallback. But for now, let's proceed to OEmbed only if scrape worked partially? - // Or if scrape failed completely, just try OEmbed as last resort. - // For simplicity, let's initialize map if nil if meta == nil { meta = make(map[string]string) } } - // 2. Fetch OEmbed for the Embed HTML (Video Player) - // We manually fetch from Reddit's OEmbed endpoint to bypass the provider check in global tryOEmbed. - oembedEndpoint := "https://www.reddit.com/oembed" - reqURL := fmt.Sprintf("%s?url=%s&format=json", oembedEndpoint, url.QueryEscape(targetURL)) + // 2. Check for video post via OG tags first, then Reddit JSON API + postID := extractRedditPostID(resolvedURL) + if postID != "" { + // Enhance metadata with uncropped image from Reddit JSON API + // This is critical for vertical videos where og:image is cropped to 16:9 + if imgURL, w, h, err := h.fetchRedditJSONDetails(postID); err == nil && imgURL != "" { + meta["image"] = imgURL + meta["embed_width"] = fmt.Sprintf("%d", w) + meta["embed_height"] = fmt.Sprintf("%d", h) + } - oembedMeta, err2 := h.manualFetchOEmbed(reqURL) - if err2 == nil { - // Merge OEmbed HTML into Scraped Meta - if html, ok := oembedMeta["html"]; ok { - meta["embed_html"] = html + isVideo := isRedditVideoPost(meta) + var videoInfo *redditVideoInfo + if !isVideo { + videoInfo = h.getRedditVideoInfo(postID) + isVideo = videoInfo != nil } - // If Scrape failed to get type, use OEmbed type - if _, ok := meta["type"]; !ok { - if t, ok := oembedMeta["type"]; ok { - meta["type"] = t + if isVideo { + embedURL := fmt.Sprintf("https://www.redditmedia.com/mediaembed/%s", postID) + meta["embed_html"] = fmt.Sprintf( + ``, + embedURL, + ) + meta["type"] = "video" + // Note: fetchRedditJSONDetails already sets embed_width/height if available, + // but we keep the fallback to videoInfo if needed (though JSON is preferred). + if _, ok := meta["embed_width"]; !ok { + if videoInfo != nil && videoInfo.Width > 0 && videoInfo.Height > 0 { + meta["embed_width"] = fmt.Sprintf("%d", videoInfo.Width) + meta["embed_height"] = fmt.Sprintf("%d", videoInfo.Height) + } } } - // Force type to video/rich if we have html - if meta["embed_html"] != "" { - meta["type"] = "rich" + } + + // 3. If no video embed, fall back to OEmbed for blockquote embed + if meta["embed_html"] == "" { + oembedEndpoint := "https://www.reddit.com/oembed" + reqURL := fmt.Sprintf("%s?url=%s&format=json", oembedEndpoint, url.QueryEscape(resolvedURL)) + + oembedMeta, err2 := h.manualFetchOEmbed(reqURL) + if err2 == nil { + if html, ok := oembedMeta["html"]; ok { + meta["embed_html"] = html + } + if _, ok := meta["type"]; !ok { + if t, ok := oembedMeta["type"]; ok { + meta["type"] = t + } + } + if meta["embed_html"] != "" { + meta["type"] = "rich" + } } } + // Clean up internal-only OG keys before responding + delete(meta, "video") + delete(meta, "video_secure_url") + delete(meta, "video_width") + delete(meta, "video_height") + delete(meta, "og_type") + // Ensure provider_name is always set for Reddit URLs if meta["provider_name"] == "" { meta["provider_name"] = "Reddit" @@ -53,6 +103,108 @@ func (h *Handler) GetRedditPreview(targetURL string) (map[string]string, error) return meta, nil } +// resolveRedirect follows HTTP redirects and returns the final URL. +func (h *Handler) resolveRedirect(targetURL string) (string, error) { + req, err := http.NewRequest("HEAD", targetURL, nil) + if err != nil { + return "", err + } + req.Header.Set("User-Agent", "Slackbot-LinkExpanding 1.0 (+https://api.slack.com/robots)") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return "", err + } + defer resp.Body.Close() + + return resp.Request.URL.String(), nil +} + +// isRedditVideoPost checks OG metadata for indicators that the post contains video. +func isRedditVideoPost(meta map[string]string) bool { + if v := meta["video"]; v != "" { + return true + } + if v := meta["video_secure_url"]; v != "" { + return true + } + if t := meta["og_type"]; strings.Contains(t, "video") { + return true + } + return false +} + +type redditVideoInfo struct { + Width int // embed player's data-video-width (NOT the raw video dimensions) + Height int // embed player's data-video-height +} + +var ( + dataVideoWidthRe = regexp.MustCompile(`data-video-width="(\d+)"`) + dataVideoHeightRe = regexp.MustCompile(`data-video-height="(\d+)"`) +) + +// getRedditVideoInfo fetches the redditmedia embed page to detect video posts +// and extract the embed player's actual display dimensions. +// Returns nil if the post has no video embed. +func (h *Handler) getRedditVideoInfo(postID string) *redditVideoInfo { + embedURL := fmt.Sprintf("https://www.redditmedia.com/mediaembed/%s", postID) + + req, err := http.NewRequest("GET", embedURL, nil) + if err != nil { + return nil + } + req.Header.Set("User-Agent", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return nil + } + defer resp.Body.Close() + + if resp.StatusCode >= 400 { + return nil + } + + body, err := io.ReadAll(resp.Body) + if err != nil { + return nil + } + html := string(body) + + // Only treat as video if the embed page contains a video player + if !strings.Contains(html, "video-player") { + return nil + } + + info := &redditVideoInfo{} + if m := dataVideoWidthRe.FindStringSubmatch(html); len(m) >= 2 { + info.Width, _ = strconv.Atoi(m[1]) + } + if m := dataVideoHeightRe.FindStringSubmatch(html); len(m) >= 2 { + info.Height, _ = strconv.Atoi(m[1]) + } + return info +} + +var redditPostIDRe = regexp.MustCompile(`/comments/([a-z0-9]+)`) + +// extractRedditPostID extracts the post ID from a canonical Reddit URL. +// URL format: https://www.reddit.com/r/{subreddit}/comments/{post_id}/{slug}/ +func extractRedditPostID(rawURL string) string { + matches := redditPostIDRe.FindStringSubmatch(rawURL) + if len(matches) >= 2 { + return matches[1] + } + return "" +} + // manualFetchOEmbed fetches OEmbed data from a fully constructed URL func (h *Handler) manualFetchOEmbed(reqURL string) (map[string]string, error) { req, err := http.NewRequest("GET", reqURL, nil) @@ -91,3 +243,93 @@ func (h *Handler) manualFetchOEmbed(reqURL string) (map[string]string, error) { return m, nil } + +// fetchRedditJSONDetails fetches the uncropped image and dimensions from the Reddit JSON API. +// This is critical because the standard og:image is often cropped to 16:9, breaking vertical video embeds. +func (h *Handler) fetchRedditJSONDetails(postID string) (imageURL string, width, height int, err error) { + fmt.Printf("[Debug] fetchRedditJSONDetails called for %s\n", postID) + jsonURL := fmt.Sprintf("https://www.reddit.com/comments/%s.json", postID) + req, err := http.NewRequest("GET", jsonURL, nil) + if err != nil { + return "", 0, 0, err + } + // Use a standard browser UA to avoid bot detection/rate limiting + req.Header.Set("User-Agent", "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") + + client := &http.Client{ + Timeout: h.Config.RequestTimeout, + } + resp, err := client.Do(req) + if err != nil { + return "", 0, 0, err + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return "", 0, 0, fmt.Errorf("reddit json status: %d", resp.StatusCode) + } + + // Define a minimal struct to parse the deep JSON structure + var response []struct { + Data struct { + Children []struct { + Data struct { + Preview struct { + Images []struct { + Source struct { + URL string `json:"url"` + Width int `json:"width"` + Height int `json:"height"` + } `json:"source"` + } `json:"images"` + } `json:"preview"` + Media struct { + RedditVideo struct { + Width int `json:"width"` + Height int `json:"height"` + DashURL string `json:"dash_url"` + HLSURL string `json:"hls_url"` + FallbackURL string `json:"fallback_url"` + } `json:"reddit_video"` + } `json:"media"` + } `json:"data"` + } `json:"children"` + } `json:"data"` + } + + if err := json.NewDecoder(resp.Body).Decode(&response); err != nil { + return "", 0, 0, err + } + + if len(response) > 0 && len(response[0].Data.Children) > 0 { + post := response[0].Data.Children[0].Data + + // Priority 1: Check deep media info (most accurate for videos) + if post.Media.RedditVideo.Width > 0 && post.Media.RedditVideo.Height > 0 { + // Use the fallback URL (mp4) or HLS if needed + url := post.Media.RedditVideo.FallbackURL + if url == "" { + url = post.Media.RedditVideo.HLSURL + } + if url == "" { + url = post.Media.RedditVideo.DashURL + } + if url == "" { + // We need *some* string for the caller to accept the dimensions + // Use the post ID as a placeholder if nothing else + url = fmt.Sprintf("https://v.redd.it/%s", postID) + } + return url, post.Media.RedditVideo.Width, post.Media.RedditVideo.Height, nil + } + + // Priority 2: Check preview images + if len(post.Preview.Images) > 0 { + source := post.Preview.Images[0].Source + // Reddit JSON URLs often contain & encoding + url := strings.ReplaceAll(source.URL, "&", "&") + return url, source.Width, source.Height, nil + } + } + + return "", 0, 0, fmt.Errorf("no preview image found") +} diff --git a/internal/templates/views/index.html b/internal/templates/views/index.html index adc078b..1a6e0be 100644 --- a/internal/templates/views/index.html +++ b/internal/templates/views/index.html @@ -169,7 +169,8 @@ } // Create preview endpoint URL - var previewUrl = "/api/v1/preview?url=" + encodeURIComponent(url); + // Append timestamp to bypass previously aggressive browser-cached responses + var previewUrl = "/api/v1/preview?url=" + encodeURIComponent(url) + "&t=" + new Date().getTime(); // Load preview asynchronously var xhr = new XMLHttpRequest(); @@ -343,6 +344,8 @@ var embedHtmlAttr = ''; if (data.embed_html) { embedHtmlAttr = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; + if (data.embed_width) embedHtmlAttr += ' data-embed-width="' + data.embed_width + '"'; + if (data.embed_height) embedHtmlAttr += ' data-embed-height="' + data.embed_height + '"'; } // Determine aspect ratio @@ -368,8 +371,10 @@ ""; } else if (!imageUrl && data.embed_html && (data.type === "video" || data.type === "rich")) { // Embed exists but no thumbnail image — create container for auto-expand - var embedHtmlAttr = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; - previewHTML += '
'; + var embedHtmlAttr2 = ' data-embed-html="' + encodeURIComponent(data.embed_html) + '"'; + if (data.embed_width) embedHtmlAttr2 += ' data-embed-width="' + data.embed_width + '"'; + if (data.embed_height) embedHtmlAttr2 += ' data-embed-height="' + data.embed_height + '"'; + previewHTML += ''; } // Text Content @@ -420,6 +425,91 @@ if (embedHtmlEncoded) { var embedHtml = decodeURIComponent(embedHtmlEncoded); + // Reddit video embeds (redditmedia.com/mediaembed/) use src directly + // to avoid iframe-in-iframe and ensure the video player works correctly + var redditVideoMatch = embedHtml.match(/src="(https:\/\/www\.redditmedia\.com\/mediaembed\/[^"]+)"/); + if (redditVideoMatch) { + var iframe = document.createElement('iframe'); + iframe.sandbox = 'allow-scripts allow-same-origin allow-popups allow-presentation'; + iframe.src = redditVideoMatch[1]; + iframe.style.width = '100%'; + iframe.style.border = 'none'; + iframe.style.display = 'block'; + iframe.allowFullscreen = true; + container.setAttribute('data-embed-type', 'reddit'); + + // Size iframe to match the embed player's aspect ratio. + // Dimensions come from the embed page's data-video-width/height + // (the player's display ratio, not the raw video dimensions). + // Controls overlay on the video so no extra padding is needed. + var ew = parseInt(container.getAttribute('data-embed-width')); + var eh = parseInt(container.getAttribute('data-embed-height')); + console.log("[Debug] replaceWithVideo: ew=" + ew + ", eh=" + eh + ", html=" + container.outerHTML); + window.__redditDebug = window.__redditDebug || []; + window.__redditDebug.push({ew: ew, eh: eh, html: container.outerHTML}); + + // Capture image dimensions before clearing container + var img = container.querySelector('img'); + var naturalRatio = (img && img.naturalWidth) ? (img.naturalHeight / img.naturalWidth) : 0; + + // Essential for CSS to allow height > 600px (see screen.css .og-image[data-embed-type="reddit"]) + container.setAttribute('data-embed-type', 'reddit'); + + // If image dimensions aren't available yet, verify them via a new Image object + if (!naturalRatio && img && img.src) { + var tester = new Image(); + tester.onload = function() { + if (this.naturalWidth > 0) { + naturalRatio = this.naturalHeight / this.naturalWidth; + iframe.setAttribute('data-debug-loaded', 'true'); + iframe.setAttribute('data-debug-ratio', naturalRatio); + iframe.setAttribute('data-debug-img-src', img.src); + sizeRedditEmbed(); // Recalculate with new ratio + } + }; + tester.src = img.src; + } + + function sizeRedditEmbed() { + var cw = container.offsetWidth || 640; + iframe.setAttribute('data-debug-ew', ew); + iframe.setAttribute('data-debug-eh', eh); + + // Priority 1: Use explicit embed dimensions if they exist + if (ew && eh) { + var calculatedHeight = Math.round(cw * eh / ew); + // Reddit's mediaembed player doesn't fill portrait iframes properly (it letterboxes). + // Clamp the aspect ratio to a maximum of 1:1 (square) to avoid huge whitespace below portrait videos. + if (calculatedHeight > cw) { + iframe.style.height = cw + 'px'; + } else { + iframe.style.height = calculatedHeight + 'px'; + } + } + // Priority 2: Use natural image aspect ratio if valid + else if (naturalRatio) { + iframe.style.height = Math.round(cw * naturalRatio) + 'px'; + } + // Priority 3: Default to 16:9 + else { + iframe.style.height = Math.round(cw * 9 / 16) + 'px'; + } + iframe.setAttribute('data-debug-final-height', iframe.style.height); + } + sizeRedditEmbed(); + + if (window.ResizeObserver) { + new ResizeObserver(sizeRedditEmbed).observe(container); + } + + container.innerHTML = ''; + container.appendChild(iframe); + container.onclick = null; + container.classList.remove('is-video'); + container.style.cursor = 'default'; + return; + } + // Security: Use sandboxed iframe to isolate external embed content // This prevents embedded content from accessing the parent page var iframe = document.createElement('iframe');