From a9d5390e2673662abc6df81a5049e8bcfdaeaa7f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andri=20=C3=93skarsson?= Date: Thu, 26 Feb 2026 09:28:10 +0100 Subject: [PATCH] Strip script, iframe, object, and embed elements with blocked resource URLs During HTML rewriting, elements that load external resources are now checked against URL blocking rules. If the resolved URL is blocked, the element is stripped from the DOM (replaced with an HTML comment). This provides defense-in-depth alongside network-level blocking: the browser never sees the element so it doesn't attempt the load. srcBlockableTags maps each tag to its URL attribute (src for most, data for object). Void elements (embed) skip the closing-tag scan. Relative and protocol-relative URLs are resolved against page origin. --- DECISIONS.md | 1 + elemhide_inject.go | 117 ++++++++- examples/dk.rules | 2 + proxy_test.go | 631 +++++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 742 insertions(+), 9 deletions(-) diff --git a/DECISIONS.md b/DECISIONS.md index 3db15bf..67387d3 100644 --- a/DECISIONS.md +++ b/DECISIONS.md @@ -16,3 +16,4 @@ - 2026-02-26 m+git@andri.dk — Added `golang.org/x/net` (zero transitive runtime deps, Go team maintained) for HTML tokenization in element replacement. Selectors are classified at cache time into simple (single-element matchable) and complex (needs CSS fallback). - 2026-02-26 m+git@andri.dk — Blocked CONNECT requests return 403 Forbidden instead of 204 No Content. Browsers hang on 204 because they expect a tunnel; 403 makes them fail fast. - 2026-02-26 m+git@andri.dk — Removed Playwright from the project. Playwright's screenshot and snapshot commands wait for `document.fonts.ready`, which never resolves when the MITM proxy blocks ad-network domains that serve fonts. curl is a better fit for proxy testing. +- 2026-02-26 m+git@andri.dk — Elements that load external resources (script, iframe, object, embed) are stripped during HTML rewriting when their resource URL resolves to a blocked address. Defense-in-depth alongside network-level blocking: stripping the element from the DOM prevents the browser from attempting the load. `srcBlockableTags` maps tag name to URL attribute (`src` for most, `data` for object). Void elements (embed) skip the `skipUntilClose` step. Relative and protocol-relative URLs are resolved against the page origin. Inline scripts (no `src`) are not affected. diff --git a/elemhide_inject.go b/elemhide_inject.go index 8361708..ae9b4c8 100644 --- a/elemhide_inject.go +++ b/elemhide_inject.go @@ -22,9 +22,44 @@ var voidElements = map[string]bool{ "track": true, "wbr": true, } +// srcBlockableTags maps element names to the attribute that carries their +// external resource URL. If the resolved URL is blocked, the element is stripped. +var srcBlockableTags = map[string]string{ + "script": "src", + "iframe": "src", + "object": "data", + "embed": "src", +} + +// srcBlockContext carries the page context needed to resolve relative src +// attributes and check them against URL blocking rules. +type srcBlockContext struct { + scheme string + host string + rules *blocklist.RuleSet +} + +// resolveSrc resolves an element's src attribute to an absolute URL. +// Protocol-relative (//host/path), absolute (/path), and fully qualified +// URLs are all handled. +func (sc srcBlockContext) resolveSrc(src string) string { + if strings.HasPrefix(src, "//") { + return sc.scheme + ":" + src + } + if strings.HasPrefix(src, "http://") || strings.HasPrefix(src, "https://") { + return src + } + if strings.HasPrefix(src, "/") { + return sc.scheme + "://" + sc.host + src + } + return sc.scheme + "://" + sc.host + "/" + src +} + // applyElementHiding checks if the response is HTML and applies element hiding -// if applicable. Matched elements are replaced with placeholder divs; complex -// selectors that can't be matched on a single element fall back to CSS injection. +// and src-based stripping if applicable. Matched elements are replaced with +// placeholder divs; complex selectors that can't be matched on a single element +// fall back to CSS injection. Elements that load external resources (script, +// iframe, object, embed) whose URL resolves to a blocked address are stripped. // Returns the modified body and true, or nil and false if unmodified. // Handles gzip and brotli compressed responses transparently. func (p *proxyHandler) applyElementHiding(resp *http.Response, host string) ([]byte, bool) { @@ -38,7 +73,11 @@ func (p *proxyHandler) applyElementHiding(resp *http.Response, host string) ([]b } eh := p.rules.ElementHidingForDomain(host) - if eh == nil { + + // Nothing to do if there are no element hiding rules and no URL rules + // that could match resource URLs on blockable elements. + hasURLRules := p.rules.HostCount() > 0 || p.rules.RuleCount() > 0 + if eh == nil && !hasURLRules { return nil, false } @@ -65,9 +104,15 @@ func (p *proxyHandler) applyElementHiding(resp *http.Response, host string) ([]b return nil, false } - modified := replaceElements(body, eh.Matchers) + var matchers []blocklist.SelectorMatch + if eh != nil { + matchers = eh.Matchers + } + + sc := srcBlockContext{scheme: "https", host: host, rules: p.rules} + modified := replaceElements(body, matchers, sc) - if eh.FallbackCSS != "" { + if eh != nil && eh.FallbackCSS != "" { styleTag := []byte("") modified = injectStyleTag(modified, styleTag) } @@ -81,8 +126,10 @@ func (p *proxyHandler) applyElementHiding(resp *http.Response, host string) ([]b // replaceElements uses the html tokenizer to walk through the HTML and replace // elements matching any of the simple selectors with placeholder divs. -func replaceElements(src []byte, matchers []blocklist.SelectorMatch) []byte { - if len(matchers) == 0 { +// It also strips elements (script, iframe, object, embed) whose resource URL +// resolves to a blocked address. +func replaceElements(src []byte, matchers []blocklist.SelectorMatch, sc srcBlockContext) []byte { + if len(matchers) == 0 && sc.rules == nil { return src } @@ -108,6 +155,36 @@ func replaceElements(src []byte, matchers []blocklist.SelectorMatch) []byte { tagName := string(tn) tagNameLower := strings.ToLower(tagName) + // For elements with a blockable resource URL (script, iframe, + // object, embed), check if the URL resolves to a blocked + // address before falling through to element hiding. + if urlAttr, blockable := srcBlockableTags[tagNameLower]; blockable && hasAttr && sc.rules != nil { + attrs := collectAttrs(tokenizer, hasAttr) + if urlVal, ok := attrs[urlAttr]; ok && urlVal != "" { + resolved := sc.resolveSrc(urlVal) + ctx := blocklist.MatchContext{PageDomain: sc.host} + if sc.rules.ShouldBlockRequest(resolved, ctx) { + replacement := fmt.Sprintf("", tagNameLower, urlVal) + buf.WriteString(replacement) + if !voidElements[tagNameLower] { + skipUntilClose(tokenizer, tagNameLower) + } + continue + } + } + // Not blocked — check element hiding with pre-collected attrs + if matched, selector := matchesAnyWithAttrs(tagNameLower, attrs, matchers); matched { + replacement := fmt.Sprintf("
", selector) + buf.WriteString(replacement) + if !voidElements[tagNameLower] { + skipUntilClose(tokenizer, tagNameLower) + } + continue + } + buf.Write(tokenizer.Raw()) + continue + } + if matched, selector := matchesAny(tagNameLower, tokenizer, hasAttr, matchers); matched { replacement := fmt.Sprintf("
", selector) buf.WriteString(replacement) @@ -140,15 +217,14 @@ func replaceElements(src []byte, matchers []blocklist.SelectorMatch) []byte { } // matchesAny checks if the current element matches any of the simple selectors. +// Collects attributes lazily from the tokenizer on first need. // Returns true and the matched selector string, or false. func matchesAny(tagName string, tokenizer *html.Tokenizer, hasAttr bool, matchers []blocklist.SelectorMatch) (bool, string) { - // Collect attributes lazily — only if we need them var attrs map[string]string for i := range matchers { sm := &matchers[i] - // Quick tag-name check before collecting attributes if sm.Tag != "" && sm.Tag != tagName { continue } @@ -169,6 +245,29 @@ func matchesAny(tagName string, tokenizer *html.Tokenizer, hasAttr bool, matcher return false, "" } +// matchesAnyWithAttrs checks if the element matches any selector using +// a pre-collected attribute map. Used when attributes were already read +// from the tokenizer (e.g. for script src checking). +func matchesAnyWithAttrs(tagName string, attrs map[string]string, matchers []blocklist.SelectorMatch) (bool, string) { + for i := range matchers { + sm := &matchers[i] + + if sm.Tag != "" && sm.Tag != tagName { + continue + } + + attrFn := func(name string) string { + return attrs[name] + } + + if sm.MatchesAttrs(tagName, attrFn) { + return true, sm.Selector + } + } + + return false, "" +} + // collectAttrs reads all attributes from the tokenizer for the current tag. func collectAttrs(tokenizer *html.Tokenizer, hasAttr bool) map[string]string { attrs := make(map[string]string) diff --git a/examples/dk.rules b/examples/dk.rules index f1734ef..697a2fb 100644 --- a/examples/dk.rules +++ b/examples/dk.rules @@ -3,6 +3,8 @@ ! For testing ublproxy element replacement ! ---- Hostname blocking (ad networks and trackers) ---- +! Scripts with src matching these hostnames are also stripped from HTML +! responses, providing defense-in-depth alongside network-level blocking. ||functions.adnami.io^ ||macro.adnami.io^ diff --git a/proxy_test.go b/proxy_test.go index 2ef996b..0538eed 100644 --- a/proxy_test.go +++ b/proxy_test.go @@ -1088,6 +1088,637 @@ func TestHeadRequestNoInjection(t *testing.T) { } } +// --- Script stripping tests --- + +func TestScriptStrippingBlockedHost(t *testing.T) { + rs := blocklist.NewRuleSet() + rs.AddLine("||ads.tracker.com^") + + upstream := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/html; charset=utf-8") + w.WriteHeader(http.StatusOK) + w.Write([]byte(`` + + `` + + `` + + `

Content

`)) + }) + + env := startTestEnv(t, upstream, rs) + client := env.httpClient(t) + + resp, err := client.Get(env.httpURL + "/page.html") + if err != nil { + t.Fatalf("GET: %v", err) + } + defer resp.Body.Close() + + body, _ := io.ReadAll(resp.Body) + bodyStr := string(body) + + // Blocked script should be stripped with a comment + if !strings.Contains(bodyStr, "") { + t.Errorf("element hiding should still work, got:\n%s", bodyStr) + } + + if !strings.Contains(bodyStr, "Content") { + t.Errorf("page content should be preserved, got:\n%s", bodyStr) + } +} + +func TestScriptStrippingHTTPS(t *testing.T) { + rs := blocklist.NewRuleSet() + rs.AddLine("||ads.tracker.com^") + + upstream := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/html; charset=utf-8") + w.WriteHeader(http.StatusOK) + w.Write([]byte(`` + + `` + + `` + + `

Content

`)) + }) + + env := startTestEnv(t, upstream, rs) + client := env.httpClient(t) + + resp, err := client.Get(env.httpsURL + "/page.html") + if err != nil { + t.Fatalf("GET: %v", err) + } + defer resp.Body.Close() + + body, _ := io.ReadAll(resp.Body) + bodyStr := string(body) + + // Blocked script should be stripped + if !strings.Contains(bodyStr, "