diff --git a/pkg/blocklist/pattern.go b/pkg/blocklist/pattern.go index 26319a3..9d431b4 100644 --- a/pkg/blocklist/pattern.go +++ b/pkg/blocklist/pattern.go @@ -193,7 +193,30 @@ func (r *Rule) matchURL(url string) bool { return matchSegments(r.segments, url, r.anchorEnd) } - // No start anchor: try matching at every position + // No start anchor: find substring match. Use strings.Index to skip to + // positions where the leading literal actually appears. + if len(r.segments) > 0 && r.segments[0].kind == segLiteral { + lit := r.segments[0].literal + rest := url + offset := 0 + for { + idx := strings.Index(rest, lit) + if idx < 0 { + return false + } + pos := offset + idx + if matchSegments(r.segments, url[pos:], r.anchorEnd) { + return true + } + offset = pos + 1 + if offset >= len(url) { + return false + } + rest = url[offset:] + } + } + + // Leading segment is separator or wildcard — fall back to scanning for i := range len(url) { if matchSegments(r.segments, url[i:], r.anchorEnd) { return true @@ -385,12 +408,29 @@ func matchSegmentsAt(segments []segment, si int, text string, ti int, anchorEnd si++ // If wildcard is the last segment, it matches everything if si == len(segments) { - if anchorEnd { - return true - } return true } - // Try matching the rest of the pattern at every position + // If the next segment is a literal, use strings.Index to skip + // to positions where it could match instead of trying every byte + if segments[si].kind == segLiteral { + lit := segments[si].literal + pos := ti + for { + idx := strings.Index(text[pos:], lit) + if idx < 0 { + return false + } + candidate := pos + idx + if matchSegmentsAt(segments, si, text, candidate, anchorEnd) { + return true + } + pos = candidate + 1 + if pos > len(text) { + return false + } + } + } + // Next segment is separator or wildcard — scan all positions for pos := ti; pos <= len(text); pos++ { if matchSegmentsAt(segments, si, text, pos, anchorEnd) { return true diff --git a/pkg/blocklist/pattern_test.go b/pkg/blocklist/pattern_test.go index f2baf9b..c18c457 100644 --- a/pkg/blocklist/pattern_test.go +++ b/pkg/blocklist/pattern_test.go @@ -176,6 +176,46 @@ func TestDomainOption(t *testing.T) { } } +func BenchmarkMatchSubstring(b *testing.B) { + // Non-anchored pattern: must scan the URL for a substring match + rule, _ := blocklist.Compile("/ads/banner*.gif") + url := "http://cdn.example.com/static/resources/images/v3/long/path/that/forces/many/scans/ads/banner123.gif" + b.ResetTimer() + for range b.N { + rule.Match(url) + } +} + +func BenchmarkMatchWildcard(b *testing.B) { + // Pattern with multiple wildcards: forces backtracking + rule, _ := blocklist.Compile("ad*track*pixel*.gif") + url := "http://example.com/ad-network/track-event/pixel-data.gif?v=123" + b.ResetTimer() + for range b.N { + rule.Match(url) + } +} + +func BenchmarkMatchDomainAnchor(b *testing.B) { + // Domain-anchored with path: typical adblock rule + rule, _ := blocklist.Compile("||example.com/ads/*.gif") + url := "http://example.com/ads/banner123.gif" + b.ResetTimer() + for range b.N { + rule.Match(url) + } +} + +func BenchmarkMatchNoMatch(b *testing.B) { + // Non-anchored pattern against a long URL that doesn't match: worst case + rule, _ := blocklist.Compile("zzz-nonexistent-pattern") + url := "http://cdn.example.com/static/resources/images/v3/long/path/that/forces/many/scans/and/never/matches.gif" + b.ResetTimer() + for range b.N { + rule.Match(url) + } +} + func TestOptionsStrippedFromPattern(t *testing.T) { // The $options suffix should not be part of the URL pattern rule, _ := blocklist.Compile("/ads/banner.gif$match-case")