//! result filters — ported from the retired rust backend, with one deliberate narrowing. //! //! the rust api accepted arbitrary regex for `exclude`/`include`. logfire //! showed the only real-world pattern was `^bigbufo_`, which is now a //! server-side default. this port supports a documented subset instead of //! pulling in a regex engine: //! //! - `^` at the start anchors to the beginning of the name //! - `$` at the end anchors to the end //! - `\b` at the start and/or end requires a word boundary (name chars are //! `[a-z0-9]`; `-`, `_`, `'` and the edges are boundaries) — the bufo bot's //! exclude list uses `\brip\b` //! - `|` separates alternatives (inside one comma-separated item) //! - everything else is a literal substring match //! //! any other regex metacharacter is matched literally and logged once per //! request so we can see whether anyone actually needs more. const std = @import("std"); const mem = std.mem; const Allocator = mem.Allocator; const Edge = enum { none, anchor, word_boundary }; pub const Pattern = struct { start: Edge, end: Edge, literal: []const u8, pub fn parse(raw: []const u8) Pattern { var s = mem.trim(u8, raw, " "); var start: Edge = .none; var end: Edge = .none; if (mem.startsWith(u8, s, "^")) { start = .anchor; s = s[1..]; } else if (mem.startsWith(u8, s, "\\b")) { start = .word_boundary; s = s[2..]; } if (mem.endsWith(u8, s, "$")) { end = .anchor; s = s[0 .. s.len - 1]; } else if (mem.endsWith(u8, s, "\\b")) { end = .word_boundary; s = s[0 .. s.len - 2]; } return .{ .start = start, .end = end, .literal = s }; } fn isWordChar(c: u8) bool { return std.ascii.isAlphanumeric(c); } fn edgeOk(edge: Edge, name: []const u8, at_start: bool, idx: usize) bool { return switch (edge) { .none => true, .anchor => if (at_start) idx == 0 else idx == name.len, .word_boundary => if (at_start) idx == 0 or !isWordChar(name[idx - 1]) else idx == name.len or !isWordChar(name[idx]), }; } pub fn matches(self: Pattern, name: []const u8) bool { if (self.literal.len == 0) return false; var from: usize = 0; while (mem.indexOfPos(u8, name, from, self.literal)) |i| { const after = i + self.literal.len; if (edgeOk(self.start, name, true, i) and edgeOk(self.end, name, false, after)) return true; if (self.start == .anchor) return false; from = i + 1; } return false; } /// diagnostic only: does this pattern use regex syntax the subset above /// doesn't implement? matching proceeds literally either way; the caller /// logs so we find out if a real client ever needs more. pub fn hasUnsupportedMeta(raw: []const u8) bool { if (mem.indexOfAny(u8, raw, ".*+?()[]{}") != null) return true; // the only escape we implement is \b; any other backslash is regex var rest = raw; while (mem.indexOfScalar(u8, rest, '\\')) |i| { if (i + 1 >= rest.len or rest[i + 1] != 'b') return true; rest = rest[i + 2 ..]; } return false; } }; /// parse a comma-separated list where each item may carry `|` alternatives /// into a flat pattern list (an item matches if any alternative matches). pub fn parsePatterns(alloc: Allocator, raw: ?[]const u8) ![]Pattern { var out: std.ArrayList(Pattern) = .empty; const text = raw orelse return out.items; var items = mem.splitScalar(u8, text, ','); while (items.next()) |item| { if (Pattern.hasUnsupportedMeta(item)) { std.log.warn("filter pattern uses unsupported regex syntax, matching literally: {s}", .{item}); } var alts = mem.splitScalar(u8, item, '|'); while (alts.next()) |alt| { const p = Pattern.parse(alt); if (p.literal.len > 0) try out.append(alloc, p); } } return out.items; } fn anyMatch(patterns: []const Pattern, name: []const u8) bool { for (patterns) |p| if (p.matches(name)) return true; return false; } /// the family-friendly blocklist, inherited from the rust backend const inappropriate = [_][]const u8{ "bufo-juicy", "good-news-bufo-offers-suppository", "bufo-declines-your-suppository-offer", "tsa-bufo-gropes-you", }; /// the 4x4 big-bufo tile set, denied for every client unless `include`d const tile_deny = Pattern{ .start = .anchor, .end = .none, .literal = "bigbufo_" }; pub const ContentFilter = struct { family_friendly: bool, exclude: []const Pattern, include: []const Pattern, formats: []const []const u8, pub fn init( alloc: Allocator, family_friendly: bool, exclude: ?[]const u8, include: ?[]const u8, formats: ?[]const u8, ) !ContentFilter { return .{ .family_friendly = family_friendly, .exclude = try parsePatterns(alloc, exclude), .include = try parsePatterns(alloc, include), .formats = try parseFormats(alloc, formats), }; } pub fn keeps(self: ContentFilter, name: []const u8, url: []const u8) bool { if (!matchesFormats(url, self.formats)) return false; if (self.family_friendly) { for (inappropriate) |blocked| { if (mem.indexOf(u8, name, blocked) != null) return false; } } if (anyMatch(self.include, name)) return true; if (tile_deny.matches(name)) return false; return !anyMatch(self.exclude, name); } }; fn parseFormats(alloc: Allocator, raw: ?[]const u8) ![]const []const u8 { var out: std.ArrayList([]const u8) = .empty; const text = raw orelse return out.items; var items = mem.splitScalar(u8, text, ','); while (items.next()) |item| { var f = mem.trim(u8, item, " "); f = mem.trimStart(u8, f, "."); if (f.len == 0) continue; const lowered = try alloc.alloc(u8, f.len); for (f, 0..) |c, i| lowered[i] = std.ascii.toLower(c); try out.append(alloc, lowered); } return out.items; } fn matchesFormats(url: []const u8, allowed: []const []const u8) bool { if (allowed.len == 0) return true; const dot = mem.lastIndexOfScalar(u8, url, '.') orelse return false; const ext = url[dot + 1 ..]; for (allowed) |f| { if (std.ascii.eqlIgnoreCase(ext, f)) return true; } return false; } // --- tests --- const t = std.testing; test "tiles denied by default, include restores" { var arena = std.heap.ArenaAllocator.init(t.allocator); defer arena.deinit(); const a = arena.allocator(); const default_filter = try ContentFilter.init(a, true, null, null, null); try t.expect(!default_filter.keeps("bigbufo_2_3", "https://x/bigbufo_2_3.png")); try t.expect(default_filter.keeps("bufo-happy", "https://x/bufo-happy.png")); const opted_in = try ContentFilter.init(a, true, null, "bigbufo", null); try t.expect(opted_in.keeps("bigbufo_2_3", "https://x/bigbufo_2_3.png")); } test "family-friendly blocklist" { var arena = std.heap.ArenaAllocator.init(t.allocator); defer arena.deinit(); const a = arena.allocator(); const strict = try ContentFilter.init(a, true, null, null, null); try t.expect(!strict.keeps("bufo-juicy", "https://x/bufo-juicy.png")); const lax = try ContentFilter.init(a, false, null, null, null); try t.expect(lax.keeps("bufo-juicy", "https://x/bufo-juicy.png")); } test "include overrides exclude" { var arena = std.heap.ArenaAllocator.init(t.allocator); defer arena.deinit(); const a = arena.allocator(); const f = try ContentFilter.init(a, false, "party", "birthday-party", null); try t.expect(!f.keeps("bufo-party", "https://x/bufo-party.png")); try t.expect(f.keeps("bufo-birthday-party", "https://x/bufo-birthday-party.png")); } test "pattern subset: anchors and alternation" { try t.expect(Pattern.parse("^bigbufo_").matches("bigbufo_0_0")); try t.expect(!Pattern.parse("^bigbufo_").matches("not-bigbufo_0_0")); try t.expect(Pattern.parse("party$").matches("bufo-party")); try t.expect(!Pattern.parse("party$").matches("bufo-party-hat")); try t.expect(Pattern.parse("^bufo-lgtm$").matches("bufo-lgtm")); try t.expect(!Pattern.parse("^bufo$").matches("bufo-lgtm")); } test "word boundaries: the bot's \\brip\\b" { const rip = Pattern.parse("\\brip\\b"); try t.expect(rip.matches("bufo-rip")); try t.expect(rip.matches("rip-bufo")); try t.expect(rip.matches("bufo-rip-in-peace")); try t.expect(rip.matches("rip")); try t.expect(!rip.matches("bufo-trip")); try t.expect(!rip.matches("bufo-ripple")); try t.expect(!rip.matches("bufo-strips")); // a later occurrence can satisfy the boundary even if the first doesn't try t.expect(rip.matches("bufo-trips-then-rip")); try t.expect(!Pattern.hasUnsupportedMeta("\\brip\\b")); try t.expect(Pattern.hasUnsupportedMeta("\\d+")); var arena = std.heap.ArenaAllocator.init(t.allocator); defer arena.deinit(); const ps = try parsePatterns(arena.allocator(), "excited|party, ^tsa"); try t.expectEqual(@as(usize, 3), ps.len); try t.expect(anyMatch(ps, "bufo-party")); try t.expect(anyMatch(ps, "tsa-bufo")); try t.expect(!anyMatch(ps, "bufo-tsa")); } test "formats filter by extension, case-insensitive, empty allows all" { var arena = std.heap.ArenaAllocator.init(t.allocator); defer arena.deinit(); const a = arena.allocator(); const all = try ContentFilter.init(a, false, null, null, null); try t.expect(all.keeps("x", "https://x/bufo.gif")); const png = try ContentFilter.init(a, false, null, null, " PNG, .jpg "); try t.expect(png.keeps("x", "https://x/bufo.png")); try t.expect(png.keeps("x", "https://x/bufo.PNG")); try t.expect(!png.keeps("x", "https://x/bufo.gif")); try t.expect(!png.keeps("x", "https://x/bufo")); }