diff --git a/src/adblock/adblock.zig b/src/adblock/adblock.zig deleted file mode 100644 index 2963354ad..000000000 --- a/src/adblock/adblock.zig +++ /dev/null @@ -1,302 +0,0 @@ -// Copyright (C) 2023-2026 Lightpanda (Selecy SAS) -// -// Francis Bouvier -// Pierre Tachoire -// -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU Affero General Public License as -// published by the Free Software Foundation, either version 3 of the -// License, or (at your option) any later version. -// -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU Affero General Public License for more details. -// -// You should have received a copy of the GNU Affero General Public License -// along with this program. If not, see . - -const std = @import("std"); - -pub const AdBlocker = @import("AdBlocker.zig"); -pub const NetworkFilter = @import("NetworkFilter.zig"); -pub const domain = @import("domain.zig"); - -pub const LineClass = enum(u3) { - empty, - comment, - network, - /// Recognized syntax we deliberately do not support (AdGuard HTML filtering `$$`). - unsupported, - - /// Classifies one trimmed line. There is no cosmetic-separator scan. - pub fn fromLine(line: []const u8) LineClass { - if (line.len == 0) return .empty; - - switch (line[0]) { - '!', '#' => return .comment, - '[' => { - if (std.ascii.startsWithIgnoreCase(line, "[adblock")) return .comment; - }, - else => {}, - } - - if (line[0] == '|' or std.mem.startsWith(u8, line, "@@|")) return .network; - if (std.mem.indexOf(u8, line, "$$") != null) return .unsupported; - return .network; - } -}; - -pub const ParseStats = struct { - lines: usize = 0, - network: usize = 0, - comments: usize = 0, - /// Valid syntax outside the supported subset (cosmetic filters, - /// scriptlets, AdGuard forms, modifier options, ...). - unsupported: usize = 0, - /// Lines with an option name we do not recognize at all. - unknown_option: usize = 0, - /// Malformed lines. - invalid: usize = 0, - /// Hosts-file noise ("127.0.0.1 localhost"). - ignored: usize = 0, -}; - -pub const Parser = struct { - reader: *std.Io.Reader, - stats: ParseStats = .{}, - title: ?[]const u8 = null, - expires: ?[]const u8 = null, - homepage: ?[]const u8 = null, - version: ?[]const u8 = null, - /// Metadata headers only count until the first filter line. - in_header: bool = true, - first_line: bool = true, - - pub fn init(reader: *std.Io.Reader) Parser { - return .{ .reader = reader }; - } - - pub const Error = error{ OutOfMemory, ReadFailed, StreamTooLong }; - - /// Returns the next network filter, or null at end of list. - pub fn next(self: *Parser, arena: std.mem.Allocator) Error!?NetworkFilter { - while (try self.reader.takeDelimiter('\n')) |raw_line| { - var stripped: []const u8 = raw_line; - if (self.first_line) { - self.first_line = false; - if (std.mem.startsWith(u8, stripped, "\xEF\xBB\xBF")) { - stripped = stripped[3..]; - } - } - const line = std.mem.trim(u8, stripped, &std.ascii.whitespace); - self.stats.lines += 1; - - switch (LineClass.fromLine(line)) { - .empty => {}, - .comment => { - self.stats.comments += 1; - if (self.in_header) try self.parseMetadata(arena, line); - }, - .unsupported => { - self.in_header = false; - self.stats.unsupported += 1; - }, - .network => { - self.in_header = false; - if (NetworkFilter.parse(arena, line)) |filter| { - self.stats.network += 1; - return filter; - } else |err| switch (err) { - error.OutOfMemory => return error.OutOfMemory, - error.Ignored => self.stats.ignored += 1, - error.UnknownOption => self.stats.unknown_option += 1, - error.UnsupportedOption, - error.UnsupportedPattern, - error.NoSupportedDomains, - => self.stats.unsupported += 1, - error.InvalidPattern, - error.InvalidOption, - error.InvalidDomainList, - => self.stats.invalid += 1, - } - }, - } - } - return null; - } - - /// `! Key: value` headers in the leading comment block. First - /// occurrence wins; identical text later in the file is just a comment. - fn parseMetadata(self: *Parser, arena: std.mem.Allocator, line: []const u8) std.mem.Allocator.Error!void { - if (line.len == 0 or line[0] != '!') return; - const rest = std.mem.trimStart(u8, line[1..], &std.ascii.whitespace); - const colon = std.mem.indexOfScalar(u8, rest, ':') orelse return; - const key = std.mem.trim(u8, rest[0..colon], &std.ascii.whitespace); - const value = std.mem.trim(u8, rest[colon + 1 ..], &std.ascii.whitespace); - if (value.len == 0) return; - - const slot: *?[]const u8 = if (std.ascii.eqlIgnoreCase(key, "title")) - &self.title - else if (std.ascii.eqlIgnoreCase(key, "expires")) - &self.expires - else if (std.ascii.eqlIgnoreCase(key, "homepage")) - &self.homepage - else if (std.ascii.eqlIgnoreCase(key, "version")) - &self.version - else - return; - if (slot.* == null) slot.* = try arena.dupe(u8, value); - } -}; - -pub const List = struct { - network: []NetworkFilter, - title: ?[]const u8 = null, - expires: ?[]const u8 = null, - homepage: ?[]const u8 = null, - version: ?[]const u8 = null, - stats: ParseStats, - - pub fn parse(arena: std.mem.Allocator, text: []const u8) std.mem.Allocator.Error!List { - var reader: std.Io.Reader = .fixed(text); - var parser: Parser = .init(&reader); - - var network: std.ArrayList(NetworkFilter) = .empty; - while (parser.next(arena) catch |err| switch (err) { - error.OutOfMemory => return error.OutOfMemory, - // A fixed reader's buffer is the whole text: reads cannot fail and no line can outgrow the buffer. - error.ReadFailed, error.StreamTooLong => unreachable, - }) |filter| { - try network.append(arena, filter); - } - - return .{ - .network = try network.toOwnedSlice(arena), - .title = parser.title, - .expires = parser.expires, - .homepage = parser.homepage, - .version = parser.version, - .stats = parser.stats, - }; - } -}; - -const testing = std.testing; - -test "adblock: line classification" { - try testing.expectEqual(.empty, LineClass.fromLine("")); - try testing.expectEqual(.comment, LineClass.fromLine("! EasyList")); - try testing.expectEqual(.comment, LineClass.fromLine("[Adblock Plus 2.0]")); - // Pre-parsing directives are plain comments; their blocks parse - // unconditionally. - try testing.expectEqual(.comment, LineClass.fromLine("!#if env_mobile")); - try testing.expectEqual(.comment, LineClass.fromLine("!#endif")); - // Every '#'-prefixed line is a comment, including generic cosmetic - // filters — there is no separator scan. - try testing.expectEqual(.comment, LineClass.fromLine("# hosts-style comment")); - try testing.expectEqual(.comment, LineClass.fromLine("#### section")); - try testing.expectEqual(.comment, LineClass.fromLine("#nosep")); - try testing.expectEqual(.comment, LineClass.fromLine("## heading text")); - try testing.expectEqual(.comment, LineClass.fromLine("##.ad-banner")); - try testing.expectEqual(.comment, LineClass.fromLine("###banner")); - - try testing.expectEqual(.network, LineClass.fromLine("||ads.example.com^")); - try testing.expectEqual(.network, LineClass.fromLine("@@|https://example.com/path#frag|")); - try testing.expectEqual(.network, LineClass.fromLine("0.0.0.0 tracker.com")); - // Domain-prefixed cosmetic lines classify as network; the filter - // parser drops them via their '#' (see NetworkFilter tests). - try testing.expectEqual(.network, LineClass.fromLine("example.com#@#.ad")); - try testing.expectEqual(.network, LineClass.fromLine("example.com#?#.ad:has-text(x)")); - try testing.expectEqual(.network, LineClass.fromLine("example.com#$#body { padding: 0 }")); - - try testing.expectEqual(.unsupported, LineClass.fromLine("example.com$$script[data-x]")); -} - -test "adblock: Parser yields one filter per next() call" { - var arena_state = std.heap.ArenaAllocator.init(testing.allocator); - defer arena_state.deinit(); - const arena = arena_state.allocator(); - - // BOM-prefixed, no trailing newline on the last line. - var reader: std.Io.Reader = .fixed("\xEF\xBB\xBF! Title: Streamed\n" ++ - "||ads.example.com^\n" ++ - "example.com##.ad-banner\n" ++ - "||tracker.net^"); - var parser: Parser = .init(&reader); - - const first = (try parser.next(arena)).?; - try testing.expectEqualStrings("ads.example.com", first.hostname); - - const second = (try parser.next(arena)).?; - try testing.expectEqualStrings("tracker.net", second.hostname); - - try testing.expect(try parser.next(arena) == null); - try testing.expect(try parser.next(arena) == null); - - try testing.expectEqualStrings("Streamed", parser.title.?); - try testing.expectEqual(2, parser.stats.network); - try testing.expectEqual(1, parser.stats.unsupported); // cosmetic line - try testing.expectEqual(4, parser.stats.lines); -} - -test "adblock: List.parse end to end" { - var arena_state = std.heap.ArenaAllocator.init(testing.allocator); - defer arena_state.deinit(); - const arena = arena_state.allocator(); - - const text = - "[Adblock Plus 2.0]\n" ++ - "! Title: Test List\n" ++ - "! Expires: 4 days (update frequency)\n" ++ - "! Homepage: https://example.org\n" ++ - "!\n" ++ - "||ads.example.com^\n" ++ - "||tracker.net^$third-party,script\n" ++ - "@@||cdn.example.com^$script\n" ++ - "-banner-468x60.\n" ++ - "0.0.0.0 telemetry.example.io\n" ++ - "127.0.0.1 localhost\n" ++ - "##.ad-banner\n" ++ - "example.com###sidebar-ad\n" ++ - "example.com#@#.sponsored\n" ++ - "example.com##+js(no-fetch-if, ads)\n" ++ - "||modifier.example.com^$removeparam=utm_source\n" ++ - "||bogus.example.com^$notarealoption\n" ++ - "!#if env_mobile\n" ++ - "||mobile-only.example.com^\n" ++ - "!#else\n" ++ - "||desktop-only.example.com^\n" ++ - "!#endif\n" ++ - "! Title: not metadata anymore\n"; - - const list = try List.parse(arena, text); - - try testing.expectEqualStrings("Test List", list.title.?); - try testing.expectEqualStrings("4 days (update frequency)", list.expires.?); - try testing.expectEqualStrings("https://example.org", list.homepage.?); - - // 5 direct network filters + both branches of the !#if block: the - // directives are comments, their contents parse unconditionally. - try testing.expectEqual(7, list.network.len); - try testing.expectEqual(7, list.stats.network); - // Domain-prefixed cosmetic lines (###sidebar-ad, #@#.sponsored) and - // $removeparam land in unsupported; the generic ##.ad-banner is a - // comment; the whitespace-carrying scriptlet joins localhost in - // ignored. - try testing.expectEqual(3, list.stats.unsupported); - try testing.expectEqual(2, list.stats.ignored); - try testing.expectEqual(1, list.stats.unknown_option); - try testing.expectEqual(0, list.stats.invalid); - - const desktop = list.network[list.network.len - 1]; - try testing.expectEqualStrings("desktop-only.example.com", desktop.hostname); - - // "-banner-468x60." is not hostname-shaped (leading '-'): plain pattern. - try testing.expectEqual(.plain, list.network[3].kind); - try testing.expectEqualStrings("-banner-468x60.", list.network[3].pattern); -} - -test { - std.testing.refAllDecls(@This()); -} diff --git a/src/adblock/AdBlocker.zig b/src/network/adblock/AdBlocker.zig similarity index 69% rename from src/adblock/AdBlocker.zig rename to src/network/adblock/AdBlocker.zig index 64b1ecb95..2077e34d6 100644 --- a/src/adblock/AdBlocker.zig +++ b/src/network/adblock/AdBlocker.zig @@ -20,16 +20,13 @@ const std = @import("std"); const Io = std.Io; const Allocator = std.mem.Allocator; -const adblock = @import("adblock.zig"); +const Parser = @import("Parser.zig"); const HostnameTrie = @import("HostnameTrie.zig"); const NetworkFilter = @import("NetworkFilter.zig"); const AdBlocker = @This(); -arena: std.heap.ArenaAllocator, -/// Network filters the tries cannot express: patterns, type/party/domain -/// constraints, badfilter directives. -filters: std.ArrayList(NetworkFilter), +allocator: Allocator, trie: HostnameTrie, blocked: u32, /// $important hostname blocks: they beat exceptions. @@ -45,8 +42,7 @@ pub fn init(allocator: Allocator) Allocator.Error!AdBlocker { const allowed = try trie.createTrie(allocator); return .{ - .arena = std.heap.ArenaAllocator.init(allocator), - .filters = .empty, + .allocator = allocator, .trie = trie, .blocked = blocked, .blocked_important = blocked_important, @@ -55,40 +51,38 @@ pub fn init(allocator: Allocator) Allocator.Error!AdBlocker { } pub fn deinit(self: *AdBlocker) void { - self.trie.deinit(self.arena.child_allocator); - self.arena.deinit(); + self.trie.deinit(self.allocator); } -pub fn parse(self: *AdBlocker, reader: *Io.Reader) !adblock.ParseStats { - var scratch_instance = std.heap.ArenaAllocator.init(self.arena.child_allocator); +pub fn parse(self: *AdBlocker, reader: *Io.Reader) !void { + var scratch_instance = std.heap.ArenaAllocator.init(self.allocator); defer scratch_instance.deinit(); const scratch = scratch_instance.allocator(); - const arena = self.arena.allocator(); - var parser: adblock.Parser = .init(reader); - - var parsed: std.ArrayList(NetworkFilter) = .empty; + var parser: Parser = .init(reader); while (try parser.next(scratch)) |filter| { - if (self.trieRoot(&filter)) |root| { - // Duplicate and subdomain-of-existing entries drop. - self.trie.add(self.arena.child_allocator, root, filter.hostname) catch |err| switch (err) { - error.OutOfMemory, error.TrieFull => |e| return e, - // The parser never yields a .hostname filter without one. - error.InvalidHostname => unreachable, - }; - continue; - } - try parsed.append(scratch, try filter.dupe(arena)); - } + // Nothing survives the iteration but the trie entry, so the + // scratch arena is recycled rather than grown per filter. + defer _ = scratch_instance.reset(.retain_capacity); - try self.filters.appendSlice(arena, parsed.items); - return parser.stats; + // Filters the tries cannot express (patterns, type/party/domain + // constraints, badfilter directives) are dropped: deciding those + // needs a request engine we do not have yet, and retaining them + // costs tens of MB on a list like EasyList. + const root = self.trieRoot(&filter) orelse continue; + // Duplicate and subdomain-of-existing entries drop. + self.trie.add(self.allocator, root, filter.hostname) catch |err| switch (err) { + error.OutOfMemory, error.TrieFull => |e| return e, + // The parser never yields a .hostname filter without one. + error.InvalidHostname => unreachable, + }; + } } pub const Verdict = enum { none, allowed, blocked }; -/// `.none` means no hostname-wide filter applies; the filters kept in `filters` may -/// still have an opinion once the request engine exists. +/// `.none` means no hostname-wide filter applies. Filters that need more +/// than a hostname match are not represented here at all. pub fn matchHostname(self: *const AdBlocker, hostname: []const u8) Verdict { if (self.trie.matches(self.blocked_important, hostname) != null) return .blocked; if (self.trie.matches(self.allowed, hostname) != null) return .allowed; @@ -97,7 +91,7 @@ pub fn matchHostname(self: *const AdBlocker, hostname: []const u8) Verdict { } /// The trie holding this filter, or null when the filter's behavior is -/// more than a hostname-wide match and must stay in `filters`. +/// more than a hostname-wide match. fn trieRoot(self: *const AdBlocker, filter: *const NetworkFilter) ?u32 { if (filter.kind != .hostname) return null; // `||host` without '^' also matches hostnames merely *starting* with @@ -117,9 +111,9 @@ fn trieRoot(self: *const AdBlocker, filter: *const NetworkFilter) ?u32 { return if (filter.important) self.blocked_important else self.blocked; } -const testing = std.testing; +const testing = @import("../../testing.zig"); -test "adblock.AdBlocker: parse accumulates filters across lists" { +test "adblock.AdBlocker: parse accumulates across lists" { var blocker: AdBlocker = try .init(testing.allocator); defer blocker.deinit(); @@ -129,13 +123,10 @@ test "adblock.AdBlocker: parse accumulates filters across lists" { \\@@||cdn.example.com^$script \\example.com##.ad-banner ); - const first_stats = try blocker.parse(&first); + try blocker.parse(&first); // The pure-hostname block went into the trie; the $script exception - // is type-restricted and stays a filter. - try testing.expectEqual(1, blocker.filters.items.len); - try testing.expectEqual(2, first_stats.network); - try testing.expectEqual(1, first_stats.unsupported); // cosmetic line + // is type-restricted, so it is dropped rather than allowing the host. try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com")); try testing.expectEqual(.blocked, blocker.matchHostname("sub.ads.example.com")); try testing.expectEqual(.none, blocker.matchHostname("cdn.example.com")); @@ -143,21 +134,11 @@ test "adblock.AdBlocker: parse accumulates filters across lists" { var second: Io.Reader = .fixed( \\||tracker.net^$third-party,domain=news.com|~sports.news.com ); - const second_stats = try blocker.parse(&second); + try blocker.parse(&second); - try testing.expectEqual(2, blocker.filters.items.len); - try testing.expectEqual(1, second_stats.network); try testing.expectEqual(.none, blocker.matchHostname("tracker.net")); - - // The scratch arena holding each list's text is gone: every retained - // string must have been deep-copied. - try testing.expect(blocker.filters.items[0].exception); - try testing.expectEqualStrings("cdn.example.com", blocker.filters.items[0].hostname); - const tracker = blocker.filters.items[1]; - try testing.expectEqualStrings("tracker.net", tracker.hostname); - try testing.expect(!tracker.first_party); - try testing.expectEqualStrings("news.com", tracker.domains.included[0].value); - try testing.expectEqualStrings("sports.news.com", tracker.domains.excluded[0].value); + // The first list's entries survived the second parse. + try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com")); } test "adblock.AdBlocker: hostname verdict precedence" { @@ -170,7 +151,7 @@ test "adblock.AdBlocker: hostname verdict precedence" { \\||evil.com^$important \\@@||evil.com^ ); - _ = try blocker.parse(&list); + try blocker.parse(&list); try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com")); // The exception wins over the plain block... @@ -191,9 +172,8 @@ test "adblock.AdBlocker: trie absorbs every pure-hostname form" { \\bare-hostname.example.com \\0.0.0.0 hosts-style.example.io ); - _ = try blocker.parse(&list); + try blocker.parse(&list); - try testing.expectEqual(0, blocker.filters.items.len); try testing.expectEqual(.blocked, blocker.matchHostname("anchored.example.com")); try testing.expectEqual(.blocked, blocker.matchHostname("bare-hostname.example.com")); try testing.expectEqual(.blocked, blocker.matchHostname("hosts-style.example.io")); @@ -211,9 +191,8 @@ test "adblock.AdBlocker: constrained filters stay out of the trie" { \\||bad.example.com^$badfilter \\@@||cosmetic.example.com^$generichide ); - _ = try blocker.parse(&list); + try blocker.parse(&list); - try testing.expectEqual(6, blocker.filters.items.len); try testing.expectEqual(.none, blocker.matchHostname("no-caret.example.com")); try testing.expectEqual(.none, blocker.matchHostname("typed.example.com")); try testing.expectEqual(.none, blocker.matchHostname("party.example.com")); diff --git a/src/adblock/HostnameTrie.zig b/src/network/adblock/HostnameTrie.zig similarity index 99% rename from src/adblock/HostnameTrie.zig rename to src/network/adblock/HostnameTrie.zig index e41f33302..23fa5653a 100644 --- a/src/adblock/HostnameTrie.zig +++ b/src/network/adblock/HostnameTrie.zig @@ -251,7 +251,7 @@ fn addSegment(self: *HostnameTrie, allocator: Allocator, hostname: []const u8, l return .{ .len = @intCast(len), .boundary = false, .offset = @intCast(offset) }; } -const testing = std.testing; +const testing = @import("../../testing.zig"); test "adblock.HostnameTrie: exact and subdomain matching" { var trie: HostnameTrie = try .init(testing.allocator); @@ -266,7 +266,7 @@ test "adblock.HostnameTrie: exact and subdomain matching" { const needle = "metrics.ssl.doubleclick.net"; const offset = trie.matches(root, needle).?; - try testing.expectEqualStrings("doubleclick.net", needle[offset..]); + try testing.expectString("doubleclick.net", needle[offset..]); try testing.expectEqual(null, trie.matches(root, "google.net")); try testing.expectEqual(null, trie.matches(root, "evilgoogle.com")); diff --git a/src/adblock/NetworkFilter.zig b/src/network/adblock/NetworkFilter.zig similarity index 94% rename from src/adblock/NetworkFilter.zig rename to src/network/adblock/NetworkFilter.zig index 5f439731c..c178fdad2 100644 --- a/src/adblock/NetworkFilter.zig +++ b/src/network/adblock/NetworkFilter.zig @@ -281,23 +281,14 @@ pub fn parse(arena: std.mem.Allocator, line: []const u8) ParseError!NetworkFilte return filter; } -/// Deep-copies the filter so it outlives the list text and arena it was -/// parsed from (parsed slices may alias both). -pub fn dupe(self: *const NetworkFilter, arena: std.mem.Allocator) std.mem.Allocator.Error!NetworkFilter { - var out = self.*; - out.pattern = try arena.dupe(u8, self.pattern); - out.hostname = try arena.dupe(u8, self.hostname); - out.domains = try self.domains.dupe(arena); - return out; -} - /// uBO allows a trailing " # comment" on lines containing whitespace. fn stripInlineComment(line: []const u8) []const u8 { - var i: usize = 1; - while (i < line.len) : (i += 1) { - if (line[i] == '#' and std.ascii.isWhitespace(line[i - 1])) { - return std.mem.trimEnd(u8, line[0..i], &std.ascii.whitespace); + var start: usize = 1; + while (std.mem.indexOfScalarPos(u8, line, start, '#')) |pos| { + if (std.ascii.isWhitespace(line[pos - 1])) { + return std.mem.trimEnd(u8, line[0..pos], &std.ascii.whitespace); } + start = pos + 1; } return line; } @@ -682,7 +673,7 @@ fn isRedirectHostName(host: []const u8) bool { return false; } -const testing = std.testing; +const testing = @import("../../testing.zig"); fn testParse(arena: std.mem.Allocator, line: []const u8) ParseError!NetworkFilter { return parse(arena, line); @@ -696,7 +687,7 @@ test "adblock.NetworkFilter: pure hostname forms" { // The single most common rule shape: `||host^`. var f = try testParse(arena, "||ads.example.com^"); try testing.expectEqual(.hostname, f.kind); - try testing.expectEqualStrings("ads.example.com", f.hostname); + try testing.expectString("ads.example.com", f.hostname); try testing.expect(f.require_separator); try testing.expect(!f.exception); // Implicit "strict" blocking: documents included. @@ -712,13 +703,13 @@ test "adblock.NetworkFilter: pure hostname forms" { // Bare hostname line == ||host^ (uBO divergence from ABP). f = try testParse(arena, "tracker.example.net"); try testing.expectEqual(.hostname, f.kind); - try testing.expectEqualStrings("tracker.example.net", f.hostname); + try testing.expectString("tracker.example.net", f.hostname); try testing.expect(f.require_separator); // Raw IPv4 lines (URLhaus style). f = try testParse(arena, "101.126.11.168"); try testing.expectEqual(.hostname, f.kind); - try testing.expectEqualStrings("101.126.11.168", f.hostname); + try testing.expectString("101.126.11.168", f.hostname); } test "adblock.NetworkFilter: hosts-file lines" { @@ -728,11 +719,11 @@ test "adblock.NetworkFilter: hosts-file lines" { var f = try testParse(arena, "0.0.0.0 ads.tracker.com"); try testing.expectEqual(.hostname, f.kind); - try testing.expectEqualStrings("ads.tracker.com", f.hostname); + try testing.expectString("ads.tracker.com", f.hostname); try testing.expectEqual(ResourceTypes.all.bits(), f.types.bits()); f = try testParse(arena, "127.0.0.1 AdServer.Example.com # inline comment"); - try testing.expectEqualStrings("adserver.example.com", f.hostname); + try testing.expectString("adserver.example.com", f.hostname); // Hosts noise is silently ignored, not an error. try testing.expectError(error.Ignored, testParse(arena, "127.0.0.1 localhost")); @@ -748,27 +739,27 @@ test "adblock.NetworkFilter: anchors and pattern kinds" { var f = try testParse(arena, "/banner/ads."); try testing.expectEqual(.plain, f.kind); - try testing.expectEqualStrings("/banner/ads.", f.pattern); + try testing.expectString("/banner/ads.", f.pattern); // Starting AND ending with '/' means regex, not path substring — lists // write "*/banner/" or "/banner/*" to force substring semantics. f = try testParse(arena, "/banner/ads/"); try testing.expectEqual(.regex, f.kind); - try testing.expectEqualStrings("banner/ads", f.pattern); + try testing.expectString("banner/ads", f.pattern); f = try testParse(arena, "|https://ads."); try testing.expectEqual(.plain, f.kind); try testing.expect(f.left_anchor); - try testing.expectEqualStrings("https://ads.", f.pattern); + try testing.expectString("https://ads.", f.pattern); f = try testParse(arena, "-Ad-300x250.gif|"); try testing.expect(f.right_anchor); - try testing.expectEqualStrings("-ad-300x250.gif", f.pattern); + try testing.expectString("-ad-300x250.gif", f.pattern); f = try testParse(arena, "||example.com/ads/*.js"); try testing.expectEqual(.wildcard, f.kind); - try testing.expectEqualStrings("example.com", f.hostname); - try testing.expectEqualStrings("/ads/*.js", f.pattern); + try testing.expectString("example.com", f.hostname); + try testing.expectString("/ads/*.js", f.pattern); f = try testParse(arena, "/ads/banner^"); try testing.expectEqual(.wildcard, f.kind); @@ -777,19 +768,19 @@ test "adblock.NetworkFilter: anchors and pattern kinds" { f = try testParse(arena, "*-ads-*|"); try testing.expectEqual(.plain, f.kind); try testing.expect(!f.right_anchor); - try testing.expectEqualStrings("-ads-", f.pattern); + try testing.expectString("-ads-", f.pattern); // A pattern that trims down to a bare hostname shape gets promoted // (uBO flavor rules), even a single label. f = try testParse(arena, "*ads*|"); try testing.expectEqual(.hostname, f.kind); - try testing.expectEqualStrings("ads", f.hostname); + try testing.expectString("ads", f.hostname); // '||' hostname region containing '*' stays a generic pattern. f = try testParse(arena, "||example.*/ads"); try testing.expectEqual(.wildcard, f.kind); try testing.expect(f.hostname_anchor); - try testing.expectEqualStrings("", f.hostname); + try testing.expectString("", f.hostname); } test "adblock.NetworkFilter: regex literals" { @@ -799,12 +790,12 @@ test "adblock.NetworkFilter: regex literals" { var f = try testParse(arena, "/banner\\d+/"); try testing.expectEqual(.regex, f.kind); - try testing.expectEqualStrings("banner\\d+", f.pattern); + try testing.expectString("banner\\d+", f.pattern); // '$' inside a regex must not be mistaken for an options separator. f = try testParse(arena, "/ads\\$/"); try testing.expectEqual(.regex, f.kind); - try testing.expectEqualStrings("ads\\$", f.pattern); + try testing.expectString("ads\\$", f.pattern); // ... but a real options list after a regex still splits. f = try testParse(arena, "/^https?:.*banner/$image"); @@ -891,7 +882,7 @@ test "adblock.NetworkFilter: domain option" { const f = try testParse(arena, "||ads.com^$script,domain=news.com|~sports.news.com|google.*"); try testing.expectEqual(2, f.domains.included.len); try testing.expectEqual(1, f.domains.excluded.len); - try testing.expectEqualStrings("news.com", f.domains.included[0].value); + try testing.expectString("news.com", f.domains.included[0].value); try testing.expect(f.domains.included[1].entity); try testing.expectError(error.InvalidOption, testParse(arena, "||ads.com^$domain=")); @@ -931,7 +922,7 @@ test "adblock.NetworkFilter: unsupported and modifier options" { // $redirect keeps its blocking half; the directive itself is ignored. var f = try testParse(arena, "||ads.com/ad.js$script,redirect=noopjs"); try testing.expect(f.types.script); - try testing.expectEqualStrings("ads.com", f.hostname); + try testing.expectString("ads.com", f.hostname); f = try testParse(arena, "||ads.com/v.mp4$mp4"); try testing.expect(f.types.media); @@ -972,7 +963,7 @@ test "adblock.NetworkFilter: uppercase patterns are normalized" { const arena = arena_state.allocator(); const f = try testParse(arena, "||Ads.Example.COM^"); - try testing.expectEqualStrings("ads.example.com", f.hostname); + try testing.expectString("ads.example.com", f.hostname); // Option names are lowercase in the wild; uppercase names are unknown. try testing.expectError(error.UnknownOption, testParse(arena, "||ads.com^$Script")); diff --git a/src/network/adblock/Parser.zig b/src/network/adblock/Parser.zig new file mode 100644 index 000000000..793816301 --- /dev/null +++ b/src/network/adblock/Parser.zig @@ -0,0 +1,250 @@ +// Copyright (C) 2023-2026 Lightpanda (Selecy SAS) +// +// Francis Bouvier +// Pierre Tachoire +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU Affero General Public License as +// published by the Free Software Foundation, either version 3 of the +// License, or (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU Affero General Public License for more details. +// +// You should have received a copy of the GNU Affero General Public License +// along with this program. If not, see . + +//! Streams an EasyList-syntax filter list, yielding one supported network +//! filter per `next` call. Lines we do not support are skipped. + +const std = @import("std"); + +const NetworkFilter = @import("NetworkFilter.zig"); + +const Allocator = std.mem.Allocator; + +const Parser = @This(); + +reader: *std.Io.Reader, +first_line: bool = true, + +pub const Error = error{ OutOfMemory, ReadFailed }; + +pub fn init(reader: *std.Io.Reader) Parser { + return .{ .reader = reader }; +} + +/// Returns the next network filter, or null at end of list. Allocations come +/// from `arena`; the returned filter borrows from it and from the reader's +/// buffer, so it only stays valid until the following call. +pub fn next(self: *Parser, arena: Allocator) Error!?NetworkFilter { + while (try self.takeLine()) |raw_line| { + var stripped: []const u8 = raw_line; + if (self.first_line) { + self.first_line = false; + if (std.mem.startsWith(u8, stripped, "\xEF\xBB\xBF")) { + stripped = stripped[3..]; + } + } + const line = std.mem.trim(u8, stripped, &std.ascii.whitespace); + + switch (LineClass.fromLine(line)) { + .empty, .comment, .unsupported => {}, + .network => { + if (NetworkFilter.parse(arena, line)) |filter| { + return filter; + } else |err| switch (err) { + error.OutOfMemory => return error.OutOfMemory, + // Anything else is a line outside the supported subset + // or malformed; either way it is not ours to enforce. + else => {}, + } + }, + } + } + return null; +} + +/// `Io.Reader.takeDelimiter`, except that a line too long to fit the reader's +/// buffer is dropped instead of aborting the list. Nothing that long is a +/// filter we could support, and a single monster line should not cost us the +/// rest of the file. +fn takeLine(self: *Parser) error{ReadFailed}!?[]u8 { + while (true) { + if (self.reader.takeDelimiter('\n')) |line| { + return line; + } else |err| switch (err) { + error.ReadFailed => return error.ReadFailed, + error.StreamTooLong => { + // takeDelimiter leaves the stream untouched on StreamTooLong, + // so the oversized line still has to be stepped over. + _ = self.reader.discardDelimiterInclusive('\n') catch |e| switch (e) { + error.ReadFailed => return error.ReadFailed, + error.EndOfStream => return null, + }; + }, + } + } +} + +const LineClass = enum { + empty, + comment, + network, + /// Recognized syntax we deliberately do not support (AdGuard HTML filtering `$$`). + unsupported, + + /// Classifies one trimmed line. There is no cosmetic-separator scan. + fn fromLine(line: []const u8) LineClass { + if (line.len == 0) return .empty; + + switch (line[0]) { + '!', '#' => return .comment, + '[' => { + if (std.ascii.startsWithIgnoreCase(line, "[adblock")) return .comment; + }, + else => {}, + } + + if (line[0] == '|' or std.mem.startsWith(u8, line, "@@|")) return .network; + if (std.mem.indexOf(u8, line, "$$") != null) return .unsupported; + return .network; + } +}; + +const testing = @import("../../testing.zig"); + +test "adblock.Parser: line classification" { + try testing.expectEqual(.empty, LineClass.fromLine("")); + try testing.expectEqual(.comment, LineClass.fromLine("! EasyList")); + try testing.expectEqual(.comment, LineClass.fromLine("[Adblock Plus 2.0]")); + // Pre-parsing directives are plain comments; their blocks parse + // unconditionally. + try testing.expectEqual(.comment, LineClass.fromLine("!#if env_mobile")); + try testing.expectEqual(.comment, LineClass.fromLine("!#endif")); + // Every '#'-prefixed line is a comment, including generic cosmetic + // filters — there is no separator scan. + try testing.expectEqual(.comment, LineClass.fromLine("# hosts-style comment")); + try testing.expectEqual(.comment, LineClass.fromLine("#### section")); + try testing.expectEqual(.comment, LineClass.fromLine("#nosep")); + try testing.expectEqual(.comment, LineClass.fromLine("## heading text")); + try testing.expectEqual(.comment, LineClass.fromLine("##.ad-banner")); + try testing.expectEqual(.comment, LineClass.fromLine("###banner")); + + try testing.expectEqual(.network, LineClass.fromLine("||ads.example.com^")); + try testing.expectEqual(.network, LineClass.fromLine("@@|https://example.com/path#frag|")); + try testing.expectEqual(.network, LineClass.fromLine("0.0.0.0 tracker.com")); + // Domain-prefixed cosmetic lines classify as network; the filter + // parser drops them via their '#' (see NetworkFilter tests). + try testing.expectEqual(.network, LineClass.fromLine("example.com#@#.ad")); + try testing.expectEqual(.network, LineClass.fromLine("example.com#?#.ad:has-text(x)")); + try testing.expectEqual(.network, LineClass.fromLine("example.com#$#body { padding: 0 }")); + + try testing.expectEqual(.unsupported, LineClass.fromLine("example.com$$script[data-x]")); +} + +test "adblock.Parser: yields one filter per next() call" { + var arena_state = std.heap.ArenaAllocator.init(testing.allocator); + defer arena_state.deinit(); + const arena = arena_state.allocator(); + + // BOM-prefixed, no trailing newline on the last line. + var reader: std.Io.Reader = .fixed("\xEF\xBB\xBF! Title: Streamed\n" ++ + "||ads.example.com^\n" ++ + "example.com##.ad-banner\n" ++ + "||tracker.net^"); + var parser: Parser = .init(&reader); + + const first = (try parser.next(arena)).?; + try testing.expectString("ads.example.com", first.hostname); + + const second = (try parser.next(arena)).?; + try testing.expectString("tracker.net", second.hostname); + + // The header, the cosmetic line and the end of the list all yield + // nothing, and next() keeps returning null once drained. + try testing.expect(try parser.next(arena) == null); + try testing.expect(try parser.next(arena) == null); +} + +test "adblock.Parser: a line too long for the buffer is skipped" { + var arena_state = std.heap.ArenaAllocator.init(testing.allocator); + defer arena_state.deinit(); + const arena = arena_state.allocator(); + + var text: std.ArrayList(u8) = .empty; + defer text.deinit(testing.allocator); + try text.appendSlice(testing.allocator, "||ads.example.com^\n||"); + try text.appendNTimes(testing.allocator, 'a', 128); + try text.appendSlice(testing.allocator, ".example.com^\n||tracker.net^\n"); + + // A buffer too small to ever hold the middle line. + var buf: [64]u8 = undefined; + var text_reader: std.Io.Reader = .fixed(text.items); + var reader = text_reader.limited(.unlimited, &buf); + var parser: Parser = .init(&reader.interface); + + const first = (try parser.next(arena)).?; + try testing.expectString("ads.example.com", first.hostname); + + // The oversized line was stepped over, not treated as end of list. + const second = (try parser.next(arena)).?; + try testing.expectString("tracker.net", second.hostname); + + try testing.expect(try parser.next(arena) == null); +} + +test "adblock.Parser: full list" { + var arena_state = std.heap.ArenaAllocator.init(testing.allocator); + defer arena_state.deinit(); + const arena = arena_state.allocator(); + + var reader: std.Io.Reader = .fixed( + "[Adblock Plus 2.0]\n" ++ + "! Title: Test List\n" ++ + "! Expires: 4 days (update frequency)\n" ++ + "!\n" ++ + "||ads.example.com^\n" ++ + "||tracker.net^$third-party,script\n" ++ + "@@||cdn.example.com^$script\n" ++ + "-banner-468x60.\n" ++ + "0.0.0.0 telemetry.example.io\n" ++ + "127.0.0.1 localhost\n" ++ + "##.ad-banner\n" ++ + "example.com###sidebar-ad\n" ++ + "example.com#@#.sponsored\n" ++ + "example.com##+js(no-fetch-if, ads)\n" ++ + "||modifier.example.com^$removeparam=utm_source\n" ++ + "||bogus.example.com^$notarealoption\n" ++ + "!#if env_mobile\n" ++ + "||mobile-only.example.com^\n" ++ + "!#else\n" ++ + "||desktop-only.example.com^\n" ++ + "!#endif\n", + ); + var parser: Parser = .init(&reader); + + // A filter only borrows the reader's buffer until the next call, so + // assert on each one as it comes out. + var count: usize = 0; + while (try parser.next(arena)) |filter| : (count += 1) { + switch (count) { + // "-banner-468x60." is not hostname-shaped (leading '-'), so it + // stays a plain pattern rather than becoming a hostname filter. + 3 => { + try testing.expectEqual(.plain, filter.kind); + try testing.expectString("-banner-468x60.", filter.pattern); + }, + 6 => try testing.expectString("desktop-only.example.com", filter.hostname), + else => {}, + } + } + + // 5 direct network filters + both branches of the !#if block: the + // directives are comments, their contents parse unconditionally. + // Everything else in the list — cosmetic lines, the scriptlet, the + // hosts-file noise, $removeparam and the unknown option — is skipped. + try testing.expectEqual(7, count); +}