src/adblock -> src/network/adblock

This commit is contained in:
Halil Durak committed 2026-08-13 16:39:14 +03:00
1 parent 5b5cf9b304
commit 0d8e926bae
5 files changed
+312 -394

No files matched your search

-302
View File
@@ -1,302 +0,0 @@
// Copyright (C) 2023-2026 Lightpanda (Selecy SAS)
//
// Francis Bouvier <francis@lightpanda.io>
// Pierre Tachoire <pierre@lightpanda.io>
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as
// published by the Free Software Foundation, either version 3 of the
// License, or (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.
const std = @import("std");
pub const AdBlocker = @import("AdBlocker.zig");
pub const NetworkFilter = @import("NetworkFilter.zig");
pub const domain = @import("domain.zig");
pub const LineClass = enum(u3) {
empty,
comment,
network,
/// Recognized syntax we deliberately do not support (AdGuard HTML filtering `$$`).
unsupported,
/// Classifies one trimmed line. There is no cosmetic-separator scan.
pub fn fromLine(line: []const u8) LineClass {
if (line.len == 0) return .empty;
switch (line[0]) {
'!', '#' => return .comment,
'[' => {
if (std.ascii.startsWithIgnoreCase(line, "[adblock")) return .comment;
},
else => {},
}
if (line[0] == '|' or std.mem.startsWith(u8, line, "@@|")) return .network;
if (std.mem.indexOf(u8, line, "$$") != null) return .unsupported;
return .network;
}
};
pub const ParseStats = struct {
lines: usize = 0,
network: usize = 0,
comments: usize = 0,
/// Valid syntax outside the supported subset (cosmetic filters,
/// scriptlets, AdGuard forms, modifier options, ...).
unsupported: usize = 0,
/// Lines with an option name we do not recognize at all.
unknown_option: usize = 0,
/// Malformed lines.
invalid: usize = 0,
/// Hosts-file noise ("127.0.0.1 localhost").
ignored: usize = 0,
};
pub const Parser = struct {
reader: *std.Io.Reader,
stats: ParseStats = .{},
title: ?[]const u8 = null,
expires: ?[]const u8 = null,
homepage: ?[]const u8 = null,
version: ?[]const u8 = null,
/// Metadata headers only count until the first filter line.
in_header: bool = true,
first_line: bool = true,
pub fn init(reader: *std.Io.Reader) Parser {
return .{ .reader = reader };
}
pub const Error = error{ OutOfMemory, ReadFailed, StreamTooLong };
/// Returns the next network filter, or null at end of list.
pub fn next(self: *Parser, arena: std.mem.Allocator) Error!?NetworkFilter {
while (try self.reader.takeDelimiter('\n')) |raw_line| {
var stripped: []const u8 = raw_line;
if (self.first_line) {
self.first_line = false;
if (std.mem.startsWith(u8, stripped, "\xEF\xBB\xBF")) {
stripped = stripped[3..];
}
}
const line = std.mem.trim(u8, stripped, &std.ascii.whitespace);
self.stats.lines += 1;
switch (LineClass.fromLine(line)) {
.empty => {},
.comment => {
self.stats.comments += 1;
if (self.in_header) try self.parseMetadata(arena, line);
},
.unsupported => {
self.in_header = false;
self.stats.unsupported += 1;
},
.network => {
self.in_header = false;
if (NetworkFilter.parse(arena, line)) |filter| {
self.stats.network += 1;
return filter;
} else |err| switch (err) {
error.OutOfMemory => return error.OutOfMemory,
error.Ignored => self.stats.ignored += 1,
error.UnknownOption => self.stats.unknown_option += 1,
error.UnsupportedOption,
error.UnsupportedPattern,
error.NoSupportedDomains,
=> self.stats.unsupported += 1,
error.InvalidPattern,
error.InvalidOption,
error.InvalidDomainList,
=> self.stats.invalid += 1,
}
},
}
}
return null;
}
/// `! Key: value` headers in the leading comment block. First
/// occurrence wins; identical text later in the file is just a comment.
fn parseMetadata(self: *Parser, arena: std.mem.Allocator, line: []const u8) std.mem.Allocator.Error!void {
if (line.len == 0 or line[0] != '!') return;
const rest = std.mem.trimStart(u8, line[1..], &std.ascii.whitespace);
const colon = std.mem.indexOfScalar(u8, rest, ':') orelse return;
const key = std.mem.trim(u8, rest[0..colon], &std.ascii.whitespace);
const value = std.mem.trim(u8, rest[colon + 1 ..], &std.ascii.whitespace);
if (value.len == 0) return;
const slot: *?[]const u8 = if (std.ascii.eqlIgnoreCase(key, "title"))
&self.title
else if (std.ascii.eqlIgnoreCase(key, "expires"))
&self.expires
else if (std.ascii.eqlIgnoreCase(key, "homepage"))
&self.homepage
else if (std.ascii.eqlIgnoreCase(key, "version"))
&self.version
else
return;
if (slot.* == null) slot.* = try arena.dupe(u8, value);
}
};
pub const List = struct {
network: []NetworkFilter,
title: ?[]const u8 = null,
expires: ?[]const u8 = null,
homepage: ?[]const u8 = null,
version: ?[]const u8 = null,
stats: ParseStats,
pub fn parse(arena: std.mem.Allocator, text: []const u8) std.mem.Allocator.Error!List {
var reader: std.Io.Reader = .fixed(text);
var parser: Parser = .init(&reader);
var network: std.ArrayList(NetworkFilter) = .empty;
while (parser.next(arena) catch |err| switch (err) {
error.OutOfMemory => return error.OutOfMemory,
// A fixed reader's buffer is the whole text: reads cannot fail and no line can outgrow the buffer.
error.ReadFailed, error.StreamTooLong => unreachable,
}) |filter| {
try network.append(arena, filter);
}
return .{
.network = try network.toOwnedSlice(arena),
.title = parser.title,
.expires = parser.expires,
.homepage = parser.homepage,
.version = parser.version,
.stats = parser.stats,
};
}
};
const testing = std.testing;
test "adblock: line classification" {
try testing.expectEqual(.empty, LineClass.fromLine(""));
try testing.expectEqual(.comment, LineClass.fromLine("! EasyList"));
try testing.expectEqual(.comment, LineClass.fromLine("[Adblock Plus 2.0]"));
// Pre-parsing directives are plain comments; their blocks parse
// unconditionally.
try testing.expectEqual(.comment, LineClass.fromLine("!#if env_mobile"));
try testing.expectEqual(.comment, LineClass.fromLine("!#endif"));
// Every '#'-prefixed line is a comment, including generic cosmetic
// filters — there is no separator scan.
try testing.expectEqual(.comment, LineClass.fromLine("# hosts-style comment"));
try testing.expectEqual(.comment, LineClass.fromLine("#### section"));
try testing.expectEqual(.comment, LineClass.fromLine("#nosep"));
try testing.expectEqual(.comment, LineClass.fromLine("## heading text"));
try testing.expectEqual(.comment, LineClass.fromLine("##.ad-banner"));
try testing.expectEqual(.comment, LineClass.fromLine("###banner"));
try testing.expectEqual(.network, LineClass.fromLine("||ads.example.com^"));
try testing.expectEqual(.network, LineClass.fromLine("@@|https://example.com/path#frag|"));
try testing.expectEqual(.network, LineClass.fromLine("0.0.0.0 tracker.com"));
// Domain-prefixed cosmetic lines classify as network; the filter
// parser drops them via their '#' (see NetworkFilter tests).
try testing.expectEqual(.network, LineClass.fromLine("example.com#@#.ad"));
try testing.expectEqual(.network, LineClass.fromLine("example.com#?#.ad:has-text(x)"));
try testing.expectEqual(.network, LineClass.fromLine("example.com#$#body { padding: 0 }"));
try testing.expectEqual(.unsupported, LineClass.fromLine("example.com$$script[data-x]"));
}
test "adblock: Parser yields one filter per next() call" {
var arena_state = std.heap.ArenaAllocator.init(testing.allocator);
defer arena_state.deinit();
const arena = arena_state.allocator();
// BOM-prefixed, no trailing newline on the last line.
var reader: std.Io.Reader = .fixed("\xEF\xBB\xBF! Title: Streamed\n" ++
"||ads.example.com^\n" ++
"example.com##.ad-banner\n" ++
"||tracker.net^");
var parser: Parser = .init(&reader);
const first = (try parser.next(arena)).?;
try testing.expectEqualStrings("ads.example.com", first.hostname);
const second = (try parser.next(arena)).?;
try testing.expectEqualStrings("tracker.net", second.hostname);
try testing.expect(try parser.next(arena) == null);
try testing.expect(try parser.next(arena) == null);
try testing.expectEqualStrings("Streamed", parser.title.?);
try testing.expectEqual(2, parser.stats.network);
try testing.expectEqual(1, parser.stats.unsupported); // cosmetic line
try testing.expectEqual(4, parser.stats.lines);
}
test "adblock: List.parse end to end" {
var arena_state = std.heap.ArenaAllocator.init(testing.allocator);
defer arena_state.deinit();
const arena = arena_state.allocator();
const text =
"[Adblock Plus 2.0]\n" ++
"! Title: Test List\n" ++
"! Expires: 4 days (update frequency)\n" ++
"! Homepage: https://example.org\n" ++
"!\n" ++
"||ads.example.com^\n" ++
"||tracker.net^$third-party,script\n" ++
"@@||cdn.example.com^$script\n" ++
"-banner-468x60.\n" ++
"0.0.0.0 telemetry.example.io\n" ++
"127.0.0.1 localhost\n" ++
"##.ad-banner\n" ++
"example.com###sidebar-ad\n" ++
"example.com#@#.sponsored\n" ++
"example.com##+js(no-fetch-if, ads)\n" ++
"||modifier.example.com^$removeparam=utm_source\n" ++
"||bogus.example.com^$notarealoption\n" ++
"!#if env_mobile\n" ++
"||mobile-only.example.com^\n" ++
"!#else\n" ++
"||desktop-only.example.com^\n" ++
"!#endif\n" ++
"! Title: not metadata anymore\n";
const list = try List.parse(arena, text);
try testing.expectEqualStrings("Test List", list.title.?);
try testing.expectEqualStrings("4 days (update frequency)", list.expires.?);
try testing.expectEqualStrings("https://example.org", list.homepage.?);
// 5 direct network filters + both branches of the !#if block: the
// directives are comments, their contents parse unconditionally.
try testing.expectEqual(7, list.network.len);
try testing.expectEqual(7, list.stats.network);
// Domain-prefixed cosmetic lines (###sidebar-ad, #@#.sponsored) and
// $removeparam land in unsupported; the generic ##.ad-banner is a
// comment; the whitespace-carrying scriptlet joins localhost in
// ignored.
try testing.expectEqual(3, list.stats.unsupported);
try testing.expectEqual(2, list.stats.ignored);
try testing.expectEqual(1, list.stats.unknown_option);
try testing.expectEqual(0, list.stats.invalid);
const desktop = list.network[list.network.len - 1];
try testing.expectEqualStrings("desktop-only.example.com", desktop.hostname);
// "-banner-468x60." is not hostname-shaped (leading '-'): plain pattern.
try testing.expectEqual(.plain, list.network[3].kind);
try testing.expectEqualStrings("-banner-468x60.", list.network[3].pattern);
}
test {
std.testing.refAllDecls(@This());
}
@@ -20,16 +20,13 @@ const std = @import("std");
const Io = std.Io;
const Allocator = std.mem.Allocator;
const adblock = @import("adblock.zig");
const Parser = @import("Parser.zig");
const HostnameTrie = @import("HostnameTrie.zig");
const NetworkFilter = @import("NetworkFilter.zig");
const AdBlocker = @This();
arena: std.heap.ArenaAllocator,
/// Network filters the tries cannot express: patterns, type/party/domain
/// constraints, badfilter directives.
filters: std.ArrayList(NetworkFilter),
allocator: Allocator,
trie: HostnameTrie,
blocked: u32,
/// $important hostname blocks: they beat exceptions.
@@ -45,8 +42,7 @@ pub fn init(allocator: Allocator) Allocator.Error!AdBlocker {
const allowed = try trie.createTrie(allocator);
return .{
.arena = std.heap.ArenaAllocator.init(allocator),
.filters = .empty,
.allocator = allocator,
.trie = trie,
.blocked = blocked,
.blocked_important = blocked_important,
@@ -55,40 +51,38 @@ pub fn init(allocator: Allocator) Allocator.Error!AdBlocker {
}
pub fn deinit(self: *AdBlocker) void {
self.trie.deinit(self.arena.child_allocator);
self.arena.deinit();
self.trie.deinit(self.allocator);
}
pub fn parse(self: *AdBlocker, reader: *Io.Reader) !adblock.ParseStats {
var scratch_instance = std.heap.ArenaAllocator.init(self.arena.child_allocator);
pub fn parse(self: *AdBlocker, reader: *Io.Reader) !void {
var scratch_instance = std.heap.ArenaAllocator.init(self.allocator);
defer scratch_instance.deinit();
const scratch = scratch_instance.allocator();
const arena = self.arena.allocator();
var parser: adblock.Parser = .init(reader);
var parsed: std.ArrayList(NetworkFilter) = .empty;
var parser: Parser = .init(reader);
while (try parser.next(scratch)) |filter| {
if (self.trieRoot(&filter)) |root| {
// Duplicate and subdomain-of-existing entries drop.
self.trie.add(self.arena.child_allocator, root, filter.hostname) catch |err| switch (err) {
error.OutOfMemory, error.TrieFull => |e| return e,
// The parser never yields a .hostname filter without one.
error.InvalidHostname => unreachable,
};
continue;
}
try parsed.append(scratch, try filter.dupe(arena));
}
// Nothing survives the iteration but the trie entry, so the
// scratch arena is recycled rather than grown per filter.
defer _ = scratch_instance.reset(.retain_capacity);
try self.filters.appendSlice(arena, parsed.items);
return parser.stats;
// Filters the tries cannot express (patterns, type/party/domain
// constraints, badfilter directives) are dropped: deciding those
// needs a request engine we do not have yet, and retaining them
// costs tens of MB on a list like EasyList.
const root = self.trieRoot(&filter) orelse continue;
// Duplicate and subdomain-of-existing entries drop.
self.trie.add(self.allocator, root, filter.hostname) catch |err| switch (err) {
error.OutOfMemory, error.TrieFull => |e| return e,
// The parser never yields a .hostname filter without one.
error.InvalidHostname => unreachable,
};
}
}
pub const Verdict = enum { none, allowed, blocked };
/// `.none` means no hostname-wide filter applies; the filters kept in `filters` may
/// still have an opinion once the request engine exists.
/// `.none` means no hostname-wide filter applies. Filters that need more
/// than a hostname match are not represented here at all.
pub fn matchHostname(self: *const AdBlocker, hostname: []const u8) Verdict {
if (self.trie.matches(self.blocked_important, hostname) != null) return .blocked;
if (self.trie.matches(self.allowed, hostname) != null) return .allowed;
@@ -97,7 +91,7 @@ pub fn matchHostname(self: *const AdBlocker, hostname: []const u8) Verdict {
}
/// The trie holding this filter, or null when the filter's behavior is
/// more than a hostname-wide match and must stay in `filters`.
/// more than a hostname-wide match.
fn trieRoot(self: *const AdBlocker, filter: *const NetworkFilter) ?u32 {
if (filter.kind != .hostname) return null;
// `||host` without '^' also matches hostnames merely *starting* with
@@ -117,9 +111,9 @@ fn trieRoot(self: *const AdBlocker, filter: *const NetworkFilter) ?u32 {
return if (filter.important) self.blocked_important else self.blocked;
}
const testing = std.testing;
const testing = @import("../../testing.zig");
test "adblock.AdBlocker: parse accumulates filters across lists" {
test "adblock.AdBlocker: parse accumulates across lists" {
var blocker: AdBlocker = try .init(testing.allocator);
defer blocker.deinit();
@@ -129,13 +123,10 @@ test "adblock.AdBlocker: parse accumulates filters across lists" {
\\@@||cdn.example.com^$script
\\example.com##.ad-banner
);
const first_stats = try blocker.parse(&first);
try blocker.parse(&first);
// The pure-hostname block went into the trie; the $script exception
// is type-restricted and stays a filter.
try testing.expectEqual(1, blocker.filters.items.len);
try testing.expectEqual(2, first_stats.network);
try testing.expectEqual(1, first_stats.unsupported); // cosmetic line
// is type-restricted, so it is dropped rather than allowing the host.
try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com"));
try testing.expectEqual(.blocked, blocker.matchHostname("sub.ads.example.com"));
try testing.expectEqual(.none, blocker.matchHostname("cdn.example.com"));
@@ -143,21 +134,11 @@ test "adblock.AdBlocker: parse accumulates filters across lists" {
var second: Io.Reader = .fixed(
\\||tracker.net^$third-party,domain=news.com|~sports.news.com
);
const second_stats = try blocker.parse(&second);
try blocker.parse(&second);
try testing.expectEqual(2, blocker.filters.items.len);
try testing.expectEqual(1, second_stats.network);
try testing.expectEqual(.none, blocker.matchHostname("tracker.net"));
// The scratch arena holding each list's text is gone: every retained
// string must have been deep-copied.
try testing.expect(blocker.filters.items[0].exception);
try testing.expectEqualStrings("cdn.example.com", blocker.filters.items[0].hostname);
const tracker = blocker.filters.items[1];
try testing.expectEqualStrings("tracker.net", tracker.hostname);
try testing.expect(!tracker.first_party);
try testing.expectEqualStrings("news.com", tracker.domains.included[0].value);
try testing.expectEqualStrings("sports.news.com", tracker.domains.excluded[0].value);
// The first list's entries survived the second parse.
try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com"));
}
test "adblock.AdBlocker: hostname verdict precedence" {
@@ -170,7 +151,7 @@ test "adblock.AdBlocker: hostname verdict precedence" {
\\||evil.com^$important
\\@@||evil.com^
);
_ = try blocker.parse(&list);
try blocker.parse(&list);
try testing.expectEqual(.blocked, blocker.matchHostname("ads.example.com"));
// The exception wins over the plain block...
@@ -191,9 +172,8 @@ test "adblock.AdBlocker: trie absorbs every pure-hostname form" {
\\bare-hostname.example.com
\\0.0.0.0 hosts-style.example.io
);
_ = try blocker.parse(&list);
try blocker.parse(&list);
try testing.expectEqual(0, blocker.filters.items.len);
try testing.expectEqual(.blocked, blocker.matchHostname("anchored.example.com"));
try testing.expectEqual(.blocked, blocker.matchHostname("bare-hostname.example.com"));
try testing.expectEqual(.blocked, blocker.matchHostname("hosts-style.example.io"));
@@ -211,9 +191,8 @@ test "adblock.AdBlocker: constrained filters stay out of the trie" {
\\||bad.example.com^$badfilter
\\@@||cosmetic.example.com^$generichide
);
_ = try blocker.parse(&list);
try blocker.parse(&list);
try testing.expectEqual(6, blocker.filters.items.len);
try testing.expectEqual(.none, blocker.matchHostname("no-caret.example.com"));
try testing.expectEqual(.none, blocker.matchHostname("typed.example.com"));
try testing.expectEqual(.none, blocker.matchHostname("party.example.com"));
@@ -251,7 +251,7 @@ fn addSegment(self: *HostnameTrie, allocator: Allocator, hostname: []const u8, l
return .{ .len = @intCast(len), .boundary = false, .offset = @intCast(offset) };
}
const testing = std.testing;
const testing = @import("../../testing.zig");
test "adblock.HostnameTrie: exact and subdomain matching" {
var trie: HostnameTrie = try .init(testing.allocator);
@@ -266,7 +266,7 @@ test "adblock.HostnameTrie: exact and subdomain matching" {
const needle = "metrics.ssl.doubleclick.net";
const offset = trie.matches(root, needle).?;
try testing.expectEqualStrings("doubleclick.net", needle[offset..]);
try testing.expectString("doubleclick.net", needle[offset..]);
try testing.expectEqual(null, trie.matches(root, "google.net"));
try testing.expectEqual(null, trie.matches(root, "evilgoogle.com"));
@@ -281,23 +281,14 @@ pub fn parse(arena: std.mem.Allocator, line: []const u8) ParseError!NetworkFilte
return filter;
}
/// Deep-copies the filter so it outlives the list text and arena it was
/// parsed from (parsed slices may alias both).
pub fn dupe(self: *const NetworkFilter, arena: std.mem.Allocator) std.mem.Allocator.Error!NetworkFilter {
var out = self.*;
out.pattern = try arena.dupe(u8, self.pattern);
out.hostname = try arena.dupe(u8, self.hostname);
out.domains = try self.domains.dupe(arena);
return out;
}
/// uBO allows a trailing " # comment" on lines containing whitespace.
fn stripInlineComment(line: []const u8) []const u8 {
var i: usize = 1;
while (i < line.len) : (i += 1) {
if (line[i] == '#' and std.ascii.isWhitespace(line[i - 1])) {
return std.mem.trimEnd(u8, line[0..i], &std.ascii.whitespace);
var start: usize = 1;
while (std.mem.indexOfScalarPos(u8, line, start, '#')) |pos| {
if (std.ascii.isWhitespace(line[pos - 1])) {
return std.mem.trimEnd(u8, line[0..pos], &std.ascii.whitespace);
}
start = pos + 1;
}
return line;
}
@@ -682,7 +673,7 @@ fn isRedirectHostName(host: []const u8) bool {
return false;
}
const testing = std.testing;
const testing = @import("../../testing.zig");
fn testParse(arena: std.mem.Allocator, line: []const u8) ParseError!NetworkFilter {
return parse(arena, line);
@@ -696,7 +687,7 @@ test "adblock.NetworkFilter: pure hostname forms" {
// The single most common rule shape: `||host^`.
var f = try testParse(arena, "||ads.example.com^");
try testing.expectEqual(.hostname, f.kind);
try testing.expectEqualStrings("ads.example.com", f.hostname);
try testing.expectString("ads.example.com", f.hostname);
try testing.expect(f.require_separator);
try testing.expect(!f.exception);
// Implicit "strict" blocking: documents included.
@@ -712,13 +703,13 @@ test "adblock.NetworkFilter: pure hostname forms" {
// Bare hostname line == ||host^ (uBO divergence from ABP).
f = try testParse(arena, "tracker.example.net");
try testing.expectEqual(.hostname, f.kind);
try testing.expectEqualStrings("tracker.example.net", f.hostname);
try testing.expectString("tracker.example.net", f.hostname);
try testing.expect(f.require_separator);
// Raw IPv4 lines (URLhaus style).
f = try testParse(arena, "101.126.11.168");
try testing.expectEqual(.hostname, f.kind);
try testing.expectEqualStrings("101.126.11.168", f.hostname);
try testing.expectString("101.126.11.168", f.hostname);
}
test "adblock.NetworkFilter: hosts-file lines" {
@@ -728,11 +719,11 @@ test "adblock.NetworkFilter: hosts-file lines" {
var f = try testParse(arena, "0.0.0.0 ads.tracker.com");
try testing.expectEqual(.hostname, f.kind);
try testing.expectEqualStrings("ads.tracker.com", f.hostname);
try testing.expectString("ads.tracker.com", f.hostname);
try testing.expectEqual(ResourceTypes.all.bits(), f.types.bits());
f = try testParse(arena, "127.0.0.1 AdServer.Example.com # inline comment");
try testing.expectEqualStrings("adserver.example.com", f.hostname);
try testing.expectString("adserver.example.com", f.hostname);
// Hosts noise is silently ignored, not an error.
try testing.expectError(error.Ignored, testParse(arena, "127.0.0.1 localhost"));
@@ -748,27 +739,27 @@ test "adblock.NetworkFilter: anchors and pattern kinds" {
var f = try testParse(arena, "/banner/ads.");
try testing.expectEqual(.plain, f.kind);
try testing.expectEqualStrings("/banner/ads.", f.pattern);
try testing.expectString("/banner/ads.", f.pattern);
// Starting AND ending with '/' means regex, not path substring — lists
// write "*/banner/" or "/banner/*" to force substring semantics.
f = try testParse(arena, "/banner/ads/");
try testing.expectEqual(.regex, f.kind);
try testing.expectEqualStrings("banner/ads", f.pattern);
try testing.expectString("banner/ads", f.pattern);
f = try testParse(arena, "|https://ads.");
try testing.expectEqual(.plain, f.kind);
try testing.expect(f.left_anchor);
try testing.expectEqualStrings("https://ads.", f.pattern);
try testing.expectString("https://ads.", f.pattern);
f = try testParse(arena, "-Ad-300x250.gif|");
try testing.expect(f.right_anchor);
try testing.expectEqualStrings("-ad-300x250.gif", f.pattern);
try testing.expectString("-ad-300x250.gif", f.pattern);
f = try testParse(arena, "||example.com/ads/*.js");
try testing.expectEqual(.wildcard, f.kind);
try testing.expectEqualStrings("example.com", f.hostname);
try testing.expectEqualStrings("/ads/*.js", f.pattern);
try testing.expectString("example.com", f.hostname);
try testing.expectString("/ads/*.js", f.pattern);
f = try testParse(arena, "/ads/banner^");
try testing.expectEqual(.wildcard, f.kind);
@@ -777,19 +768,19 @@ test "adblock.NetworkFilter: anchors and pattern kinds" {
f = try testParse(arena, "*-ads-*|");
try testing.expectEqual(.plain, f.kind);
try testing.expect(!f.right_anchor);
try testing.expectEqualStrings("-ads-", f.pattern);
try testing.expectString("-ads-", f.pattern);
// A pattern that trims down to a bare hostname shape gets promoted
// (uBO flavor rules), even a single label.
f = try testParse(arena, "*ads*|");
try testing.expectEqual(.hostname, f.kind);
try testing.expectEqualStrings("ads", f.hostname);
try testing.expectString("ads", f.hostname);
// '||' hostname region containing '*' stays a generic pattern.
f = try testParse(arena, "||example.*/ads");
try testing.expectEqual(.wildcard, f.kind);
try testing.expect(f.hostname_anchor);
try testing.expectEqualStrings("", f.hostname);
try testing.expectString("", f.hostname);
}
test "adblock.NetworkFilter: regex literals" {
@@ -799,12 +790,12 @@ test "adblock.NetworkFilter: regex literals" {
var f = try testParse(arena, "/banner\\d+/");
try testing.expectEqual(.regex, f.kind);
try testing.expectEqualStrings("banner\\d+", f.pattern);
try testing.expectString("banner\\d+", f.pattern);
// '$' inside a regex must not be mistaken for an options separator.
f = try testParse(arena, "/ads\\$/");
try testing.expectEqual(.regex, f.kind);
try testing.expectEqualStrings("ads\\$", f.pattern);
try testing.expectString("ads\\$", f.pattern);
// ... but a real options list after a regex still splits.
f = try testParse(arena, "/^https?:.*banner/$image");
@@ -891,7 +882,7 @@ test "adblock.NetworkFilter: domain option" {
const f = try testParse(arena, "||ads.com^$script,domain=news.com|~sports.news.com|google.*");
try testing.expectEqual(2, f.domains.included.len);
try testing.expectEqual(1, f.domains.excluded.len);
try testing.expectEqualStrings("news.com", f.domains.included[0].value);
try testing.expectString("news.com", f.domains.included[0].value);
try testing.expect(f.domains.included[1].entity);
try testing.expectError(error.InvalidOption, testParse(arena, "||ads.com^$domain="));
@@ -931,7 +922,7 @@ test "adblock.NetworkFilter: unsupported and modifier options" {
// $redirect keeps its blocking half; the directive itself is ignored.
var f = try testParse(arena, "||ads.com/ad.js$script,redirect=noopjs");
try testing.expect(f.types.script);
try testing.expectEqualStrings("ads.com", f.hostname);
try testing.expectString("ads.com", f.hostname);
f = try testParse(arena, "||ads.com/v.mp4$mp4");
try testing.expect(f.types.media);
@@ -972,7 +963,7 @@ test "adblock.NetworkFilter: uppercase patterns are normalized" {
const arena = arena_state.allocator();
const f = try testParse(arena, "||Ads.Example.COM^");
try testing.expectEqualStrings("ads.example.com", f.hostname);
try testing.expectString("ads.example.com", f.hostname);
// Option names are lowercase in the wild; uppercase names are unknown.
try testing.expectError(error.UnknownOption, testParse(arena, "||ads.com^$Script"));
+250
View File
@@ -0,0 +1,250 @@
// Copyright (C) 2023-2026 Lightpanda (Selecy SAS)
//
// Francis Bouvier <francis@lightpanda.io>
// Pierre Tachoire <pierre@lightpanda.io>
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License as
// published by the Free Software Foundation, either version 3 of the
// License, or (at your option) any later version.
//
// This program is distributed in the hope that it will be useful,
// but WITHOUT ANY WARRANTY; without even the implied warranty of
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
// GNU Affero General Public License for more details.
//
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see <https://www.gnu.org/licenses/>.
//! Streams an EasyList-syntax filter list, yielding one supported network
//! filter per `next` call. Lines we do not support are skipped.
const std = @import("std");
const NetworkFilter = @import("NetworkFilter.zig");
const Allocator = std.mem.Allocator;
const Parser = @This();
reader: *std.Io.Reader,
first_line: bool = true,
pub const Error = error{ OutOfMemory, ReadFailed };
pub fn init(reader: *std.Io.Reader) Parser {
return .{ .reader = reader };
}
/// Returns the next network filter, or null at end of list. Allocations come
/// from `arena`; the returned filter borrows from it and from the reader's
/// buffer, so it only stays valid until the following call.
pub fn next(self: *Parser, arena: Allocator) Error!?NetworkFilter {
while (try self.takeLine()) |raw_line| {
var stripped: []const u8 = raw_line;
if (self.first_line) {
self.first_line = false;
if (std.mem.startsWith(u8, stripped, "\xEF\xBB\xBF")) {
stripped = stripped[3..];
}
}
const line = std.mem.trim(u8, stripped, &std.ascii.whitespace);
switch (LineClass.fromLine(line)) {
.empty, .comment, .unsupported => {},
.network => {
if (NetworkFilter.parse(arena, line)) |filter| {
return filter;
} else |err| switch (err) {
error.OutOfMemory => return error.OutOfMemory,
// Anything else is a line outside the supported subset
// or malformed; either way it is not ours to enforce.
else => {},
}
},
}
}
return null;
}
/// `Io.Reader.takeDelimiter`, except that a line too long to fit the reader's
/// buffer is dropped instead of aborting the list. Nothing that long is a
/// filter we could support, and a single monster line should not cost us the
/// rest of the file.
fn takeLine(self: *Parser) error{ReadFailed}!?[]u8 {
while (true) {
if (self.reader.takeDelimiter('\n')) |line| {
return line;
} else |err| switch (err) {
error.ReadFailed => return error.ReadFailed,
error.StreamTooLong => {
// takeDelimiter leaves the stream untouched on StreamTooLong,
// so the oversized line still has to be stepped over.
_ = self.reader.discardDelimiterInclusive('\n') catch |e| switch (e) {
error.ReadFailed => return error.ReadFailed,
error.EndOfStream => return null,
};
},
}
}
}
const LineClass = enum {
empty,
comment,
network,
/// Recognized syntax we deliberately do not support (AdGuard HTML filtering `$$`).
unsupported,
/// Classifies one trimmed line. There is no cosmetic-separator scan.
fn fromLine(line: []const u8) LineClass {
if (line.len == 0) return .empty;
switch (line[0]) {
'!', '#' => return .comment,
'[' => {
if (std.ascii.startsWithIgnoreCase(line, "[adblock")) return .comment;
},
else => {},
}
if (line[0] == '|' or std.mem.startsWith(u8, line, "@@|")) return .network;
if (std.mem.indexOf(u8, line, "$$") != null) return .unsupported;
return .network;
}
};
const testing = @import("../../testing.zig");
test "adblock.Parser: line classification" {
try testing.expectEqual(.empty, LineClass.fromLine(""));
try testing.expectEqual(.comment, LineClass.fromLine("! EasyList"));
try testing.expectEqual(.comment, LineClass.fromLine("[Adblock Plus 2.0]"));
// Pre-parsing directives are plain comments; their blocks parse
// unconditionally.
try testing.expectEqual(.comment, LineClass.fromLine("!#if env_mobile"));
try testing.expectEqual(.comment, LineClass.fromLine("!#endif"));
// Every '#'-prefixed line is a comment, including generic cosmetic
// filters — there is no separator scan.
try testing.expectEqual(.comment, LineClass.fromLine("# hosts-style comment"));
try testing.expectEqual(.comment, LineClass.fromLine("#### section"));
try testing.expectEqual(.comment, LineClass.fromLine("#nosep"));
try testing.expectEqual(.comment, LineClass.fromLine("## heading text"));
try testing.expectEqual(.comment, LineClass.fromLine("##.ad-banner"));
try testing.expectEqual(.comment, LineClass.fromLine("###banner"));
try testing.expectEqual(.network, LineClass.fromLine("||ads.example.com^"));
try testing.expectEqual(.network, LineClass.fromLine("@@|https://example.com/path#frag|"));
try testing.expectEqual(.network, LineClass.fromLine("0.0.0.0 tracker.com"));
// Domain-prefixed cosmetic lines classify as network; the filter
// parser drops them via their '#' (see NetworkFilter tests).
try testing.expectEqual(.network, LineClass.fromLine("example.com#@#.ad"));
try testing.expectEqual(.network, LineClass.fromLine("example.com#?#.ad:has-text(x)"));
try testing.expectEqual(.network, LineClass.fromLine("example.com#$#body { padding: 0 }"));
try testing.expectEqual(.unsupported, LineClass.fromLine("example.com$$script[data-x]"));
}
test "adblock.Parser: yields one filter per next() call" {
var arena_state = std.heap.ArenaAllocator.init(testing.allocator);
defer arena_state.deinit();
const arena = arena_state.allocator();
// BOM-prefixed, no trailing newline on the last line.
var reader: std.Io.Reader = .fixed("\xEF\xBB\xBF! Title: Streamed\n" ++
"||ads.example.com^\n" ++
"example.com##.ad-banner\n" ++
"||tracker.net^");
var parser: Parser = .init(&reader);
const first = (try parser.next(arena)).?;
try testing.expectString("ads.example.com", first.hostname);
const second = (try parser.next(arena)).?;
try testing.expectString("tracker.net", second.hostname);
// The header, the cosmetic line and the end of the list all yield
// nothing, and next() keeps returning null once drained.
try testing.expect(try parser.next(arena) == null);
try testing.expect(try parser.next(arena) == null);
}
test "adblock.Parser: a line too long for the buffer is skipped" {
var arena_state = std.heap.ArenaAllocator.init(testing.allocator);
defer arena_state.deinit();
const arena = arena_state.allocator();
var text: std.ArrayList(u8) = .empty;
defer text.deinit(testing.allocator);
try text.appendSlice(testing.allocator, "||ads.example.com^\n||");
try text.appendNTimes(testing.allocator, 'a', 128);
try text.appendSlice(testing.allocator, ".example.com^\n||tracker.net^\n");
// A buffer too small to ever hold the middle line.
var buf: [64]u8 = undefined;
var text_reader: std.Io.Reader = .fixed(text.items);
var reader = text_reader.limited(.unlimited, &buf);
var parser: Parser = .init(&reader.interface);
const first = (try parser.next(arena)).?;
try testing.expectString("ads.example.com", first.hostname);
// The oversized line was stepped over, not treated as end of list.
const second = (try parser.next(arena)).?;
try testing.expectString("tracker.net", second.hostname);
try testing.expect(try parser.next(arena) == null);
}
test "adblock.Parser: full list" {
var arena_state = std.heap.ArenaAllocator.init(testing.allocator);
defer arena_state.deinit();
const arena = arena_state.allocator();
var reader: std.Io.Reader = .fixed(
"[Adblock Plus 2.0]\n" ++
"! Title: Test List\n" ++
"! Expires: 4 days (update frequency)\n" ++
"!\n" ++
"||ads.example.com^\n" ++
"||tracker.net^$third-party,script\n" ++
"@@||cdn.example.com^$script\n" ++
"-banner-468x60.\n" ++
"0.0.0.0 telemetry.example.io\n" ++
"127.0.0.1 localhost\n" ++
"##.ad-banner\n" ++
"example.com###sidebar-ad\n" ++
"example.com#@#.sponsored\n" ++
"example.com##+js(no-fetch-if, ads)\n" ++
"||modifier.example.com^$removeparam=utm_source\n" ++
"||bogus.example.com^$notarealoption\n" ++
"!#if env_mobile\n" ++
"||mobile-only.example.com^\n" ++
"!#else\n" ++
"||desktop-only.example.com^\n" ++
"!#endif\n",
);
var parser: Parser = .init(&reader);
// A filter only borrows the reader's buffer until the next call, so
// assert on each one as it comes out.
var count: usize = 0;
while (try parser.next(arena)) |filter| : (count += 1) {
switch (count) {
// "-banner-468x60." is not hostname-shaped (leading '-'), so it
// stays a plain pattern rather than becoming a hostname filter.
3 => {
try testing.expectEqual(.plain, filter.kind);
try testing.expectString("-banner-468x60.", filter.pattern);
},
6 => try testing.expectString("desktop-only.example.com", filter.hostname),
else => {},
}
}
// 5 direct network filters + both branches of the !#if block: the
// directives are comments, their contents parse unconditionally.
// Everything else in the list — cosmetic lines, the scriptlet, the
// hosts-file noise, $removeparam and the unknown option — is skipped.
try testing.expectEqual(7, count);
}