Adblocker: changes on request construction & tokenization

* Engine.Request.fromHttp(req, source_url, buffers) now builds the adblock request straight from HttpClient.Request.
* The URL is tokenized once per request (hashed into the Request, shared by all engines); capped at 128 tokens (same as adblock-rust).
* Document hostname longer than 253 bytes now skips adblocking.
This commit is contained in:
Halil Durak committed 2026-09-07 10:57:36 +03:00
1 parent 6a85b1f386
commit b60f790a03
2 files changed
+198 -75

No files matched your search

+108 -72
View File
@@ -398,59 +398,32 @@ fn clearUrlBlocklist(self: *Client) void {
/// Every reason a request is refused before it reaches the network:
/// `--block-urls` patterns and the `--adblock-lists` filters both land here
/// so that no call site can apply one without the other.
fn isUrlBlocked(self: *const Client, req: *const Request) bool {
fn isUrlBlocked(self: *const Client, transfer: *const Transfer) bool {
const req = &transfer.req;
if (req.internal) return false;
if (self.url_blocklist) |*blocklist| {
if (blocklist.isBlocked(req.url)) return true;
}
return self.isAdblocked(req);
return self.isAdblocked(transfer);
}
/// Longest URL we will match filters against.
/// Anything past this is not a resource a filter list has an opinion about.
const ADBLOCK_URL_MAX = 8 * 1024;
fn isAdblocked(self: *const Client, req: *const Request) bool {
fn isAdblocked(self: *const Client, transfer: *const Transfer) bool {
const blocker = if (self.network.adblocker) |*b| b else return false;
var url_buf: [ADBLOCK_URL_MAX]u8 = undefined;
const url = normalizeForAdblock(req.url, &url_buf) orelse return false;
var source_buf: [253]u8 = undefined;
const top_level = req.resource_type == .document and !req.is_subframe;
const source_host = if (top_level) "" else URL.getOriginHostname(req.cookie_origin);
const source = if (source_host.len == 0 or source_host.len > source_buf.len)
""
else
std.ascii.lowerString(&source_buf, source_host);
return blocker.match(.init(url, source, adblockResourceType(req))) == .blocked;
var buffers: AdBlocker.Request.Buffers = undefined;
const target = AdBlocker.Request.fromHttp(
&transfer.req,
adblockSourceUrl(transfer),
&buffers,
) orelse return false;
return blocker.match(target) == .blocked;
}
fn normalizeForAdblock(url: [:0]const u8, buf: []u8) ?[]const u8 {
const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len;
const trimmed = url[0..end];
const upper = for (trimmed, 0..) |c, i| {
if (std.ascii.isUpper(c)) break i;
} else return trimmed;
if (trimmed.len > buf.len) return null;
const out = buf[0..trimmed.len];
@memcpy(out[0..upper], trimmed[0..upper]);
_ = std.ascii.lowerString(out[upper..], trimmed[upper..]);
return out;
}
fn adblockResourceType(req: *const Request) AdBlocker.ResourceTypes {
return switch (req.resource_type) {
.document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true },
.script => .{ .script = true },
.stylesheet => .{ .stylesheet = true },
.xhr, .fetch => .{ .xmlhttprequest = true },
.image => .{ .image = true },
.eventsource => .{ .other = true },
};
fn adblockSourceUrl(transfer: *const Transfer) ?[]const u8 {
var owner: *const Owner = transfer.owner orelse return null;
if (transfer.req.resource_type == .document) {
owner = owner.parent orelse return null;
}
return owner.documentUrl();
}
fn isCrossOriginModeAllowed(transfer: *const Transfer) bool {
@@ -1063,7 +1036,7 @@ fn pipeline(self: *Client, transfer: *Transfer, from: SubmitFrom) !void {
continue :sw SubmitFrom.after_intercept;
},
.after_intercept => {
if (self.isUrlBlocked(&transfer.req)) {
if (self.isUrlBlocked(transfer)) {
log.info(.http, "blocked url", .{ .url = transfer.req.url });
return transfer.failAsync(error.UrlBlocked);
}
@@ -2147,6 +2120,18 @@ pub const Owner = struct {
const Blob = @import("../browser/webapi/Blob.zig");
/// The URL of the document this owner's requests belong to.
/// Handles `about:` case also.
pub fn documentUrl(self: *const Owner) ?[:0]const u8 {
var source = self;
while (true) {
if (source.url) |url| {
if (!std.mem.startsWith(u8, url.*, "about:")) return url.*;
}
source = source.parent orelse return null;
}
}
// RFC 6265bis "site for cookies"
pub fn siteForCookies(self: *const Owner) Cookie.SiteForCookies {
var source = self;
@@ -4031,14 +4016,14 @@ const Synthetic = struct {
const testing = @import("../testing.zig");
// Only the transfer list matters to the tests using it: they build their
// transfers by hand and never go through newRequest.
fn testOwner() Owner {
// Only the transfer list, the url and the parent matter to the tests using
// it: they build their transfers by hand and never go through newRequest.
fn testOwner(url: ?*const [:0]const u8, parent: ?*const Owner) Owner {
return .{
.blob_urls = undefined,
.origin = undefined,
.url = null,
.parent = null,
.url = url,
.parent = parent,
.frame_id = 0,
.document_frame_id = 0,
.loader_id = 0,
@@ -4261,26 +4246,48 @@ test "HttpClient: setBlockedUrls owns, replaces, and clears patterns" {
const TestRequest = struct {
url: [:0]const u8,
document: [:0]const u8 = "",
/// The page embedding `document`, when the test wants a deeper chain.
parent_document: [:0]const u8 = "",
resource_type: Request.ResourceType = .document,
is_subframe: bool = false,
internal: bool = false,
};
fn testIsUrlBlocked(client: *const Client, opts: TestRequest) bool {
const req: Request = .{
.frame_id = 0,
.loader_id = 0,
.method = .GET,
.url = opts.url,
.cookie_jar = null,
.cookie_origin = opts.document,
.resource_type = opts.resource_type,
.is_subframe = opts.is_subframe,
.internal = opts.internal,
.notification = undefined,
.shutdown_callback = noopShutdown,
// The owner chain a real request carries: [0] the embedding page,
// [1] the request's document, [2] the frame being navigated — whose url
// slot already holds the target, so its context is its parent's.
var chain: [3]Owner = undefined;
chain[0] = testOwner(
if (opts.parent_document.len == 0) null else &opts.parent_document,
null,
);
chain[1] = testOwner(
if (opts.document.len == 0) null else &opts.document,
if (opts.parent_document.len == 0) null else &chain[0],
);
chain[2] = testOwner(&opts.url, if (opts.document.len == 0) null else &chain[1]);
var transfer: Transfer = .{
.arena = undefined,
.owner = if (opts.resource_type == .document)
&chain[2]
else if (opts.document.len == 0)
null
else
&chain[1],
.req = .{
.method = .GET,
.url = opts.url,
.resource_type = opts.resource_type,
.is_subframe = opts.is_subframe,
.internal = opts.internal,
.shutdown_callback = noopShutdown,
},
.client = undefined,
.start_time = 0,
};
return client.isUrlBlocked(&req);
return client.isUrlBlocked(&transfer);
}
test "HttpClient: adblock verdicts apply per request" {
@@ -4354,6 +4361,35 @@ test "HttpClient: adblock verdicts apply per request" {
.document = "https://news.com/",
.is_subframe = true,
}));
// The context is the issuing frame's document even under a cross-site
// ancestor (the canonical ad iframe) — site-for-cookies semantics would
// collapse this chain to nothing and lose the party.
try testing.expect(testIsUrlBlocked(&client, .{
.url = "https://partied.example.com/x.js",
.document = "https://adprovider.com/frame.html",
.parent_document = "https://news.com/",
.resource_type = .script,
}));
// A document hostname DNS could not carry is nothing a filter list has
// an opinion about: the request is let through, not matched sourceless.
try testing.expect(testIsUrlBlocked(&client, .{
.url = "https://ads.example.com/pixel.gif",
.document = "https://news.com/",
.resource_type = .image,
}));
try testing.expect(!testIsUrlBlocked(&client, .{
.url = "https://ads.example.com/pixel.gif",
.document = "https://" ++ "a" ** 254 ++ ".com/",
.resource_type = .image,
}));
// Same for a URL too long to normalize (uppercase forces the copy).
try testing.expect(!testIsUrlBlocked(&client, .{
.url = "https://ads.example.com/" ++ "A" ** (8 * 1024),
.document = "https://news.com/",
.resource_type = .image,
}));
}
test "HttpClient: URL blocking exempts internal transfers" {
@@ -4518,21 +4554,21 @@ test "HttpClient: Fetch header overrides restore after one hop" {
test "HttpClient: Owner.siteForCookies" {
var top_url: [:0]const u8 = "http://attacker.example/attacker-nested";
var top = testOwner();
var top = testOwner(null, null);
top.url = &top_url;
var middle_url: [:0]const u8 = "http://victim.example/nested-middle";
var middle = testOwner();
var middle = testOwner(null, null);
middle.url = &middle_url;
middle.parent = ⊤
var inner_url: [:0]const u8 = "http://victim.example/inner";
var inner = testOwner();
var inner = testOwner(null, null);
inner.url = &inner_url;
inner.parent = &middle;
// A worker has no site of its own; it takes its creating document's.
var worker = testOwner();
var worker = testOwner(null, null);
worker.parent = &inner;
// A top-level document is its own site.
@@ -4585,7 +4621,7 @@ test "HttpClient: fulfillIntercepted survives a done_callback that tears down th
defer client.processGraveyard();
defer client.transfers.deinit(testing.allocator);
var owner = testOwner();
var owner = testOwner(null, null);
const Ctx = struct {
client: *Client,
@@ -4658,7 +4694,7 @@ test "HttpClient: kill during done_callback does not also fire shutdown_callback
defer client.processGraveyard();
defer client.transfers.deinit(testing.allocator);
var owner = testOwner();
var owner = testOwner(null, null);
const Ctx = struct {
client: *Client,
@@ -4738,7 +4774,7 @@ test "HttpClient: kill during a non-terminal callback defers shutdown_callback"
defer client.processGraveyard();
defer client.transfers.deinit(testing.allocator);
var owner = testOwner();
var owner = testOwner(null, null);
const Ctx = struct {
client: *Client,
@@ -5053,7 +5089,7 @@ test "HttpClient: abortParked survives an error_callback that tears down the own
defer client.processGraveyard();
defer client.transfers.deinit(testing.allocator);
var owner = testOwner();
var owner = testOwner(null, null);
const Ctx = struct {
client: *Client,
@@ -5127,7 +5163,7 @@ test "HttpClient: abort survives an error_callback that tears down the owner" {
defer client.processGraveyard();
defer client.transfers.deinit(testing.allocator);
var owner = testOwner();
var owner = testOwner(null, null);
const Ctx = struct {
client: *Client,
+90 -3
View File
@@ -34,6 +34,9 @@
const std = @import("std");
const URL = @import("../../browser/URL.zig");
const HttpClient = @import("../HttpClient.zig");
const domain = @import("domain.zig");
const pattern = @import("pattern.zig");
const NetworkFilter = @import("NetworkFilter.zig");
@@ -69,6 +72,25 @@ pub const Request = struct {
/// Exactly one bit set.
kind: NetworkFilter.ResourceTypes,
third_party: bool,
/// The URL's tokens, hashed once here so no engine retokenizes. A URL
/// long enough to overflow loses its last tokens.
tokens_buf: [URL_TOKENS_MAX]u32,
tokens_len: usize,
const URL_TOKENS_MAX = 128;
/// Longest URL `fromHttp` will normalize. Anything past this is not a
/// resource a filter list has an opinion about.
const URL_MAX = 8 * 1024;
/// DNS's own hostname limit.
const SOURCE_MAX = 253;
/// Backs the normalized text of a `fromHttp` request, which stays valid
/// only as long as the buffers do.
pub const Buffers = struct {
url: [URL_MAX]u8,
source: [SOURCE_MAX]u8,
};
/// `url` must already be lowercased and fragment-free; `source_hostname`
/// may be empty when there is no document context.
@@ -79,11 +101,77 @@ pub const Request = struct {
) Request {
const parsed: pattern.Url = .init(url);
const source = if (source_hostname.len == 0) parsed.hostname() else source_hostname;
return .{
var request: Request = .{
.url = parsed,
.source_hostname = source,
.kind = kind,
.third_party = domain.isThirdParty(parsed.hostname(), source),
.tokens_buf = undefined,
.tokens_len = 0,
};
var it: Tokens = .{ .text = url };
while (it.next()) |token| {
if (request.tokens_len == request.tokens_buf.len) break;
request.tokens_buf[request.tokens_len] = token;
request.tokens_len += 1;
}
return request;
}
/// Builds the request `req` is matched as, `source` being the URL (or
/// origin serialization) of the document it was issued from. null when
/// there is none.
pub fn fromHttp(
req: *const HttpClient.Request,
source_url: ?[]const u8,
buffers: *Buffers,
) ?Request {
const url = normalizeUrl(req.url, &buffers.url) orelse return null;
// Matching top-level nav against the page it was clicked on would make
// every link third party.
const top_level = req.resource_type == .document and !req.is_subframe;
const source_host = if (top_level)
""
else if (source_url) |u|
URL.getOriginHostname(u)
else
"";
if (source_host.len > buffers.source.len) return null;
const source = std.ascii.lowerString(&buffers.source, source_host);
return .init(url, source, resourceType(req));
}
inline fn tokens(self: *const Request) []const u32 {
return self.tokens_buf[0..self.tokens_len];
}
/// Lowercases `url` into `buf` with its fragment stripped; patterns are
/// stored lowercased, and no request URL carries a fragment onto the wire.
fn normalizeUrl(url: []const u8, buf: []u8) ?[]const u8 {
const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len;
const trimmed = url[0..end];
const upper = for (trimmed, 0..) |c, i| {
if (std.ascii.isUpper(c)) break i;
} else return trimmed;
if (trimmed.len > buf.len) return null;
const out = buf[0..trimmed.len];
@memcpy(out[0..upper], trimmed[0..upper]);
_ = std.ascii.lowerString(out[upper..], trimmed[upper..]);
return out;
}
fn resourceType(req: *const HttpClient.Request) NetworkFilter.ResourceTypes {
return switch (req.resource_type) {
.document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true },
.script => .{ .script = true },
.stylesheet => .{ .stylesheet = true },
.xhr, .fetch => .{ .xmlhttprequest = true },
.image => .{ .image = true },
.eventsource => .{ .other = true },
};
}
};
@@ -99,8 +187,7 @@ const TOKENS_MAX = 32;
/// The first filter that matches, or null.
pub fn match(self: *const Engine, request: Request) ?*const NetworkFilter {
if (self.filters.len != 0) {
var it: Tokens = .{ .text = request.url.text };
while (it.next()) |token| {
for (request.tokens()) |token| {
const bucket = self.buckets.get(token) orelse continue;
if (self.matchIn(bucket, request)) |filter| return filter;
}