From b60f790a03c4cb15bb359a629d84b9e3dad2a6bf Mon Sep 17 00:00:00 2001 From: Halil Durak Date: Wed, 2 Sep 2026 15:41:37 +0300 Subject: [PATCH] `Adblocker`: changes on request construction & tokenization * Engine.Request.fromHttp(req, source_url, buffers) now builds the adblock request straight from HttpClient.Request. * The URL is tokenized once per request (hashed into the Request, shared by all engines); capped at 128 tokens (same as adblock-rust). * Document hostname longer than 253 bytes now skips adblocking. --- src/network/HttpClient.zig | 180 ++++++++++++++++++++------------- src/network/adblock/Engine.zig | 93 ++++++++++++++++- 2 files changed, 198 insertions(+), 75 deletions(-) diff --git a/src/network/HttpClient.zig b/src/network/HttpClient.zig index 6018833b2..b4a83ade0 100644 --- a/src/network/HttpClient.zig +++ b/src/network/HttpClient.zig @@ -398,59 +398,32 @@ fn clearUrlBlocklist(self: *Client) void { /// Every reason a request is refused before it reaches the network: /// `--block-urls` patterns and the `--adblock-lists` filters both land here /// so that no call site can apply one without the other. -fn isUrlBlocked(self: *const Client, req: *const Request) bool { +fn isUrlBlocked(self: *const Client, transfer: *const Transfer) bool { + const req = &transfer.req; if (req.internal) return false; if (self.url_blocklist) |*blocklist| { if (blocklist.isBlocked(req.url)) return true; } - return self.isAdblocked(req); + return self.isAdblocked(transfer); } -/// Longest URL we will match filters against. -/// Anything past this is not a resource a filter list has an opinion about. -const ADBLOCK_URL_MAX = 8 * 1024; - -fn isAdblocked(self: *const Client, req: *const Request) bool { +fn isAdblocked(self: *const Client, transfer: *const Transfer) bool { const blocker = if (self.network.adblocker) |*b| b else return false; - - var url_buf: [ADBLOCK_URL_MAX]u8 = undefined; - const url = normalizeForAdblock(req.url, &url_buf) orelse return false; - - var source_buf: [253]u8 = undefined; - const top_level = req.resource_type == .document and !req.is_subframe; - const source_host = if (top_level) "" else URL.getOriginHostname(req.cookie_origin); - const source = if (source_host.len == 0 or source_host.len > source_buf.len) - "" - else - std.ascii.lowerString(&source_buf, source_host); - - return blocker.match(.init(url, source, adblockResourceType(req))) == .blocked; + var buffers: AdBlocker.Request.Buffers = undefined; + const target = AdBlocker.Request.fromHttp( + &transfer.req, + adblockSourceUrl(transfer), + &buffers, + ) orelse return false; + return blocker.match(target) == .blocked; } -fn normalizeForAdblock(url: [:0]const u8, buf: []u8) ?[]const u8 { - const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len; - const trimmed = url[0..end]; - - const upper = for (trimmed, 0..) |c, i| { - if (std.ascii.isUpper(c)) break i; - } else return trimmed; - - if (trimmed.len > buf.len) return null; - const out = buf[0..trimmed.len]; - @memcpy(out[0..upper], trimmed[0..upper]); - _ = std.ascii.lowerString(out[upper..], trimmed[upper..]); - return out; -} - -fn adblockResourceType(req: *const Request) AdBlocker.ResourceTypes { - return switch (req.resource_type) { - .document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true }, - .script => .{ .script = true }, - .stylesheet => .{ .stylesheet = true }, - .xhr, .fetch => .{ .xmlhttprequest = true }, - .image => .{ .image = true }, - .eventsource => .{ .other = true }, - }; +fn adblockSourceUrl(transfer: *const Transfer) ?[]const u8 { + var owner: *const Owner = transfer.owner orelse return null; + if (transfer.req.resource_type == .document) { + owner = owner.parent orelse return null; + } + return owner.documentUrl(); } fn isCrossOriginModeAllowed(transfer: *const Transfer) bool { @@ -1063,7 +1036,7 @@ fn pipeline(self: *Client, transfer: *Transfer, from: SubmitFrom) !void { continue :sw SubmitFrom.after_intercept; }, .after_intercept => { - if (self.isUrlBlocked(&transfer.req)) { + if (self.isUrlBlocked(transfer)) { log.info(.http, "blocked url", .{ .url = transfer.req.url }); return transfer.failAsync(error.UrlBlocked); } @@ -2147,6 +2120,18 @@ pub const Owner = struct { const Blob = @import("../browser/webapi/Blob.zig"); + /// The URL of the document this owner's requests belong to. + /// Handles `about:` case also. + pub fn documentUrl(self: *const Owner) ?[:0]const u8 { + var source = self; + while (true) { + if (source.url) |url| { + if (!std.mem.startsWith(u8, url.*, "about:")) return url.*; + } + source = source.parent orelse return null; + } + } + // RFC 6265bis "site for cookies" pub fn siteForCookies(self: *const Owner) Cookie.SiteForCookies { var source = self; @@ -4031,14 +4016,14 @@ const Synthetic = struct { const testing = @import("../testing.zig"); -// Only the transfer list matters to the tests using it: they build their -// transfers by hand and never go through newRequest. -fn testOwner() Owner { +// Only the transfer list, the url and the parent matter to the tests using +// it: they build their transfers by hand and never go through newRequest. +fn testOwner(url: ?*const [:0]const u8, parent: ?*const Owner) Owner { return .{ .blob_urls = undefined, .origin = undefined, - .url = null, - .parent = null, + .url = url, + .parent = parent, .frame_id = 0, .document_frame_id = 0, .loader_id = 0, @@ -4261,26 +4246,48 @@ test "HttpClient: setBlockedUrls owns, replaces, and clears patterns" { const TestRequest = struct { url: [:0]const u8, document: [:0]const u8 = "", + /// The page embedding `document`, when the test wants a deeper chain. + parent_document: [:0]const u8 = "", resource_type: Request.ResourceType = .document, is_subframe: bool = false, internal: bool = false, }; fn testIsUrlBlocked(client: *const Client, opts: TestRequest) bool { - const req: Request = .{ - .frame_id = 0, - .loader_id = 0, - .method = .GET, - .url = opts.url, - .cookie_jar = null, - .cookie_origin = opts.document, - .resource_type = opts.resource_type, - .is_subframe = opts.is_subframe, - .internal = opts.internal, - .notification = undefined, - .shutdown_callback = noopShutdown, + // The owner chain a real request carries: [0] the embedding page, + // [1] the request's document, [2] the frame being navigated — whose url + // slot already holds the target, so its context is its parent's. + var chain: [3]Owner = undefined; + chain[0] = testOwner( + if (opts.parent_document.len == 0) null else &opts.parent_document, + null, + ); + chain[1] = testOwner( + if (opts.document.len == 0) null else &opts.document, + if (opts.parent_document.len == 0) null else &chain[0], + ); + chain[2] = testOwner(&opts.url, if (opts.document.len == 0) null else &chain[1]); + + var transfer: Transfer = .{ + .arena = undefined, + .owner = if (opts.resource_type == .document) + &chain[2] + else if (opts.document.len == 0) + null + else + &chain[1], + .req = .{ + .method = .GET, + .url = opts.url, + .resource_type = opts.resource_type, + .is_subframe = opts.is_subframe, + .internal = opts.internal, + .shutdown_callback = noopShutdown, + }, + .client = undefined, + .start_time = 0, }; - return client.isUrlBlocked(&req); + return client.isUrlBlocked(&transfer); } test "HttpClient: adblock verdicts apply per request" { @@ -4354,6 +4361,35 @@ test "HttpClient: adblock verdicts apply per request" { .document = "https://news.com/", .is_subframe = true, })); + + // The context is the issuing frame's document even under a cross-site + // ancestor (the canonical ad iframe) — site-for-cookies semantics would + // collapse this chain to nothing and lose the party. + try testing.expect(testIsUrlBlocked(&client, .{ + .url = "https://partied.example.com/x.js", + .document = "https://adprovider.com/frame.html", + .parent_document = "https://news.com/", + .resource_type = .script, + })); + + // A document hostname DNS could not carry is nothing a filter list has + // an opinion about: the request is let through, not matched sourceless. + try testing.expect(testIsUrlBlocked(&client, .{ + .url = "https://ads.example.com/pixel.gif", + .document = "https://news.com/", + .resource_type = .image, + })); + try testing.expect(!testIsUrlBlocked(&client, .{ + .url = "https://ads.example.com/pixel.gif", + .document = "https://" ++ "a" ** 254 ++ ".com/", + .resource_type = .image, + })); + // Same for a URL too long to normalize (uppercase forces the copy). + try testing.expect(!testIsUrlBlocked(&client, .{ + .url = "https://ads.example.com/" ++ "A" ** (8 * 1024), + .document = "https://news.com/", + .resource_type = .image, + })); } test "HttpClient: URL blocking exempts internal transfers" { @@ -4518,21 +4554,21 @@ test "HttpClient: Fetch header overrides restore after one hop" { test "HttpClient: Owner.siteForCookies" { var top_url: [:0]const u8 = "http://attacker.example/attacker-nested"; - var top = testOwner(); + var top = testOwner(null, null); top.url = &top_url; var middle_url: [:0]const u8 = "http://victim.example/nested-middle"; - var middle = testOwner(); + var middle = testOwner(null, null); middle.url = &middle_url; middle.parent = ⊤ var inner_url: [:0]const u8 = "http://victim.example/inner"; - var inner = testOwner(); + var inner = testOwner(null, null); inner.url = &inner_url; inner.parent = &middle; // A worker has no site of its own; it takes its creating document's. - var worker = testOwner(); + var worker = testOwner(null, null); worker.parent = &inner; // A top-level document is its own site. @@ -4585,7 +4621,7 @@ test "HttpClient: fulfillIntercepted survives a done_callback that tears down th defer client.processGraveyard(); defer client.transfers.deinit(testing.allocator); - var owner = testOwner(); + var owner = testOwner(null, null); const Ctx = struct { client: *Client, @@ -4658,7 +4694,7 @@ test "HttpClient: kill during done_callback does not also fire shutdown_callback defer client.processGraveyard(); defer client.transfers.deinit(testing.allocator); - var owner = testOwner(); + var owner = testOwner(null, null); const Ctx = struct { client: *Client, @@ -4738,7 +4774,7 @@ test "HttpClient: kill during a non-terminal callback defers shutdown_callback" defer client.processGraveyard(); defer client.transfers.deinit(testing.allocator); - var owner = testOwner(); + var owner = testOwner(null, null); const Ctx = struct { client: *Client, @@ -5053,7 +5089,7 @@ test "HttpClient: abortParked survives an error_callback that tears down the own defer client.processGraveyard(); defer client.transfers.deinit(testing.allocator); - var owner = testOwner(); + var owner = testOwner(null, null); const Ctx = struct { client: *Client, @@ -5127,7 +5163,7 @@ test "HttpClient: abort survives an error_callback that tears down the owner" { defer client.processGraveyard(); defer client.transfers.deinit(testing.allocator); - var owner = testOwner(); + var owner = testOwner(null, null); const Ctx = struct { client: *Client, diff --git a/src/network/adblock/Engine.zig b/src/network/adblock/Engine.zig index 6b1cd800e..c92f7c146 100644 --- a/src/network/adblock/Engine.zig +++ b/src/network/adblock/Engine.zig @@ -34,6 +34,9 @@ const std = @import("std"); +const URL = @import("../../browser/URL.zig"); +const HttpClient = @import("../HttpClient.zig"); + const domain = @import("domain.zig"); const pattern = @import("pattern.zig"); const NetworkFilter = @import("NetworkFilter.zig"); @@ -69,6 +72,25 @@ pub const Request = struct { /// Exactly one bit set. kind: NetworkFilter.ResourceTypes, third_party: bool, + /// The URL's tokens, hashed once here so no engine retokenizes. A URL + /// long enough to overflow loses its last tokens. + tokens_buf: [URL_TOKENS_MAX]u32, + tokens_len: usize, + + const URL_TOKENS_MAX = 128; + + /// Longest URL `fromHttp` will normalize. Anything past this is not a + /// resource a filter list has an opinion about. + const URL_MAX = 8 * 1024; + /// DNS's own hostname limit. + const SOURCE_MAX = 253; + + /// Backs the normalized text of a `fromHttp` request, which stays valid + /// only as long as the buffers do. + pub const Buffers = struct { + url: [URL_MAX]u8, + source: [SOURCE_MAX]u8, + }; /// `url` must already be lowercased and fragment-free; `source_hostname` /// may be empty when there is no document context. @@ -79,11 +101,77 @@ pub const Request = struct { ) Request { const parsed: pattern.Url = .init(url); const source = if (source_hostname.len == 0) parsed.hostname() else source_hostname; - return .{ + var request: Request = .{ .url = parsed, .source_hostname = source, .kind = kind, .third_party = domain.isThirdParty(parsed.hostname(), source), + .tokens_buf = undefined, + .tokens_len = 0, + }; + var it: Tokens = .{ .text = url }; + while (it.next()) |token| { + if (request.tokens_len == request.tokens_buf.len) break; + request.tokens_buf[request.tokens_len] = token; + request.tokens_len += 1; + } + return request; + } + + /// Builds the request `req` is matched as, `source` being the URL (or + /// origin serialization) of the document it was issued from. null when + /// there is none. + pub fn fromHttp( + req: *const HttpClient.Request, + source_url: ?[]const u8, + buffers: *Buffers, + ) ?Request { + const url = normalizeUrl(req.url, &buffers.url) orelse return null; + + // Matching top-level nav against the page it was clicked on would make + // every link third party. + const top_level = req.resource_type == .document and !req.is_subframe; + const source_host = if (top_level) + "" + else if (source_url) |u| + URL.getOriginHostname(u) + else + ""; + if (source_host.len > buffers.source.len) return null; + const source = std.ascii.lowerString(&buffers.source, source_host); + + return .init(url, source, resourceType(req)); + } + + inline fn tokens(self: *const Request) []const u32 { + return self.tokens_buf[0..self.tokens_len]; + } + + /// Lowercases `url` into `buf` with its fragment stripped; patterns are + /// stored lowercased, and no request URL carries a fragment onto the wire. + fn normalizeUrl(url: []const u8, buf: []u8) ?[]const u8 { + const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len; + const trimmed = url[0..end]; + + const upper = for (trimmed, 0..) |c, i| { + if (std.ascii.isUpper(c)) break i; + } else return trimmed; + + if (trimmed.len > buf.len) return null; + const out = buf[0..trimmed.len]; + @memcpy(out[0..upper], trimmed[0..upper]); + _ = std.ascii.lowerString(out[upper..], trimmed[upper..]); + return out; + } + + fn resourceType(req: *const HttpClient.Request) NetworkFilter.ResourceTypes { + return switch (req.resource_type) { + .document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true }, + .script => .{ .script = true }, + .stylesheet => .{ .stylesheet = true }, + .xhr, .fetch => .{ .xmlhttprequest = true }, + .image => .{ .image = true }, + .eventsource => .{ .other = true }, }; } }; @@ -99,8 +187,7 @@ const TOKENS_MAX = 32; /// The first filter that matches, or null. pub fn match(self: *const Engine, request: Request) ?*const NetworkFilter { if (self.filters.len != 0) { - var it: Tokens = .{ .text = request.url.text }; - while (it.next()) |token| { + for (request.tokens()) |token| { const bucket = self.buckets.get(token) orelse continue; if (self.matchIn(bucket, request)) |filter| return filter; }