mirror of
https://github.com/lightpanda-io/browser.git
synced 2026-09-14 15:03:27 -04:00
Adblocker: changes on request construction & tokenization
* Engine.Request.fromHttp(req, source_url, buffers) now builds the adblock request straight from HttpClient.Request. * The URL is tokenized once per request (hashed into the Request, shared by all engines); capped at 128 tokens (same as adblock-rust). * Document hostname longer than 253 bytes now skips adblocking.
This commit is contained in:
2 files changed
+198
-75
No files matched your search
+108
-72
@@ -398,59 +398,32 @@ fn clearUrlBlocklist(self: *Client) void {
|
||||
/// Every reason a request is refused before it reaches the network:
|
||||
/// `--block-urls` patterns and the `--adblock-lists` filters both land here
|
||||
/// so that no call site can apply one without the other.
|
||||
fn isUrlBlocked(self: *const Client, req: *const Request) bool {
|
||||
fn isUrlBlocked(self: *const Client, transfer: *const Transfer) bool {
|
||||
const req = &transfer.req;
|
||||
if (req.internal) return false;
|
||||
if (self.url_blocklist) |*blocklist| {
|
||||
if (blocklist.isBlocked(req.url)) return true;
|
||||
}
|
||||
return self.isAdblocked(req);
|
||||
return self.isAdblocked(transfer);
|
||||
}
|
||||
|
||||
/// Longest URL we will match filters against.
|
||||
/// Anything past this is not a resource a filter list has an opinion about.
|
||||
const ADBLOCK_URL_MAX = 8 * 1024;
|
||||
|
||||
fn isAdblocked(self: *const Client, req: *const Request) bool {
|
||||
fn isAdblocked(self: *const Client, transfer: *const Transfer) bool {
|
||||
const blocker = if (self.network.adblocker) |*b| b else return false;
|
||||
|
||||
var url_buf: [ADBLOCK_URL_MAX]u8 = undefined;
|
||||
const url = normalizeForAdblock(req.url, &url_buf) orelse return false;
|
||||
|
||||
var source_buf: [253]u8 = undefined;
|
||||
const top_level = req.resource_type == .document and !req.is_subframe;
|
||||
const source_host = if (top_level) "" else URL.getOriginHostname(req.cookie_origin);
|
||||
const source = if (source_host.len == 0 or source_host.len > source_buf.len)
|
||||
""
|
||||
else
|
||||
std.ascii.lowerString(&source_buf, source_host);
|
||||
|
||||
return blocker.match(.init(url, source, adblockResourceType(req))) == .blocked;
|
||||
var buffers: AdBlocker.Request.Buffers = undefined;
|
||||
const target = AdBlocker.Request.fromHttp(
|
||||
&transfer.req,
|
||||
adblockSourceUrl(transfer),
|
||||
&buffers,
|
||||
) orelse return false;
|
||||
return blocker.match(target) == .blocked;
|
||||
}
|
||||
|
||||
fn normalizeForAdblock(url: [:0]const u8, buf: []u8) ?[]const u8 {
|
||||
const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len;
|
||||
const trimmed = url[0..end];
|
||||
|
||||
const upper = for (trimmed, 0..) |c, i| {
|
||||
if (std.ascii.isUpper(c)) break i;
|
||||
} else return trimmed;
|
||||
|
||||
if (trimmed.len > buf.len) return null;
|
||||
const out = buf[0..trimmed.len];
|
||||
@memcpy(out[0..upper], trimmed[0..upper]);
|
||||
_ = std.ascii.lowerString(out[upper..], trimmed[upper..]);
|
||||
return out;
|
||||
}
|
||||
|
||||
fn adblockResourceType(req: *const Request) AdBlocker.ResourceTypes {
|
||||
return switch (req.resource_type) {
|
||||
.document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true },
|
||||
.script => .{ .script = true },
|
||||
.stylesheet => .{ .stylesheet = true },
|
||||
.xhr, .fetch => .{ .xmlhttprequest = true },
|
||||
.image => .{ .image = true },
|
||||
.eventsource => .{ .other = true },
|
||||
};
|
||||
fn adblockSourceUrl(transfer: *const Transfer) ?[]const u8 {
|
||||
var owner: *const Owner = transfer.owner orelse return null;
|
||||
if (transfer.req.resource_type == .document) {
|
||||
owner = owner.parent orelse return null;
|
||||
}
|
||||
return owner.documentUrl();
|
||||
}
|
||||
|
||||
fn isCrossOriginModeAllowed(transfer: *const Transfer) bool {
|
||||
@@ -1063,7 +1036,7 @@ fn pipeline(self: *Client, transfer: *Transfer, from: SubmitFrom) !void {
|
||||
continue :sw SubmitFrom.after_intercept;
|
||||
},
|
||||
.after_intercept => {
|
||||
if (self.isUrlBlocked(&transfer.req)) {
|
||||
if (self.isUrlBlocked(transfer)) {
|
||||
log.info(.http, "blocked url", .{ .url = transfer.req.url });
|
||||
return transfer.failAsync(error.UrlBlocked);
|
||||
}
|
||||
@@ -2147,6 +2120,18 @@ pub const Owner = struct {
|
||||
|
||||
const Blob = @import("../browser/webapi/Blob.zig");
|
||||
|
||||
/// The URL of the document this owner's requests belong to.
|
||||
/// Handles `about:` case also.
|
||||
pub fn documentUrl(self: *const Owner) ?[:0]const u8 {
|
||||
var source = self;
|
||||
while (true) {
|
||||
if (source.url) |url| {
|
||||
if (!std.mem.startsWith(u8, url.*, "about:")) return url.*;
|
||||
}
|
||||
source = source.parent orelse return null;
|
||||
}
|
||||
}
|
||||
|
||||
// RFC 6265bis "site for cookies"
|
||||
pub fn siteForCookies(self: *const Owner) Cookie.SiteForCookies {
|
||||
var source = self;
|
||||
@@ -4031,14 +4016,14 @@ const Synthetic = struct {
|
||||
|
||||
const testing = @import("../testing.zig");
|
||||
|
||||
// Only the transfer list matters to the tests using it: they build their
|
||||
// transfers by hand and never go through newRequest.
|
||||
fn testOwner() Owner {
|
||||
// Only the transfer list, the url and the parent matter to the tests using
|
||||
// it: they build their transfers by hand and never go through newRequest.
|
||||
fn testOwner(url: ?*const [:0]const u8, parent: ?*const Owner) Owner {
|
||||
return .{
|
||||
.blob_urls = undefined,
|
||||
.origin = undefined,
|
||||
.url = null,
|
||||
.parent = null,
|
||||
.url = url,
|
||||
.parent = parent,
|
||||
.frame_id = 0,
|
||||
.document_frame_id = 0,
|
||||
.loader_id = 0,
|
||||
@@ -4261,26 +4246,48 @@ test "HttpClient: setBlockedUrls owns, replaces, and clears patterns" {
|
||||
const TestRequest = struct {
|
||||
url: [:0]const u8,
|
||||
document: [:0]const u8 = "",
|
||||
/// The page embedding `document`, when the test wants a deeper chain.
|
||||
parent_document: [:0]const u8 = "",
|
||||
resource_type: Request.ResourceType = .document,
|
||||
is_subframe: bool = false,
|
||||
internal: bool = false,
|
||||
};
|
||||
|
||||
fn testIsUrlBlocked(client: *const Client, opts: TestRequest) bool {
|
||||
const req: Request = .{
|
||||
.frame_id = 0,
|
||||
.loader_id = 0,
|
||||
.method = .GET,
|
||||
.url = opts.url,
|
||||
.cookie_jar = null,
|
||||
.cookie_origin = opts.document,
|
||||
.resource_type = opts.resource_type,
|
||||
.is_subframe = opts.is_subframe,
|
||||
.internal = opts.internal,
|
||||
.notification = undefined,
|
||||
.shutdown_callback = noopShutdown,
|
||||
// The owner chain a real request carries: [0] the embedding page,
|
||||
// [1] the request's document, [2] the frame being navigated — whose url
|
||||
// slot already holds the target, so its context is its parent's.
|
||||
var chain: [3]Owner = undefined;
|
||||
chain[0] = testOwner(
|
||||
if (opts.parent_document.len == 0) null else &opts.parent_document,
|
||||
null,
|
||||
);
|
||||
chain[1] = testOwner(
|
||||
if (opts.document.len == 0) null else &opts.document,
|
||||
if (opts.parent_document.len == 0) null else &chain[0],
|
||||
);
|
||||
chain[2] = testOwner(&opts.url, if (opts.document.len == 0) null else &chain[1]);
|
||||
|
||||
var transfer: Transfer = .{
|
||||
.arena = undefined,
|
||||
.owner = if (opts.resource_type == .document)
|
||||
&chain[2]
|
||||
else if (opts.document.len == 0)
|
||||
null
|
||||
else
|
||||
&chain[1],
|
||||
.req = .{
|
||||
.method = .GET,
|
||||
.url = opts.url,
|
||||
.resource_type = opts.resource_type,
|
||||
.is_subframe = opts.is_subframe,
|
||||
.internal = opts.internal,
|
||||
.shutdown_callback = noopShutdown,
|
||||
},
|
||||
.client = undefined,
|
||||
.start_time = 0,
|
||||
};
|
||||
return client.isUrlBlocked(&req);
|
||||
return client.isUrlBlocked(&transfer);
|
||||
}
|
||||
|
||||
test "HttpClient: adblock verdicts apply per request" {
|
||||
@@ -4354,6 +4361,35 @@ test "HttpClient: adblock verdicts apply per request" {
|
||||
.document = "https://news.com/",
|
||||
.is_subframe = true,
|
||||
}));
|
||||
|
||||
// The context is the issuing frame's document even under a cross-site
|
||||
// ancestor (the canonical ad iframe) — site-for-cookies semantics would
|
||||
// collapse this chain to nothing and lose the party.
|
||||
try testing.expect(testIsUrlBlocked(&client, .{
|
||||
.url = "https://partied.example.com/x.js",
|
||||
.document = "https://adprovider.com/frame.html",
|
||||
.parent_document = "https://news.com/",
|
||||
.resource_type = .script,
|
||||
}));
|
||||
|
||||
// A document hostname DNS could not carry is nothing a filter list has
|
||||
// an opinion about: the request is let through, not matched sourceless.
|
||||
try testing.expect(testIsUrlBlocked(&client, .{
|
||||
.url = "https://ads.example.com/pixel.gif",
|
||||
.document = "https://news.com/",
|
||||
.resource_type = .image,
|
||||
}));
|
||||
try testing.expect(!testIsUrlBlocked(&client, .{
|
||||
.url = "https://ads.example.com/pixel.gif",
|
||||
.document = "https://" ++ "a" ** 254 ++ ".com/",
|
||||
.resource_type = .image,
|
||||
}));
|
||||
// Same for a URL too long to normalize (uppercase forces the copy).
|
||||
try testing.expect(!testIsUrlBlocked(&client, .{
|
||||
.url = "https://ads.example.com/" ++ "A" ** (8 * 1024),
|
||||
.document = "https://news.com/",
|
||||
.resource_type = .image,
|
||||
}));
|
||||
}
|
||||
|
||||
test "HttpClient: URL blocking exempts internal transfers" {
|
||||
@@ -4518,21 +4554,21 @@ test "HttpClient: Fetch header overrides restore after one hop" {
|
||||
|
||||
test "HttpClient: Owner.siteForCookies" {
|
||||
var top_url: [:0]const u8 = "http://attacker.example/attacker-nested";
|
||||
var top = testOwner();
|
||||
var top = testOwner(null, null);
|
||||
top.url = &top_url;
|
||||
|
||||
var middle_url: [:0]const u8 = "http://victim.example/nested-middle";
|
||||
var middle = testOwner();
|
||||
var middle = testOwner(null, null);
|
||||
middle.url = &middle_url;
|
||||
middle.parent = ⊤
|
||||
|
||||
var inner_url: [:0]const u8 = "http://victim.example/inner";
|
||||
var inner = testOwner();
|
||||
var inner = testOwner(null, null);
|
||||
inner.url = &inner_url;
|
||||
inner.parent = &middle;
|
||||
|
||||
// A worker has no site of its own; it takes its creating document's.
|
||||
var worker = testOwner();
|
||||
var worker = testOwner(null, null);
|
||||
worker.parent = &inner;
|
||||
|
||||
// A top-level document is its own site.
|
||||
@@ -4585,7 +4621,7 @@ test "HttpClient: fulfillIntercepted survives a done_callback that tears down th
|
||||
defer client.processGraveyard();
|
||||
defer client.transfers.deinit(testing.allocator);
|
||||
|
||||
var owner = testOwner();
|
||||
var owner = testOwner(null, null);
|
||||
|
||||
const Ctx = struct {
|
||||
client: *Client,
|
||||
@@ -4658,7 +4694,7 @@ test "HttpClient: kill during done_callback does not also fire shutdown_callback
|
||||
defer client.processGraveyard();
|
||||
defer client.transfers.deinit(testing.allocator);
|
||||
|
||||
var owner = testOwner();
|
||||
var owner = testOwner(null, null);
|
||||
|
||||
const Ctx = struct {
|
||||
client: *Client,
|
||||
@@ -4738,7 +4774,7 @@ test "HttpClient: kill during a non-terminal callback defers shutdown_callback"
|
||||
defer client.processGraveyard();
|
||||
defer client.transfers.deinit(testing.allocator);
|
||||
|
||||
var owner = testOwner();
|
||||
var owner = testOwner(null, null);
|
||||
|
||||
const Ctx = struct {
|
||||
client: *Client,
|
||||
@@ -5053,7 +5089,7 @@ test "HttpClient: abortParked survives an error_callback that tears down the own
|
||||
defer client.processGraveyard();
|
||||
defer client.transfers.deinit(testing.allocator);
|
||||
|
||||
var owner = testOwner();
|
||||
var owner = testOwner(null, null);
|
||||
|
||||
const Ctx = struct {
|
||||
client: *Client,
|
||||
@@ -5127,7 +5163,7 @@ test "HttpClient: abort survives an error_callback that tears down the owner" {
|
||||
defer client.processGraveyard();
|
||||
defer client.transfers.deinit(testing.allocator);
|
||||
|
||||
var owner = testOwner();
|
||||
var owner = testOwner(null, null);
|
||||
|
||||
const Ctx = struct {
|
||||
client: *Client,
|
||||
|
||||
@@ -34,6 +34,9 @@
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
const URL = @import("../../browser/URL.zig");
|
||||
const HttpClient = @import("../HttpClient.zig");
|
||||
|
||||
const domain = @import("domain.zig");
|
||||
const pattern = @import("pattern.zig");
|
||||
const NetworkFilter = @import("NetworkFilter.zig");
|
||||
@@ -69,6 +72,25 @@ pub const Request = struct {
|
||||
/// Exactly one bit set.
|
||||
kind: NetworkFilter.ResourceTypes,
|
||||
third_party: bool,
|
||||
/// The URL's tokens, hashed once here so no engine retokenizes. A URL
|
||||
/// long enough to overflow loses its last tokens.
|
||||
tokens_buf: [URL_TOKENS_MAX]u32,
|
||||
tokens_len: usize,
|
||||
|
||||
const URL_TOKENS_MAX = 128;
|
||||
|
||||
/// Longest URL `fromHttp` will normalize. Anything past this is not a
|
||||
/// resource a filter list has an opinion about.
|
||||
const URL_MAX = 8 * 1024;
|
||||
/// DNS's own hostname limit.
|
||||
const SOURCE_MAX = 253;
|
||||
|
||||
/// Backs the normalized text of a `fromHttp` request, which stays valid
|
||||
/// only as long as the buffers do.
|
||||
pub const Buffers = struct {
|
||||
url: [URL_MAX]u8,
|
||||
source: [SOURCE_MAX]u8,
|
||||
};
|
||||
|
||||
/// `url` must already be lowercased and fragment-free; `source_hostname`
|
||||
/// may be empty when there is no document context.
|
||||
@@ -79,11 +101,77 @@ pub const Request = struct {
|
||||
) Request {
|
||||
const parsed: pattern.Url = .init(url);
|
||||
const source = if (source_hostname.len == 0) parsed.hostname() else source_hostname;
|
||||
return .{
|
||||
var request: Request = .{
|
||||
.url = parsed,
|
||||
.source_hostname = source,
|
||||
.kind = kind,
|
||||
.third_party = domain.isThirdParty(parsed.hostname(), source),
|
||||
.tokens_buf = undefined,
|
||||
.tokens_len = 0,
|
||||
};
|
||||
var it: Tokens = .{ .text = url };
|
||||
while (it.next()) |token| {
|
||||
if (request.tokens_len == request.tokens_buf.len) break;
|
||||
request.tokens_buf[request.tokens_len] = token;
|
||||
request.tokens_len += 1;
|
||||
}
|
||||
return request;
|
||||
}
|
||||
|
||||
/// Builds the request `req` is matched as, `source` being the URL (or
|
||||
/// origin serialization) of the document it was issued from. null when
|
||||
/// there is none.
|
||||
pub fn fromHttp(
|
||||
req: *const HttpClient.Request,
|
||||
source_url: ?[]const u8,
|
||||
buffers: *Buffers,
|
||||
) ?Request {
|
||||
const url = normalizeUrl(req.url, &buffers.url) orelse return null;
|
||||
|
||||
// Matching top-level nav against the page it was clicked on would make
|
||||
// every link third party.
|
||||
const top_level = req.resource_type == .document and !req.is_subframe;
|
||||
const source_host = if (top_level)
|
||||
""
|
||||
else if (source_url) |u|
|
||||
URL.getOriginHostname(u)
|
||||
else
|
||||
"";
|
||||
if (source_host.len > buffers.source.len) return null;
|
||||
const source = std.ascii.lowerString(&buffers.source, source_host);
|
||||
|
||||
return .init(url, source, resourceType(req));
|
||||
}
|
||||
|
||||
inline fn tokens(self: *const Request) []const u32 {
|
||||
return self.tokens_buf[0..self.tokens_len];
|
||||
}
|
||||
|
||||
/// Lowercases `url` into `buf` with its fragment stripped; patterns are
|
||||
/// stored lowercased, and no request URL carries a fragment onto the wire.
|
||||
fn normalizeUrl(url: []const u8, buf: []u8) ?[]const u8 {
|
||||
const end = std.mem.indexOfScalar(u8, url, '#') orelse url.len;
|
||||
const trimmed = url[0..end];
|
||||
|
||||
const upper = for (trimmed, 0..) |c, i| {
|
||||
if (std.ascii.isUpper(c)) break i;
|
||||
} else return trimmed;
|
||||
|
||||
if (trimmed.len > buf.len) return null;
|
||||
const out = buf[0..trimmed.len];
|
||||
@memcpy(out[0..upper], trimmed[0..upper]);
|
||||
_ = std.ascii.lowerString(out[upper..], trimmed[upper..]);
|
||||
return out;
|
||||
}
|
||||
|
||||
fn resourceType(req: *const HttpClient.Request) NetworkFilter.ResourceTypes {
|
||||
return switch (req.resource_type) {
|
||||
.document => if (req.is_subframe) .{ .subdocument = true } else .{ .document = true },
|
||||
.script => .{ .script = true },
|
||||
.stylesheet => .{ .stylesheet = true },
|
||||
.xhr, .fetch => .{ .xmlhttprequest = true },
|
||||
.image => .{ .image = true },
|
||||
.eventsource => .{ .other = true },
|
||||
};
|
||||
}
|
||||
};
|
||||
@@ -99,8 +187,7 @@ const TOKENS_MAX = 32;
|
||||
/// The first filter that matches, or null.
|
||||
pub fn match(self: *const Engine, request: Request) ?*const NetworkFilter {
|
||||
if (self.filters.len != 0) {
|
||||
var it: Tokens = .{ .text = request.url.text };
|
||||
while (it.next()) |token| {
|
||||
for (request.tokens()) |token| {
|
||||
const bucket = self.buckets.get(token) orelse continue;
|
||||
if (self.matchIn(bucket, request)) |filter| return filter;
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user