From 06f7f6f29579e908d7f136809e54bb44feb97582 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A0=20Arrufat?= Date: Thu, 10 Sep 2026 11:25:20 +0200 Subject: [PATCH] `Engine`: look through an optional stretch when reading a branch's ends An alternation or a repeat keeps only whether its text may start and end with a token character, and read that off one marker. An optional non-token stretch there (`\/?x`) was taken as a definite non-token, so `\/ads(\/?x|\/y)` was filed under "ads" while `/adsx` carries no such token. What follows the stretch answers now. --- src/network/adblock/AdBlocker.zig | 13 +++++++++++++ src/network/adblock/Engine.zig | 20 ++++++++++++++++++-- 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/src/network/adblock/AdBlocker.zig b/src/network/adblock/AdBlocker.zig index 4ddb4f50a..9699783aa 100644 --- a/src/network/adblock/AdBlocker.zig +++ b/src/network/adblock/AdBlocker.zig @@ -665,6 +665,19 @@ test "adblock.AdBlocker: the regex rules from EasyList" { try expectVerdict(&blocker, .none, "https://x.com/ABCD.js", "x.com", script); } +test "adblock.AdBlocker: a regex is found under every token it may match" { + var blocker: AdBlocker = try .init(testing.allocator); + defer blocker.deinit(); + + try testLoad(&blocker, + \\/\/ads(\/?x|\/y)/$script + ); + try expectVerdict(&blocker, .blocked, "https://example.com/adsx", "example.com", script); + try expectVerdict(&blocker, .blocked, "https://example.com/ads/x", "example.com", script); + try expectVerdict(&blocker, .blocked, "https://example.com/ads/y", "example.com", script); + try expectVerdict(&blocker, .none, "https://example.com/ads/z", "example.com", script); +} + test "adblock.AdBlocker: $badfilter removes a regex rule" { var blocker: AdBlocker = try .init(testing.allocator); defer blocker.deinit(); diff --git a/src/network/adblock/Engine.zig b/src/network/adblock/Engine.zig index 14fe401e4..ff19e3684 100644 --- a/src/network/adblock/Engine.zig +++ b/src/network/adblock/Engine.zig @@ -637,12 +637,25 @@ const RegexShape = struct { return c == maybe_token or c == maybe_token_optional or isTokenChar(c); } + /// Whether what `s` matches could start with a token character. An + /// optional non-token stretch may be absent, so whatever follows it + /// answers instead. fn startsTokenish(s: []const u8) bool { - return s.len != 0 and isTokenish(s[0]); + for (s) |c| { + if (isTokenish(c)) return true; + if (c != not_token_optional) return false; + } + return false; } fn endsTokenish(s: []const u8) bool { - return s.len != 0 and isTokenish(s[s.len - 1]); + var i = s.len; + while (i > 0) { + i -= 1; + if (isTokenish(s[i])) return true; + if (s[i] != not_token_optional) return false; + } + return false; } }; @@ -693,6 +706,9 @@ test "adblock.Engine: regex filters yield the tokens every match carries" { // An optional stretch may vanish and glue its neighbours: "adsbanner". tokens = try tokensOf(arena, "/\\/ads\\/?banner\\//", &buf); try testing.expectEqual(0, tokens.len); + // ... also from inside a group: "adsx" is a match. + tokens = try tokensOf(arena, "/\\/ads(\\/?x|\\/y)/", &buf); + try testing.expectEqual(0, tokens.len); // A repeat is not the literal it repeats. tokens = try tokensOf(arena, "/\\/ab+c\\//", &buf); try testing.expectEqual(0, tokens.len);