diff --git a/src/App.zig b/src/App.zig
index fe5a0effa..3d573294b 100644
--- a/src/App.zig
+++ b/src/App.zig
@@ -20,6 +20,7 @@ const std = @import("std");
const lp = @import("lightpanda");
const Config = @import("Config.zig");
+const Regex = @import("Regex.zig");
const Snapshot = @import("browser/js/Snapshot.zig");
const Platform = @import("browser/js/Platform.zig");
const Telemetry = @import("telemetry/telemetry.zig").Telemetry;
@@ -43,6 +44,8 @@ allocator: Allocator,
arena_pool: ArenaPool,
app_dir_path: ?[]const u8,
+regex_context: *Regex.Context,
+
pub fn init(allocator: Allocator, config: *const Config) !*App {
const platform = try Platform.init(.{
.v8_flags = config.v8Flags(),
@@ -54,6 +57,9 @@ pub fn init(allocator: Allocator, config: *const Config) !*App {
const snapshot = try Snapshot.load();
errdefer snapshot.deinit();
+ const regex_context: *Regex.Context = try .init(allocator);
+ errdefer regex_context.deinit();
+
const app = try allocator.create(App);
errdefer allocator.destroy(app);
@@ -62,6 +68,7 @@ pub fn init(allocator: Allocator, config: *const Config) !*App {
.allocator = allocator,
.platform = platform,
.snapshot = snapshot,
+ .regex_context = regex_context,
.network = undefined,
.app_dir_path = undefined,
.telemetry = undefined,
@@ -96,6 +103,8 @@ pub fn deinit(self: *App) void {
}
self.telemetry.deinit(allocator);
self.network.deinit();
+ // After `network`: its adblock regexes free through this context.
+ self.regex_context.deinit();
self.snapshot.deinit();
self.platform.deinit();
self.arena_pool.deinit();
diff --git a/src/network/adblock/Regex.zig b/src/Regex.zig
similarity index 52%
rename from src/network/adblock/Regex.zig
rename to src/Regex.zig
index 76806f15d..a5c567f60 100644
--- a/src/network/adblock/Regex.zig
+++ b/src/Regex.zig
@@ -16,23 +16,18 @@
// You should have received a copy of the GNU Affero General Public License
// along with this program. If not, see .
-//! A compiled `/regex/` filter body. Filter lists write them in JavaScript
-//! `RegExp` syntax and uBO runs them with `new RegExp(src, 'i')` against the
-//! raw request URL (no flag under `$match-case`); PCRE2 reads that syntax
-//! as-is, escapes like `\/` included.
+//! A pattern in JavaScript `RegExp` syntax, run by PCRE2, which reads that
+//! syntax as-is, escapes like `\/` included.
//!
//! A compiled pattern and its `Context` are never modified after `compile`,
-//! so one `Regex` can be shared by every HTTP client thread; the per-call
-//! match data is what PCRE2 requires to be private.
+//! so one `Regex` can be shared by every thread; the per-call match data is
+//! what PCRE2 requires to be private.
const std = @import("std");
-const lp = @import("lightpanda");
const pcre2 = @import("pcre2");
const Allocator = std.mem.Allocator;
-const log = lp.log;
-
const Regex = @This();
code: *pcre2.pcre2_code_8,
@@ -40,21 +35,42 @@ context: *const Context,
pub const Error = error{ InvalidRegex, OutOfMemory };
-/// What every `Regex` compiled through it shares: the allocator PCRE2 draws
-/// from, and the compile and match settings. Outlives the regexes.
+pub const Options = struct {
+ case_insensitive: bool = false,
+ /// UTF-8 aware matching: `.` consumes a code point and caseless folding
+ /// works beyond ASCII. An invalid sequence in the subject fails to match
+ /// rather than erroring. `\b` and `\w` stay ASCII, as in JavaScript.
+ unicode: bool = false,
+ /// JavaScript's `s`.
+ dot_all: bool = false,
+ /// JavaScript's `m`.
+ multiline: bool = false,
+};
+
+pub const Diagnostic = struct {
+ offset: usize = 0,
+ len: usize = 0,
+ buf: [256]u8 = undefined,
+
+ pub fn message(self: *const Diagnostic) []const u8 {
+ return self.buf[0..self.len];
+ }
+};
+
+/// Shared by every `Regex` compiled through it; outlives them.
///
-/// PCRE2 would happily use libc's malloc; it is handed the blocker's
-/// allocator so that a compiled pattern nobody freed fails a test the way
-/// any other leak does.
+/// PCRE2 would happily use libc's malloc; it is handed the owner's allocator
+/// so that a compiled pattern nobody freed fails a test the way any other
+/// leak does.
pub const Context = struct {
allocator: Allocator,
general: *pcre2.pcre2_general_context_8,
compile_context: *pcre2.pcre2_compile_context_8,
match_context: *pcre2.pcre2_match_context_8,
- // A pattern from a list that backtracks this much on one URL is broken,
- // not slow; giving up costs a false negative on that request, nothing
- // more.
+ // Patterns come from lists and prompts, never from code: one that
+ // backtracks this much on a subject is broken, not slow, and giving up
+ // costs a false negative on that subject, nothing more.
const MATCH_LIMIT = 100_000;
const DEPTH_LIMIT = 10_000;
@@ -71,8 +87,7 @@ pub const Context = struct {
const compile_context = pcre2.pcre2_compile_context_create_8(general) orelse return error.OutOfMemory;
errdefer pcre2.pcre2_compile_context_free_8(compile_context);
- // JavaScript without the `u` flag reads an unknown escape as the
- // literal character, and that is the mode uBO compiles filters in.
+ // JavaScript reads an unknown escape as the literal character.
_ = pcre2.pcre2_set_compile_extra_options_8(compile_context, pcre2.PCRE2_EXTRA_BAD_ESCAPE_IS_LITERAL);
const match_context = pcre2.pcre2_match_context_create_8(general) orelse return error.OutOfMemory;
@@ -85,6 +100,37 @@ pub const Context = struct {
return self;
}
+ /// A failed compile fills `diag`, when given, with PCRE2's message and the
+ /// offset of the offending character.
+ pub fn compile(self: *const Context, pattern: []const u8, options: Options, diag: ?*Diagnostic) Error!Regex {
+ var flags: u32 = 0;
+ if (options.case_insensitive) flags |= pcre2.PCRE2_CASELESS;
+ if (options.unicode) flags |= pcre2.PCRE2_UTF | pcre2.PCRE2_MATCH_INVALID_UTF;
+ if (options.dot_all) flags |= pcre2.PCRE2_DOTALL;
+ if (options.multiline) flags |= pcre2.PCRE2_MULTILINE;
+
+ var err_code: c_int = 0;
+ var err_offset: usize = 0;
+ const code = pcre2.pcre2_compile_8(
+ pattern.ptr,
+ pattern.len,
+ flags,
+ &err_code,
+ &err_offset,
+ self.compile_context,
+ ) orelse {
+ // A failed allocation is ours, not the pattern's.
+ if (err_code == pcre2.PCRE2_ERROR_HEAP_FAILED) return error.OutOfMemory;
+ if (diag) |d| {
+ const len = pcre2.pcre2_get_error_message_8(err_code, &d.buf, d.buf.len);
+ d.len = if (len < 0) 0 else @intCast(len);
+ d.offset = err_offset;
+ }
+ return error.InvalidRegex;
+ };
+ return .{ .code = code, .context = self };
+ }
+
pub fn deinit(self: *Context) void {
pcre2.pcre2_match_context_free_8(self.match_context);
pcre2.pcre2_compile_context_free_8(self.compile_context);
@@ -114,33 +160,6 @@ pub const Context = struct {
}
};
-pub fn compile(context: *const Context, pattern: []const u8, case_insensitive: bool) Error!Regex {
- const options: u32 = if (case_insensitive) pcre2.PCRE2_CASELESS else 0;
- var err_code: c_int = 0;
- var err_offset: usize = 0;
- const code = pcre2.pcre2_compile_8(
- pattern.ptr,
- pattern.len,
- options,
- &err_code,
- &err_offset,
- context.compile_context,
- ) orelse {
- // A failed allocation is ours, not the pattern's.
- if (err_code == pcre2.PCRE2_ERROR_HEAP_FAILED) return error.OutOfMemory;
- var buf: [256]u8 = undefined;
- const len = pcre2.pcre2_get_error_message_8(err_code, &buf, buf.len);
- const message: []const u8 = if (len < 0) "unknown error" else buf[0..@intCast(len)];
- log.debug(.app, "adblock regex rejected", .{
- .pattern = pattern,
- .err = message,
- .offset = err_offset,
- });
- return error.InvalidRegex;
- };
- return .{ .code = code, .context = context };
-}
-
pub fn deinit(self: Regex) void {
pcre2.pcre2_code_free_8(self.code);
}
@@ -152,7 +171,6 @@ const MATCH_SCRATCH = 24 * 1024;
/// Whether the pattern matches anywhere in `text`, as `RegExp.test` would
/// answer. A match that hits the backtracking limits counts as no match.
pub fn matches(self: Regex, text: []const u8) bool {
- // This runs per request; the scratch keeps the common case off the heap.
var scratch = std.heap.stackFallback(MATCH_SCRATCH, self.context.allocator);
var allocator = scratch.get();
const general = pcre2.pcre2_general_context_create_8(Context.cMalloc, Context.cFree, &allocator) orelse return false;
@@ -167,13 +185,13 @@ pub fn matches(self: Regex, text: []const u8) bool {
return rc >= 0;
}
-const testing = @import("../../testing.zig");
+const testing = @import("testing.zig");
-test "adblock.Regex: JavaScript escapes and unanchored search" {
+test "Regex: JavaScript escapes and unanchored search" {
const context: *Context = try .init(testing.allocator);
defer context.deinit();
- const regex = try Regex.compile(context, "^https?:\\/\\/[0-9a-z]{5,}\\.com\\/.*", true);
+ const regex = try context.compile("^https?:\\/\\/[0-9a-z]{5,}\\.com\\/.*", .{ .case_insensitive = true }, null);
defer regex.deinit();
try testing.expect(regex.matches("https://abcde.com/x"));
@@ -181,44 +199,104 @@ test "adblock.Regex: JavaScript escapes and unanchored search" {
try testing.expect(!regex.matches("https://abcd.com/x"));
try testing.expect(!regex.matches("https://abcde.org/x"));
- const invoke = try Regex.compile(context, "\\/[0-9a-f]{32}\\/invoke\\.js", true);
+ const invoke = try context.compile("\\/[0-9a-f]{32}\\/invoke\\.js", .{ .case_insensitive = true }, null);
defer invoke.deinit();
try testing.expect(invoke.matches("https://host.com/0123456789abcdef0123456789abcdef/invoke.js"));
try testing.expect(!invoke.matches("https://host.com/0123456789abcdef0123456789abcde/invoke.js"));
- const dash = try Regex.compile(context, "[a-z\\-]+\\?s=", true);
+ const dash = try context.compile("[a-z\\-]+\\?s=", .{ .case_insensitive = true }, null);
defer dash.deinit();
try testing.expect(dash.matches("https://x.com/a-b?s=1"));
try testing.expect(!dash.matches("https://x.com/?s=1"));
}
-test "adblock.Regex: $match-case keeps the case" {
+test "Regex: case is kept by default" {
const context: *Context = try .init(testing.allocator);
defer context.deinit();
- const exact = try Regex.compile(context, "\\/[a-z0-9]{12}\\/[a-zA-Z0-9]{20,}$", false);
+ const exact = try context.compile("\\/[a-z0-9]{12}\\/[a-zA-Z0-9]{20,}$", .{}, null);
defer exact.deinit();
try testing.expect(exact.matches("https://x.com/abcdef123456/aBcDeFgHiJkLmNoPqRsTuV"));
try testing.expect(!exact.matches("https://x.com/ABCDEF123456/aBcDeFgHiJkLmNoPqRsTuV"));
}
-test "adblock.Regex: invalid patterns are errors, runaway ones no match" {
+test "Regex: invalid patterns are errors, runaway ones no match" {
const context: *Context = try .init(testing.allocator);
defer context.deinit();
- try testing.expectError(error.InvalidRegex, Regex.compile(context, "(", true));
- try testing.expectError(error.InvalidRegex, Regex.compile(context, "a{2,1}", true));
+ try testing.expectError(error.InvalidRegex, context.compile("(", .{}, null));
+ try testing.expectError(error.InvalidRegex, context.compile("a{2,1}", .{}, null));
// An unknown alphanumeric escape is the literal, as in JavaScript.
- const literal = try Regex.compile(context, "\\q", true);
+ const literal = try context.compile("\\q", .{ .case_insensitive = true }, null);
defer literal.deinit();
try testing.expect(literal.matches("https://x.com/q"));
// Exponential backtracking stops at the match limit instead of stalling
- // the request.
- const runaway = try Regex.compile(context, "^(a+)+$", true);
+ // the caller.
+ const runaway = try context.compile("^(a+)+$", .{ .case_insensitive = true }, null);
defer runaway.deinit();
const subject = "a" ** 64 ++ "b";
try testing.expect(!runaway.matches(subject));
try testing.expect(runaway.matches("a" ** 64));
}
+
+test "Regex: dot_all and multiline follow the JavaScript flags" {
+ const context: *Context = try .init(testing.allocator);
+ defer context.deinit();
+
+ const dot = try context.compile("a.b", .{}, null);
+ defer dot.deinit();
+ try testing.expect(!dot.matches("a\nb"));
+ const dot_all = try context.compile("a.b", .{ .dot_all = true }, null);
+ defer dot_all.deinit();
+ try testing.expect(dot_all.matches("a\nb"));
+
+ const line = try context.compile("^b$", .{}, null);
+ defer line.deinit();
+ try testing.expect(!line.matches("a\nb"));
+ const multiline = try context.compile("^b$", .{ .multiline = true }, null);
+ defer multiline.deinit();
+ try testing.expect(multiline.matches("a\nb"));
+}
+
+test "Regex: a diagnostic names the fault and where it is" {
+ const context: *Context = try .init(testing.allocator);
+ defer context.deinit();
+
+ var diag: Diagnostic = .{};
+ try testing.expectError(error.InvalidRegex, context.compile("ab(", .{}, &diag));
+ try testing.expectString("missing closing parenthesis", diag.message());
+ try testing.expectEqual(3, diag.offset);
+}
+
+test "Regex: unicode folds case beyond ASCII and tolerates invalid bytes" {
+ const context: *Context = try .init(testing.allocator);
+ defer context.deinit();
+
+ const ascii = try context.compile("^реклама$", .{ .case_insensitive = true }, null);
+ defer ascii.deinit();
+ try testing.expect(ascii.matches("реклама"));
+ try testing.expect(!ascii.matches("Реклама"));
+
+ const unicode = try context.compile("^реклама$", .{ .case_insensitive = true, .unicode = true }, null);
+ defer unicode.deinit();
+ try testing.expect(unicode.matches("Реклама"));
+ try testing.expect(unicode.matches("РЕКЛАМА"));
+
+ // One code point, not one byte.
+ const single = try context.compile("^.$", .{ .unicode = true }, null);
+ defer single.deinit();
+ try testing.expect(single.matches("é"));
+ try testing.expect(!single.matches("ab"));
+
+ // Word boundaries stay ASCII, as in JavaScript.
+ const word = try context.compile("\\bshare\\b", .{ .unicode = true }, null);
+ defer word.deinit();
+ try testing.expect(word.matches("éshare"));
+
+ const sidebar = try context.compile("sidebar", .{ .unicode = true }, null);
+ defer sidebar.deinit();
+ try testing.expect(sidebar.matches("sidebar\xFF"));
+ try testing.expect(!sidebar.matches("\xFF"));
+}
diff --git a/src/browser/interactive.zig b/src/browser/interactive.zig
index 655af76b6..a25a3941e 100644
--- a/src/browser/interactive.zig
+++ b/src/browser/interactive.zig
@@ -20,6 +20,7 @@ const std = @import("std");
const Frame = @import("Frame.zig");
const URL = @import("URL.zig");
+const Regex = @import("../Regex.zig");
const TreeWalker = @import("webapi/TreeWalker.zig");
const Label = @import("webapi/element/html/Label.zig");
const AXNode = @import("../server/cdp/AXNode.zig");
@@ -149,11 +150,18 @@ pub fn collectInteractiveElements(
return walkInteractive(root, arena, frame, .{});
}
+pub const Name = union(enum) {
+ /// Case-insensitive.
+ substring: []const u8,
+ /// Unanchored.
+ regex: Regex,
+};
+
const FindFilter = struct {
/// Exact role match (case-insensitive). When null, role is not filtered.
role: ?[]const u8 = null,
- /// Accessible-name substring match (case-insensitive). When null, name is not filtered.
- name: ?[]const u8 = null,
+ /// Accessible-name match. When null, name is not filtered.
+ name: ?Name = null,
/// Stop walking once this many matches accumulate. When null, walks the full subtree.
max: ?usize = null,
};
@@ -227,7 +235,11 @@ fn walkInteractive(
if (role == null) try getTextContent(node, arena) else null;
if (filter.name) |nf| {
const n = name orelse continue;
- if (std.ascii.indexOfIgnoreCase(n, nf) == null) continue;
+ const hit = switch (nf) {
+ .substring => |s| std.ascii.indexOfIgnoreCase(n, s) != null,
+ .regex => |re| re.matches(n),
+ };
+ if (!hit) continue;
}
const listener_types = getListenerTypes(el.asEventTarget(), listener_targets);
@@ -490,6 +502,30 @@ fn testInteractiveInBody(html: []const u8) ![]InteractiveElement {
return collectInteractiveElements(div.asNode(), frame.call_arena, frame);
}
+test "browser.interactive: a name regex filters the walk" {
+ const frame = try testing.createFrame();
+ defer testing.test_session.closeAllPages();
+ const doc = frame.window._document;
+ const div = try doc.createElement("div", null, frame);
+ try Frame.parse.htmlAsChildren(frame, div.asNode(), "Add item");
+
+ const context = testing.test_app.regex_context;
+ const options: Regex.Options = .{ .case_insensitive = true, .unicode = true };
+
+ const starts_add = try context.compile("^add", options, null);
+ defer starts_add.deinit();
+ const found_add = try findInteractiveElements(div.asNode(), frame.call_arena, frame, .{ .name = .{ .regex = starts_add } });
+ try testing.expectEqual(2, found_add.len);
+ try testing.expectEqual("Add to cart", found_add[0].name.?);
+ try testing.expectEqual("Add item", found_add[1].name.?);
+
+ const only_cart = try context.compile("^cart$", options, null);
+ defer only_cart.deinit();
+ const found_cart = try findInteractiveElements(div.asNode(), frame.call_arena, frame, .{ .name = .{ .regex = only_cart } });
+ try testing.expectEqual(1, found_cart.len);
+ try testing.expectEqual("Cart", found_cart[0].name.?);
+}
+
test "browser.interactive: names come from labels, like the tree" {
const elements = try testInteractiveInBody(
\\
diff --git a/src/browser/tools.zig b/src/browser/tools.zig
index 68d7d131b..37448160d 100644
--- a/src/browser/tools.zig
+++ b/src/browser/tools.zig
@@ -50,8 +50,8 @@ pub const driver_guidance =
\\ values are already in the tree — don't re-fetch via `nodeDetails`.
\\- `nodeDetails(backendNodeId)` → a ready-to-use CSS `selector` that
\\ resolves to one node, plus its id/class/attrs.
- \\- `findElement(role, name)` → locate a candidate by role/name without
- \\ parsing the whole tree.
+ \\- `findElement(role, name)` → locate a candidate by role and name (a
+ \\ substring, or `/regex/`) without parsing the whole tree.
\\- `markdown(selector | backendNodeId)` → readable text for one
\\ subtree. Use after `tree` has shown you where the interesting
\\ region is.
@@ -683,7 +683,7 @@ pub const Tool = enum {
\\ "type": "object",
\\ "properties": {
\\ "role": { "type": "string", "description": "Optional ARIA role to match (e.g. 'button', 'link', 'textbox', 'checkbox')." },
- \\ "name": { "type": "string", "description": "Optional accessible name substring to match (case-insensitive)." }
+ \\ "name": { "type": "string", "description": "Optional accessible name to match, case-insensitive: a substring, or a JavaScript regex literal such as /sign (in|up)/ (unanchored; flags i, m, s, u accepted; case-insensitive even without i, prefix (?-i) to make it case-sensitive)." }
\\ }
\\}
),
@@ -814,8 +814,9 @@ pub fn errorMessage(err: ToolError) []const u8 {
/// Outcome of running a tool against the page. Operational failures (OOM,
/// missing page, invalid params) come out as Zig errors on the enclosing
/// `!ToolResult`; `is_error = true` is the in-band signal for a JS-level
-/// failure (V8 caught a throw inside `evaluate`/`extract`) — the LLM consumes
-/// `text` either way to self-correct. Non-evaluate tools always set `is_error =
+/// failure (V8 caught a throw inside `evaluate`/`extract`) or any failure whose
+/// message carries detail the model needs — the LLM consumes `text` either way
+/// to self-correct. Non-evaluate tools always set `is_error =
/// false` on success.
pub const ToolResult = struct {
text: []const u8,
@@ -924,7 +925,7 @@ fn dispatch(
.press => .{ .text = try execPress(arena, session, registry, substituted) },
.selectOption => .{ .text = try execSelectOption(arena, session, registry, substituted) },
.setChecked => .{ .text = try execSetChecked(arena, session, registry, substituted) },
- .findElement => .{ .text = try execFindElement(arena, session, registry, substituted) },
+ .findElement => execFindElement(arena, session, registry, substituted),
.evaluate => execEvaluate(arena, session, registry, substituted),
.extract => execExtract(arena, session, registry, substituted),
.getEnv => .{ .text = try execGetEnv(arena, substituted) },
@@ -2067,7 +2068,7 @@ fn execSetChecked(arena: std.mem.Allocator, session: *lp.Session, registry: *Nod
return finalizeAction(arena, session, registry, scope, body);
}
-fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *NodeRegistry, arguments: ?std.json.Value) ToolError![]const u8 {
+fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *NodeRegistry, arguments: ?std.json.Value) ToolError!ToolResult {
const Params = struct {
role: ?[]const u8 = null,
name: ?[]const u8 = null,
@@ -2078,14 +2079,77 @@ fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *No
const page = try requireFrame(session);
+ var name_filter: ?lp.interactive.Name = null;
+ defer if (name_filter) |nf| switch (nf) {
+ .regex => |re| re.deinit(),
+ .substring => {},
+ };
+ if (args.name) |name| {
+ if (regexLiteral(name)) |lit| {
+ var options: lp.Regex.Options = .{ .case_insensitive = true, .unicode = true };
+ for (lit.flags) |flag| switch (flag) {
+ 'i', 'u' => {},
+ 's' => options.dot_all = true,
+ 'm' => options.multiline = true,
+ else => return .{
+ .text = try std.fmt.allocPrint(arena, "findElement: unsupported regex flag '{c}' in '{s}'", .{ flag, name }),
+ .is_error = true,
+ },
+ };
+ var diag: lp.Regex.Diagnostic = .{};
+ const regex = session.browser.app.regex_context.compile(lit.body, options, &diag) catch |err| switch (err) {
+ error.OutOfMemory => return error.OutOfMemory,
+ error.InvalidRegex => return .{
+ .text = try std.fmt.allocPrint(arena, "findElement: invalid name regex '{s}': {s} at offset {d}", .{ lit.body, diag.message(), diag.offset }),
+ .is_error = true,
+ },
+ };
+ name_filter = .{ .regex = regex };
+ } else {
+ name_filter = .{ .substring = name };
+ }
+ }
+
const matched = lp.interactive.findInteractiveElements(page.document.asNode(), arena, page, .{
.role = args.role,
- .name = args.name,
+ .name = name_filter,
}) catch return ToolError.InternalError;
lp.interactive.registerNodes(matched, registry) catch
return ToolError.InternalError;
- return renderJson(arena, matched);
+ return .{ .text = try renderJson(arena, matched) };
+}
+
+const RegexLiteral = struct {
+ body: []const u8,
+ flags: []const u8,
+};
+
+/// A JavaScript `/body/flags` literal, or null for plain text. A name really
+/// written as `/foo/` still matches itself, the search being unanchored.
+fn regexLiteral(text: []const u8) ?RegexLiteral {
+ if (text.len == 0 or text[0] != '/') return null;
+ const close = std.mem.lastIndexOfScalar(u8, text, '/') orelse return null;
+ if (close < 2) return null;
+ const flags = text[close + 1 ..];
+ for (flags) |flag| {
+ if (std.mem.indexOfScalar(u8, "dgimsuvy", flag) == null) return null;
+ }
+ return .{ .body = text[1..close], .flags = flags };
+}
+
+test "regexLiteral" {
+ for ([_][]const u8{ "foo", "/", "//", "//i", "/foo", "/foo/ bar", "/usr/bin" }) |text| {
+ try std.testing.expectEqual(null, regexLiteral(text));
+ }
+
+ const plain = regexLiteral("/foo/").?;
+ try std.testing.expectEqualStrings("foo", plain.body);
+ try std.testing.expectEqualStrings("", plain.flags);
+
+ const flagged = regexLiteral("/a/b/gi").?;
+ try std.testing.expectEqualStrings("a/b", flagged.body);
+ try std.testing.expectEqualStrings("gi", flagged.flags);
}
fn execGetEnv(arena: std.mem.Allocator, arguments: ?std.json.Value) ToolError![]const u8 {
diff --git a/src/lightpanda.zig b/src/lightpanda.zig
index 45cef95f8..c26d734f3 100644
--- a/src/lightpanda.zig
+++ b/src/lightpanda.zig
@@ -22,6 +22,7 @@ pub const log = @import("log.zig");
pub const mcp = @import("mcp.zig");
pub const App = @import("App.zig");
pub const Arena = @import("Arena.zig");
+pub const Regex = @import("Regex.zig");
pub const Config = @import("Config.zig");
pub const cookies = @import("cookies.zig");
pub const datetime = @import("datetime.zig");
diff --git a/src/mcp/tools.zig b/src/mcp/tools.zig
index 08d7e01ca..e6812042b 100644
--- a/src/mcp/tools.zig
+++ b/src/mcp/tools.zig
@@ -1448,6 +1448,36 @@ test "MCP - findElement" {
try testing.expect(std.mem.indexOf(u8, out.written(), "error") != null);
out.clearRetainingCapacity();
}
+
+ {
+ const msg =
+ \\{"jsonrpc":"2.0","id":5,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/^PREVENT.*default$/i"}}}
+ ;
+ try router.handleMessage(server, aa, msg);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "Prevent Default") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "Click Me") == null);
+ out.clearRetainingCapacity();
+ }
+
+ {
+ const msg =
+ \\{"jsonrpc":"2.0","id":6,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/(/"}}}
+ ;
+ try router.handleMessage(server, aa, msg);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "\"isError\":true") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "missing closing parenthesis at offset 1") != null);
+ out.clearRetainingCapacity();
+ }
+
+ {
+ const msg =
+ \\{"jsonrpc":"2.0","id":8,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/prevent/g"}}}
+ ;
+ try router.handleMessage(server, aa, msg);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "\"isError\":true") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "unsupported regex flag 'g' in '/prevent/g'") != null);
+ out.clearRetainingCapacity();
+ }
}
test "MCP - waitForSelector: existing element" {
diff --git a/src/network/HttpClient.zig b/src/network/HttpClient.zig
index 2c1400b37..ceb22282f 100644
--- a/src/network/HttpClient.zig
+++ b/src/network/HttpClient.zig
@@ -4431,7 +4431,7 @@ test "HttpClient: adblock verdicts apply per request" {
var client: Client = undefined;
initTestClient(&client, &pool);
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
var list: std.Io.Reader = .fixed(
\\||ads.example.com^
diff --git a/src/network/Network.zig b/src/network/Network.zig
index b180c4936..71ba1fe3f 100644
--- a/src/network/Network.zig
+++ b/src/network/Network.zig
@@ -106,7 +106,7 @@ pub fn init(app: *App) !Network {
null;
errdefer if (web_bot_auth) |wba| wba.deinit(allocator);
- var adblocker = try AdBlocker.fromConfig(allocator, config);
+ var adblocker = try AdBlocker.fromConfig(allocator, config, app.regex_context);
errdefer if (adblocker) |*blocker| blocker.deinit();
var cache = try Cache.init(allocator, config);
diff --git a/src/network/adblock/AdBlocker.zig b/src/network/adblock/AdBlocker.zig
index 9699783aa..b078cfd2c 100644
--- a/src/network/adblock/AdBlocker.zig
+++ b/src/network/adblock/AdBlocker.zig
@@ -29,7 +29,7 @@ const Parser = @import("Parser.zig");
const Engine = @import("Engine.zig");
const HostnameTrie = @import("HostnameTrie.zig");
const NetworkFilter = @import("NetworkFilter.zig");
-const Regex = @import("Regex.zig");
+const Regex = lp.Regex;
const log = lp.log;
@@ -52,7 +52,7 @@ badfilters: std.AutoHashMapUnmanaged(u64, void),
/// filter; they are freed here, not through `filters`.
regexes: std.ArrayList(*const Regex),
/// What the regexes are compiled and run with; PCRE2 allocates through it.
-regex_context: *Regex.Context,
+regex_context: *const Regex.Context,
built: bool,
trie: HostnameTrie,
blocked: u32,
@@ -82,9 +82,7 @@ rules_cosmetic: usize,
/// Real-world rules are well under 1KB; the parser skips anything longer.
const LINE_MAX = 8 * 1024;
-pub fn init(allocator: Allocator) Allocator.Error!AdBlocker {
- const regex_context: *Regex.Context = try .init(allocator);
- errdefer regex_context.deinit();
+pub fn init(allocator: Allocator, regex_context: *const Regex.Context) Allocator.Error!AdBlocker {
var trie: HostnameTrie = try .init(allocator);
errdefer trie.deinit(allocator);
const blocked = try trie.createTrie(allocator);
@@ -121,7 +119,6 @@ pub fn deinit(self: *AdBlocker) void {
self.badfilters.deinit(self.allocator);
for (self.regexes.items) |regex| regex.deinit();
self.regexes.deinit(self.allocator);
- self.regex_context.deinit();
self.filters.deinit(self.allocator);
self.trie.deinit(self.allocator);
self.arena.deinit();
@@ -129,7 +126,7 @@ pub fn deinit(self: *AdBlocker) void {
/// Builds the blocker from `--adblock-lists`, or null when the option is
/// unset. Lists accumulate into the one instance.
-pub fn fromConfig(allocator: Allocator, config: *const Config) !?AdBlocker {
+pub fn fromConfig(allocator: Allocator, config: *const Config, regex_context: *const Regex.Context) !?AdBlocker {
var paths = config.adblockLists() orelse return null;
var adblocker: ?AdBlocker = null;
@@ -140,7 +137,7 @@ pub fn fromConfig(allocator: Allocator, config: *const Config) !?AdBlocker {
while (paths.next()) |path| {
if (path.len == 0) continue;
- if (adblocker == null) adblocker = try AdBlocker.init(allocator);
+ if (adblocker == null) adblocker = try AdBlocker.init(allocator, regex_context);
loadList(&adblocker.?, path, buf) catch |err| {
log.err(.app, "adblock list load failed", .{ .path = path, .err = err });
return err;
@@ -225,9 +222,15 @@ pub fn parse(self: *AdBlocker, reader: *Io.Reader) !void {
if (filter.kind == .regex) {
const body = filter.pattern[1 .. filter.pattern.len - 1];
- const compiled = Regex.compile(self.regex_context, body, !filter.match_case) catch |err| switch (err) {
+ var diag: Regex.Diagnostic = .{};
+ const compiled = self.regex_context.compile(body, .{ .case_insensitive = !filter.match_case }, &diag) catch |err| switch (err) {
error.OutOfMemory => return error.OutOfMemory,
error.InvalidRegex => {
+ log.debug(.app, "adblock regex rejected", .{
+ .pattern = body,
+ .err = diag.message(),
+ .offset = diag.offset,
+ });
self.rules_skipped += 1;
continue;
},
@@ -481,7 +484,7 @@ const document: ResourceTypes = .{ .document = true };
const frame: ResourceTypes = .{ .subdocument = true };
test "adblock.AdBlocker: the doubleclick rules from EasyList" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -523,7 +526,7 @@ test "adblock.AdBlocker: the doubleclick rules from EasyList" {
}
test "adblock.AdBlocker: the youtube rules from EasyList" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -559,7 +562,7 @@ test "adblock.AdBlocker: the youtube rules from EasyList" {
}
test "adblock.AdBlocker: parse accumulates across lists" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
var first: Io.Reader = .fixed(
@@ -596,7 +599,7 @@ test "adblock.AdBlocker: parse accumulates across lists" {
}
test "adblock.AdBlocker: tokens past the request buffer still match" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -616,7 +619,7 @@ test "adblock.AdBlocker: tokens past the request buffer still match" {
}
test "adblock.AdBlocker: cosmetic-realm rules are not skipped rules" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -633,7 +636,7 @@ test "adblock.AdBlocker: cosmetic-realm rules are not skipped rules" {
}
test "adblock.AdBlocker: the regex rules from EasyList" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -666,7 +669,7 @@ test "adblock.AdBlocker: the regex rules from EasyList" {
}
test "adblock.AdBlocker: a regex is found under every token it may match" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -679,7 +682,7 @@ test "adblock.AdBlocker: a regex is found under every token it may match" {
}
test "adblock.AdBlocker: $badfilter removes a regex rule" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -692,7 +695,7 @@ test "adblock.AdBlocker: $badfilter removes a regex rule" {
}
test "adblock.AdBlocker: verdict precedence" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -717,7 +720,7 @@ test "adblock.AdBlocker: verdict precedence" {
}
test "adblock.AdBlocker: $badfilter removes its target" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -743,7 +746,7 @@ test "adblock.AdBlocker: $badfilter removes its target" {
}
test "adblock.AdBlocker: exceptions we cannot read suppress their hostname" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -769,7 +772,7 @@ test "adblock.AdBlocker: exceptions we cannot read suppress their hostname" {
}
test "adblock.AdBlocker: trie absorbs every pure-hostname form" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try testLoad(&blocker,
@@ -795,7 +798,7 @@ test "adblock.AdBlocker: trie absorbs every pure-hostname form" {
}
test "adblock.AdBlocker: $important on an exception is invalid" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
// uBO rejects `@@…$important`, so nothing outranks a `$important`
@@ -812,7 +815,7 @@ test "adblock.AdBlocker: $important on an exception is invalid" {
}
test "adblock.AdBlocker: an empty blocker decides nothing" {
- var blocker: AdBlocker = try .init(testing.allocator);
+ var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context);
defer blocker.deinit();
try blocker.build();
diff --git a/src/network/adblock/NetworkFilter.zig b/src/network/adblock/NetworkFilter.zig
index 6140c5832..9e23e2862 100644
--- a/src/network/adblock/NetworkFilter.zig
+++ b/src/network/adblock/NetworkFilter.zig
@@ -18,7 +18,7 @@
const std = @import("std");
const domain = @import("domain.zig");
-const Regex = @import("Regex.zig");
+const Regex = @import("../../Regex.zig");
const NetworkFilter = @This();
diff --git a/src/script/skill.zig b/src/script/skill.zig
index 7ad4306c9..8de0e5329 100644
--- a/src/script/skill.zig
+++ b/src/script/skill.zig
@@ -203,7 +203,8 @@ fn note(tool: browser_tools.Tool) []const u8 {
.screenshot => "`path` is required: writes a PNG of the text layout.",
.press => "Selector first! `page.press(\"Enter\")` binds \"Enter\" to `selector` and fails — use `page.press(null, \"Enter\")` or `page.press({ key: \"Enter\" })`.",
.click, .fill, .scroll, .hover, .selectOption, .setChecked => "",
- .search, .markdown, .html, .links, .tree, .nodeDetails, .interactiveElements, .structuredData, .detectForms, .findElement, .consoleLogs, .getUrl, .getCookies, .getEnv => "",
+ .findElement => "`name` is a case-insensitive substring, or a JS regex literal like `/sign (in|up)/` (also case-insensitive).",
+ .search, .markdown, .html, .links, .tree, .nodeDetails, .interactiveElements, .structuredData, .detectForms, .consoleLogs, .getUrl, .getCookies, .getEnv => "",
};
}