diff --git a/src/App.zig b/src/App.zig index fe5a0effa..3d573294b 100644 --- a/src/App.zig +++ b/src/App.zig @@ -20,6 +20,7 @@ const std = @import("std"); const lp = @import("lightpanda"); const Config = @import("Config.zig"); +const Regex = @import("Regex.zig"); const Snapshot = @import("browser/js/Snapshot.zig"); const Platform = @import("browser/js/Platform.zig"); const Telemetry = @import("telemetry/telemetry.zig").Telemetry; @@ -43,6 +44,8 @@ allocator: Allocator, arena_pool: ArenaPool, app_dir_path: ?[]const u8, +regex_context: *Regex.Context, + pub fn init(allocator: Allocator, config: *const Config) !*App { const platform = try Platform.init(.{ .v8_flags = config.v8Flags(), @@ -54,6 +57,9 @@ pub fn init(allocator: Allocator, config: *const Config) !*App { const snapshot = try Snapshot.load(); errdefer snapshot.deinit(); + const regex_context: *Regex.Context = try .init(allocator); + errdefer regex_context.deinit(); + const app = try allocator.create(App); errdefer allocator.destroy(app); @@ -62,6 +68,7 @@ pub fn init(allocator: Allocator, config: *const Config) !*App { .allocator = allocator, .platform = platform, .snapshot = snapshot, + .regex_context = regex_context, .network = undefined, .app_dir_path = undefined, .telemetry = undefined, @@ -96,6 +103,8 @@ pub fn deinit(self: *App) void { } self.telemetry.deinit(allocator); self.network.deinit(); + // After `network`: its adblock regexes free through this context. + self.regex_context.deinit(); self.snapshot.deinit(); self.platform.deinit(); self.arena_pool.deinit(); diff --git a/src/network/adblock/Regex.zig b/src/Regex.zig similarity index 52% rename from src/network/adblock/Regex.zig rename to src/Regex.zig index 76806f15d..a5c567f60 100644 --- a/src/network/adblock/Regex.zig +++ b/src/Regex.zig @@ -16,23 +16,18 @@ // You should have received a copy of the GNU Affero General Public License // along with this program. If not, see . -//! A compiled `/regex/` filter body. Filter lists write them in JavaScript -//! `RegExp` syntax and uBO runs them with `new RegExp(src, 'i')` against the -//! raw request URL (no flag under `$match-case`); PCRE2 reads that syntax -//! as-is, escapes like `\/` included. +//! A pattern in JavaScript `RegExp` syntax, run by PCRE2, which reads that +//! syntax as-is, escapes like `\/` included. //! //! A compiled pattern and its `Context` are never modified after `compile`, -//! so one `Regex` can be shared by every HTTP client thread; the per-call -//! match data is what PCRE2 requires to be private. +//! so one `Regex` can be shared by every thread; the per-call match data is +//! what PCRE2 requires to be private. const std = @import("std"); -const lp = @import("lightpanda"); const pcre2 = @import("pcre2"); const Allocator = std.mem.Allocator; -const log = lp.log; - const Regex = @This(); code: *pcre2.pcre2_code_8, @@ -40,21 +35,42 @@ context: *const Context, pub const Error = error{ InvalidRegex, OutOfMemory }; -/// What every `Regex` compiled through it shares: the allocator PCRE2 draws -/// from, and the compile and match settings. Outlives the regexes. +pub const Options = struct { + case_insensitive: bool = false, + /// UTF-8 aware matching: `.` consumes a code point and caseless folding + /// works beyond ASCII. An invalid sequence in the subject fails to match + /// rather than erroring. `\b` and `\w` stay ASCII, as in JavaScript. + unicode: bool = false, + /// JavaScript's `s`. + dot_all: bool = false, + /// JavaScript's `m`. + multiline: bool = false, +}; + +pub const Diagnostic = struct { + offset: usize = 0, + len: usize = 0, + buf: [256]u8 = undefined, + + pub fn message(self: *const Diagnostic) []const u8 { + return self.buf[0..self.len]; + } +}; + +/// Shared by every `Regex` compiled through it; outlives them. /// -/// PCRE2 would happily use libc's malloc; it is handed the blocker's -/// allocator so that a compiled pattern nobody freed fails a test the way -/// any other leak does. +/// PCRE2 would happily use libc's malloc; it is handed the owner's allocator +/// so that a compiled pattern nobody freed fails a test the way any other +/// leak does. pub const Context = struct { allocator: Allocator, general: *pcre2.pcre2_general_context_8, compile_context: *pcre2.pcre2_compile_context_8, match_context: *pcre2.pcre2_match_context_8, - // A pattern from a list that backtracks this much on one URL is broken, - // not slow; giving up costs a false negative on that request, nothing - // more. + // Patterns come from lists and prompts, never from code: one that + // backtracks this much on a subject is broken, not slow, and giving up + // costs a false negative on that subject, nothing more. const MATCH_LIMIT = 100_000; const DEPTH_LIMIT = 10_000; @@ -71,8 +87,7 @@ pub const Context = struct { const compile_context = pcre2.pcre2_compile_context_create_8(general) orelse return error.OutOfMemory; errdefer pcre2.pcre2_compile_context_free_8(compile_context); - // JavaScript without the `u` flag reads an unknown escape as the - // literal character, and that is the mode uBO compiles filters in. + // JavaScript reads an unknown escape as the literal character. _ = pcre2.pcre2_set_compile_extra_options_8(compile_context, pcre2.PCRE2_EXTRA_BAD_ESCAPE_IS_LITERAL); const match_context = pcre2.pcre2_match_context_create_8(general) orelse return error.OutOfMemory; @@ -85,6 +100,37 @@ pub const Context = struct { return self; } + /// A failed compile fills `diag`, when given, with PCRE2's message and the + /// offset of the offending character. + pub fn compile(self: *const Context, pattern: []const u8, options: Options, diag: ?*Diagnostic) Error!Regex { + var flags: u32 = 0; + if (options.case_insensitive) flags |= pcre2.PCRE2_CASELESS; + if (options.unicode) flags |= pcre2.PCRE2_UTF | pcre2.PCRE2_MATCH_INVALID_UTF; + if (options.dot_all) flags |= pcre2.PCRE2_DOTALL; + if (options.multiline) flags |= pcre2.PCRE2_MULTILINE; + + var err_code: c_int = 0; + var err_offset: usize = 0; + const code = pcre2.pcre2_compile_8( + pattern.ptr, + pattern.len, + flags, + &err_code, + &err_offset, + self.compile_context, + ) orelse { + // A failed allocation is ours, not the pattern's. + if (err_code == pcre2.PCRE2_ERROR_HEAP_FAILED) return error.OutOfMemory; + if (diag) |d| { + const len = pcre2.pcre2_get_error_message_8(err_code, &d.buf, d.buf.len); + d.len = if (len < 0) 0 else @intCast(len); + d.offset = err_offset; + } + return error.InvalidRegex; + }; + return .{ .code = code, .context = self }; + } + pub fn deinit(self: *Context) void { pcre2.pcre2_match_context_free_8(self.match_context); pcre2.pcre2_compile_context_free_8(self.compile_context); @@ -114,33 +160,6 @@ pub const Context = struct { } }; -pub fn compile(context: *const Context, pattern: []const u8, case_insensitive: bool) Error!Regex { - const options: u32 = if (case_insensitive) pcre2.PCRE2_CASELESS else 0; - var err_code: c_int = 0; - var err_offset: usize = 0; - const code = pcre2.pcre2_compile_8( - pattern.ptr, - pattern.len, - options, - &err_code, - &err_offset, - context.compile_context, - ) orelse { - // A failed allocation is ours, not the pattern's. - if (err_code == pcre2.PCRE2_ERROR_HEAP_FAILED) return error.OutOfMemory; - var buf: [256]u8 = undefined; - const len = pcre2.pcre2_get_error_message_8(err_code, &buf, buf.len); - const message: []const u8 = if (len < 0) "unknown error" else buf[0..@intCast(len)]; - log.debug(.app, "adblock regex rejected", .{ - .pattern = pattern, - .err = message, - .offset = err_offset, - }); - return error.InvalidRegex; - }; - return .{ .code = code, .context = context }; -} - pub fn deinit(self: Regex) void { pcre2.pcre2_code_free_8(self.code); } @@ -152,7 +171,6 @@ const MATCH_SCRATCH = 24 * 1024; /// Whether the pattern matches anywhere in `text`, as `RegExp.test` would /// answer. A match that hits the backtracking limits counts as no match. pub fn matches(self: Regex, text: []const u8) bool { - // This runs per request; the scratch keeps the common case off the heap. var scratch = std.heap.stackFallback(MATCH_SCRATCH, self.context.allocator); var allocator = scratch.get(); const general = pcre2.pcre2_general_context_create_8(Context.cMalloc, Context.cFree, &allocator) orelse return false; @@ -167,13 +185,13 @@ pub fn matches(self: Regex, text: []const u8) bool { return rc >= 0; } -const testing = @import("../../testing.zig"); +const testing = @import("testing.zig"); -test "adblock.Regex: JavaScript escapes and unanchored search" { +test "Regex: JavaScript escapes and unanchored search" { const context: *Context = try .init(testing.allocator); defer context.deinit(); - const regex = try Regex.compile(context, "^https?:\\/\\/[0-9a-z]{5,}\\.com\\/.*", true); + const regex = try context.compile("^https?:\\/\\/[0-9a-z]{5,}\\.com\\/.*", .{ .case_insensitive = true }, null); defer regex.deinit(); try testing.expect(regex.matches("https://abcde.com/x")); @@ -181,44 +199,104 @@ test "adblock.Regex: JavaScript escapes and unanchored search" { try testing.expect(!regex.matches("https://abcd.com/x")); try testing.expect(!regex.matches("https://abcde.org/x")); - const invoke = try Regex.compile(context, "\\/[0-9a-f]{32}\\/invoke\\.js", true); + const invoke = try context.compile("\\/[0-9a-f]{32}\\/invoke\\.js", .{ .case_insensitive = true }, null); defer invoke.deinit(); try testing.expect(invoke.matches("https://host.com/0123456789abcdef0123456789abcdef/invoke.js")); try testing.expect(!invoke.matches("https://host.com/0123456789abcdef0123456789abcde/invoke.js")); - const dash = try Regex.compile(context, "[a-z\\-]+\\?s=", true); + const dash = try context.compile("[a-z\\-]+\\?s=", .{ .case_insensitive = true }, null); defer dash.deinit(); try testing.expect(dash.matches("https://x.com/a-b?s=1")); try testing.expect(!dash.matches("https://x.com/?s=1")); } -test "adblock.Regex: $match-case keeps the case" { +test "Regex: case is kept by default" { const context: *Context = try .init(testing.allocator); defer context.deinit(); - const exact = try Regex.compile(context, "\\/[a-z0-9]{12}\\/[a-zA-Z0-9]{20,}$", false); + const exact = try context.compile("\\/[a-z0-9]{12}\\/[a-zA-Z0-9]{20,}$", .{}, null); defer exact.deinit(); try testing.expect(exact.matches("https://x.com/abcdef123456/aBcDeFgHiJkLmNoPqRsTuV")); try testing.expect(!exact.matches("https://x.com/ABCDEF123456/aBcDeFgHiJkLmNoPqRsTuV")); } -test "adblock.Regex: invalid patterns are errors, runaway ones no match" { +test "Regex: invalid patterns are errors, runaway ones no match" { const context: *Context = try .init(testing.allocator); defer context.deinit(); - try testing.expectError(error.InvalidRegex, Regex.compile(context, "(", true)); - try testing.expectError(error.InvalidRegex, Regex.compile(context, "a{2,1}", true)); + try testing.expectError(error.InvalidRegex, context.compile("(", .{}, null)); + try testing.expectError(error.InvalidRegex, context.compile("a{2,1}", .{}, null)); // An unknown alphanumeric escape is the literal, as in JavaScript. - const literal = try Regex.compile(context, "\\q", true); + const literal = try context.compile("\\q", .{ .case_insensitive = true }, null); defer literal.deinit(); try testing.expect(literal.matches("https://x.com/q")); // Exponential backtracking stops at the match limit instead of stalling - // the request. - const runaway = try Regex.compile(context, "^(a+)+$", true); + // the caller. + const runaway = try context.compile("^(a+)+$", .{ .case_insensitive = true }, null); defer runaway.deinit(); const subject = "a" ** 64 ++ "b"; try testing.expect(!runaway.matches(subject)); try testing.expect(runaway.matches("a" ** 64)); } + +test "Regex: dot_all and multiline follow the JavaScript flags" { + const context: *Context = try .init(testing.allocator); + defer context.deinit(); + + const dot = try context.compile("a.b", .{}, null); + defer dot.deinit(); + try testing.expect(!dot.matches("a\nb")); + const dot_all = try context.compile("a.b", .{ .dot_all = true }, null); + defer dot_all.deinit(); + try testing.expect(dot_all.matches("a\nb")); + + const line = try context.compile("^b$", .{}, null); + defer line.deinit(); + try testing.expect(!line.matches("a\nb")); + const multiline = try context.compile("^b$", .{ .multiline = true }, null); + defer multiline.deinit(); + try testing.expect(multiline.matches("a\nb")); +} + +test "Regex: a diagnostic names the fault and where it is" { + const context: *Context = try .init(testing.allocator); + defer context.deinit(); + + var diag: Diagnostic = .{}; + try testing.expectError(error.InvalidRegex, context.compile("ab(", .{}, &diag)); + try testing.expectString("missing closing parenthesis", diag.message()); + try testing.expectEqual(3, diag.offset); +} + +test "Regex: unicode folds case beyond ASCII and tolerates invalid bytes" { + const context: *Context = try .init(testing.allocator); + defer context.deinit(); + + const ascii = try context.compile("^реклама$", .{ .case_insensitive = true }, null); + defer ascii.deinit(); + try testing.expect(ascii.matches("реклама")); + try testing.expect(!ascii.matches("Реклама")); + + const unicode = try context.compile("^реклама$", .{ .case_insensitive = true, .unicode = true }, null); + defer unicode.deinit(); + try testing.expect(unicode.matches("Реклама")); + try testing.expect(unicode.matches("РЕКЛАМА")); + + // One code point, not one byte. + const single = try context.compile("^.$", .{ .unicode = true }, null); + defer single.deinit(); + try testing.expect(single.matches("é")); + try testing.expect(!single.matches("ab")); + + // Word boundaries stay ASCII, as in JavaScript. + const word = try context.compile("\\bshare\\b", .{ .unicode = true }, null); + defer word.deinit(); + try testing.expect(word.matches("éshare")); + + const sidebar = try context.compile("sidebar", .{ .unicode = true }, null); + defer sidebar.deinit(); + try testing.expect(sidebar.matches("sidebar\xFF")); + try testing.expect(!sidebar.matches("\xFF")); +} diff --git a/src/browser/interactive.zig b/src/browser/interactive.zig index 655af76b6..a25a3941e 100644 --- a/src/browser/interactive.zig +++ b/src/browser/interactive.zig @@ -20,6 +20,7 @@ const std = @import("std"); const Frame = @import("Frame.zig"); const URL = @import("URL.zig"); +const Regex = @import("../Regex.zig"); const TreeWalker = @import("webapi/TreeWalker.zig"); const Label = @import("webapi/element/html/Label.zig"); const AXNode = @import("../server/cdp/AXNode.zig"); @@ -149,11 +150,18 @@ pub fn collectInteractiveElements( return walkInteractive(root, arena, frame, .{}); } +pub const Name = union(enum) { + /// Case-insensitive. + substring: []const u8, + /// Unanchored. + regex: Regex, +}; + const FindFilter = struct { /// Exact role match (case-insensitive). When null, role is not filtered. role: ?[]const u8 = null, - /// Accessible-name substring match (case-insensitive). When null, name is not filtered. - name: ?[]const u8 = null, + /// Accessible-name match. When null, name is not filtered. + name: ?Name = null, /// Stop walking once this many matches accumulate. When null, walks the full subtree. max: ?usize = null, }; @@ -227,7 +235,11 @@ fn walkInteractive( if (role == null) try getTextContent(node, arena) else null; if (filter.name) |nf| { const n = name orelse continue; - if (std.ascii.indexOfIgnoreCase(n, nf) == null) continue; + const hit = switch (nf) { + .substring => |s| std.ascii.indexOfIgnoreCase(n, s) != null, + .regex => |re| re.matches(n), + }; + if (!hit) continue; } const listener_types = getListenerTypes(el.asEventTarget(), listener_targets); @@ -490,6 +502,30 @@ fn testInteractiveInBody(html: []const u8) ![]InteractiveElement { return collectInteractiveElements(div.asNode(), frame.call_arena, frame); } +test "browser.interactive: a name regex filters the walk" { + const frame = try testing.createFrame(); + defer testing.test_session.closeAllPages(); + const doc = frame.window._document; + const div = try doc.createElement("div", null, frame); + try Frame.parse.htmlAsChildren(frame, div.asNode(), "Add item"); + + const context = testing.test_app.regex_context; + const options: Regex.Options = .{ .case_insensitive = true, .unicode = true }; + + const starts_add = try context.compile("^add", options, null); + defer starts_add.deinit(); + const found_add = try findInteractiveElements(div.asNode(), frame.call_arena, frame, .{ .name = .{ .regex = starts_add } }); + try testing.expectEqual(2, found_add.len); + try testing.expectEqual("Add to cart", found_add[0].name.?); + try testing.expectEqual("Add item", found_add[1].name.?); + + const only_cart = try context.compile("^cart$", options, null); + defer only_cart.deinit(); + const found_cart = try findInteractiveElements(div.asNode(), frame.call_arena, frame, .{ .name = .{ .regex = only_cart } }); + try testing.expectEqual(1, found_cart.len); + try testing.expectEqual("Cart", found_cart[0].name.?); +} + test "browser.interactive: names come from labels, like the tree" { const elements = try testInteractiveInBody( \\ diff --git a/src/browser/tools.zig b/src/browser/tools.zig index 68d7d131b..37448160d 100644 --- a/src/browser/tools.zig +++ b/src/browser/tools.zig @@ -50,8 +50,8 @@ pub const driver_guidance = \\ values are already in the tree — don't re-fetch via `nodeDetails`. \\- `nodeDetails(backendNodeId)` → a ready-to-use CSS `selector` that \\ resolves to one node, plus its id/class/attrs. - \\- `findElement(role, name)` → locate a candidate by role/name without - \\ parsing the whole tree. + \\- `findElement(role, name)` → locate a candidate by role and name (a + \\ substring, or `/regex/`) without parsing the whole tree. \\- `markdown(selector | backendNodeId)` → readable text for one \\ subtree. Use after `tree` has shown you where the interesting \\ region is. @@ -683,7 +683,7 @@ pub const Tool = enum { \\ "type": "object", \\ "properties": { \\ "role": { "type": "string", "description": "Optional ARIA role to match (e.g. 'button', 'link', 'textbox', 'checkbox')." }, - \\ "name": { "type": "string", "description": "Optional accessible name substring to match (case-insensitive)." } + \\ "name": { "type": "string", "description": "Optional accessible name to match, case-insensitive: a substring, or a JavaScript regex literal such as /sign (in|up)/ (unanchored; flags i, m, s, u accepted; case-insensitive even without i, prefix (?-i) to make it case-sensitive)." } \\ } \\} ), @@ -814,8 +814,9 @@ pub fn errorMessage(err: ToolError) []const u8 { /// Outcome of running a tool against the page. Operational failures (OOM, /// missing page, invalid params) come out as Zig errors on the enclosing /// `!ToolResult`; `is_error = true` is the in-band signal for a JS-level -/// failure (V8 caught a throw inside `evaluate`/`extract`) — the LLM consumes -/// `text` either way to self-correct. Non-evaluate tools always set `is_error = +/// failure (V8 caught a throw inside `evaluate`/`extract`) or any failure whose +/// message carries detail the model needs — the LLM consumes `text` either way +/// to self-correct. Non-evaluate tools always set `is_error = /// false` on success. pub const ToolResult = struct { text: []const u8, @@ -924,7 +925,7 @@ fn dispatch( .press => .{ .text = try execPress(arena, session, registry, substituted) }, .selectOption => .{ .text = try execSelectOption(arena, session, registry, substituted) }, .setChecked => .{ .text = try execSetChecked(arena, session, registry, substituted) }, - .findElement => .{ .text = try execFindElement(arena, session, registry, substituted) }, + .findElement => execFindElement(arena, session, registry, substituted), .evaluate => execEvaluate(arena, session, registry, substituted), .extract => execExtract(arena, session, registry, substituted), .getEnv => .{ .text = try execGetEnv(arena, substituted) }, @@ -2067,7 +2068,7 @@ fn execSetChecked(arena: std.mem.Allocator, session: *lp.Session, registry: *Nod return finalizeAction(arena, session, registry, scope, body); } -fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *NodeRegistry, arguments: ?std.json.Value) ToolError![]const u8 { +fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *NodeRegistry, arguments: ?std.json.Value) ToolError!ToolResult { const Params = struct { role: ?[]const u8 = null, name: ?[]const u8 = null, @@ -2078,14 +2079,77 @@ fn execFindElement(arena: std.mem.Allocator, session: *lp.Session, registry: *No const page = try requireFrame(session); + var name_filter: ?lp.interactive.Name = null; + defer if (name_filter) |nf| switch (nf) { + .regex => |re| re.deinit(), + .substring => {}, + }; + if (args.name) |name| { + if (regexLiteral(name)) |lit| { + var options: lp.Regex.Options = .{ .case_insensitive = true, .unicode = true }; + for (lit.flags) |flag| switch (flag) { + 'i', 'u' => {}, + 's' => options.dot_all = true, + 'm' => options.multiline = true, + else => return .{ + .text = try std.fmt.allocPrint(arena, "findElement: unsupported regex flag '{c}' in '{s}'", .{ flag, name }), + .is_error = true, + }, + }; + var diag: lp.Regex.Diagnostic = .{}; + const regex = session.browser.app.regex_context.compile(lit.body, options, &diag) catch |err| switch (err) { + error.OutOfMemory => return error.OutOfMemory, + error.InvalidRegex => return .{ + .text = try std.fmt.allocPrint(arena, "findElement: invalid name regex '{s}': {s} at offset {d}", .{ lit.body, diag.message(), diag.offset }), + .is_error = true, + }, + }; + name_filter = .{ .regex = regex }; + } else { + name_filter = .{ .substring = name }; + } + } + const matched = lp.interactive.findInteractiveElements(page.document.asNode(), arena, page, .{ .role = args.role, - .name = args.name, + .name = name_filter, }) catch return ToolError.InternalError; lp.interactive.registerNodes(matched, registry) catch return ToolError.InternalError; - return renderJson(arena, matched); + return .{ .text = try renderJson(arena, matched) }; +} + +const RegexLiteral = struct { + body: []const u8, + flags: []const u8, +}; + +/// A JavaScript `/body/flags` literal, or null for plain text. A name really +/// written as `/foo/` still matches itself, the search being unanchored. +fn regexLiteral(text: []const u8) ?RegexLiteral { + if (text.len == 0 or text[0] != '/') return null; + const close = std.mem.lastIndexOfScalar(u8, text, '/') orelse return null; + if (close < 2) return null; + const flags = text[close + 1 ..]; + for (flags) |flag| { + if (std.mem.indexOfScalar(u8, "dgimsuvy", flag) == null) return null; + } + return .{ .body = text[1..close], .flags = flags }; +} + +test "regexLiteral" { + for ([_][]const u8{ "foo", "/", "//", "//i", "/foo", "/foo/ bar", "/usr/bin" }) |text| { + try std.testing.expectEqual(null, regexLiteral(text)); + } + + const plain = regexLiteral("/foo/").?; + try std.testing.expectEqualStrings("foo", plain.body); + try std.testing.expectEqualStrings("", plain.flags); + + const flagged = regexLiteral("/a/b/gi").?; + try std.testing.expectEqualStrings("a/b", flagged.body); + try std.testing.expectEqualStrings("gi", flagged.flags); } fn execGetEnv(arena: std.mem.Allocator, arguments: ?std.json.Value) ToolError![]const u8 { diff --git a/src/lightpanda.zig b/src/lightpanda.zig index 45cef95f8..c26d734f3 100644 --- a/src/lightpanda.zig +++ b/src/lightpanda.zig @@ -22,6 +22,7 @@ pub const log = @import("log.zig"); pub const mcp = @import("mcp.zig"); pub const App = @import("App.zig"); pub const Arena = @import("Arena.zig"); +pub const Regex = @import("Regex.zig"); pub const Config = @import("Config.zig"); pub const cookies = @import("cookies.zig"); pub const datetime = @import("datetime.zig"); diff --git a/src/mcp/tools.zig b/src/mcp/tools.zig index 08d7e01ca..e6812042b 100644 --- a/src/mcp/tools.zig +++ b/src/mcp/tools.zig @@ -1448,6 +1448,36 @@ test "MCP - findElement" { try testing.expect(std.mem.indexOf(u8, out.written(), "error") != null); out.clearRetainingCapacity(); } + + { + const msg = + \\{"jsonrpc":"2.0","id":5,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/^PREVENT.*default$/i"}}} + ; + try router.handleMessage(server, aa, msg); + try testing.expect(std.mem.indexOf(u8, out.written(), "Prevent Default") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "Click Me") == null); + out.clearRetainingCapacity(); + } + + { + const msg = + \\{"jsonrpc":"2.0","id":6,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/(/"}}} + ; + try router.handleMessage(server, aa, msg); + try testing.expect(std.mem.indexOf(u8, out.written(), "\"isError\":true") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "missing closing parenthesis at offset 1") != null); + out.clearRetainingCapacity(); + } + + { + const msg = + \\{"jsonrpc":"2.0","id":8,"method":"tools/call","params":{"name":"findElement","arguments":{"name":"/prevent/g"}}} + ; + try router.handleMessage(server, aa, msg); + try testing.expect(std.mem.indexOf(u8, out.written(), "\"isError\":true") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "unsupported regex flag 'g' in '/prevent/g'") != null); + out.clearRetainingCapacity(); + } } test "MCP - waitForSelector: existing element" { diff --git a/src/network/HttpClient.zig b/src/network/HttpClient.zig index 2c1400b37..ceb22282f 100644 --- a/src/network/HttpClient.zig +++ b/src/network/HttpClient.zig @@ -4431,7 +4431,7 @@ test "HttpClient: adblock verdicts apply per request" { var client: Client = undefined; initTestClient(&client, &pool); - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); var list: std.Io.Reader = .fixed( \\||ads.example.com^ diff --git a/src/network/Network.zig b/src/network/Network.zig index b180c4936..71ba1fe3f 100644 --- a/src/network/Network.zig +++ b/src/network/Network.zig @@ -106,7 +106,7 @@ pub fn init(app: *App) !Network { null; errdefer if (web_bot_auth) |wba| wba.deinit(allocator); - var adblocker = try AdBlocker.fromConfig(allocator, config); + var adblocker = try AdBlocker.fromConfig(allocator, config, app.regex_context); errdefer if (adblocker) |*blocker| blocker.deinit(); var cache = try Cache.init(allocator, config); diff --git a/src/network/adblock/AdBlocker.zig b/src/network/adblock/AdBlocker.zig index 9699783aa..b078cfd2c 100644 --- a/src/network/adblock/AdBlocker.zig +++ b/src/network/adblock/AdBlocker.zig @@ -29,7 +29,7 @@ const Parser = @import("Parser.zig"); const Engine = @import("Engine.zig"); const HostnameTrie = @import("HostnameTrie.zig"); const NetworkFilter = @import("NetworkFilter.zig"); -const Regex = @import("Regex.zig"); +const Regex = lp.Regex; const log = lp.log; @@ -52,7 +52,7 @@ badfilters: std.AutoHashMapUnmanaged(u64, void), /// filter; they are freed here, not through `filters`. regexes: std.ArrayList(*const Regex), /// What the regexes are compiled and run with; PCRE2 allocates through it. -regex_context: *Regex.Context, +regex_context: *const Regex.Context, built: bool, trie: HostnameTrie, blocked: u32, @@ -82,9 +82,7 @@ rules_cosmetic: usize, /// Real-world rules are well under 1KB; the parser skips anything longer. const LINE_MAX = 8 * 1024; -pub fn init(allocator: Allocator) Allocator.Error!AdBlocker { - const regex_context: *Regex.Context = try .init(allocator); - errdefer regex_context.deinit(); +pub fn init(allocator: Allocator, regex_context: *const Regex.Context) Allocator.Error!AdBlocker { var trie: HostnameTrie = try .init(allocator); errdefer trie.deinit(allocator); const blocked = try trie.createTrie(allocator); @@ -121,7 +119,6 @@ pub fn deinit(self: *AdBlocker) void { self.badfilters.deinit(self.allocator); for (self.regexes.items) |regex| regex.deinit(); self.regexes.deinit(self.allocator); - self.regex_context.deinit(); self.filters.deinit(self.allocator); self.trie.deinit(self.allocator); self.arena.deinit(); @@ -129,7 +126,7 @@ pub fn deinit(self: *AdBlocker) void { /// Builds the blocker from `--adblock-lists`, or null when the option is /// unset. Lists accumulate into the one instance. -pub fn fromConfig(allocator: Allocator, config: *const Config) !?AdBlocker { +pub fn fromConfig(allocator: Allocator, config: *const Config, regex_context: *const Regex.Context) !?AdBlocker { var paths = config.adblockLists() orelse return null; var adblocker: ?AdBlocker = null; @@ -140,7 +137,7 @@ pub fn fromConfig(allocator: Allocator, config: *const Config) !?AdBlocker { while (paths.next()) |path| { if (path.len == 0) continue; - if (adblocker == null) adblocker = try AdBlocker.init(allocator); + if (adblocker == null) adblocker = try AdBlocker.init(allocator, regex_context); loadList(&adblocker.?, path, buf) catch |err| { log.err(.app, "adblock list load failed", .{ .path = path, .err = err }); return err; @@ -225,9 +222,15 @@ pub fn parse(self: *AdBlocker, reader: *Io.Reader) !void { if (filter.kind == .regex) { const body = filter.pattern[1 .. filter.pattern.len - 1]; - const compiled = Regex.compile(self.regex_context, body, !filter.match_case) catch |err| switch (err) { + var diag: Regex.Diagnostic = .{}; + const compiled = self.regex_context.compile(body, .{ .case_insensitive = !filter.match_case }, &diag) catch |err| switch (err) { error.OutOfMemory => return error.OutOfMemory, error.InvalidRegex => { + log.debug(.app, "adblock regex rejected", .{ + .pattern = body, + .err = diag.message(), + .offset = diag.offset, + }); self.rules_skipped += 1; continue; }, @@ -481,7 +484,7 @@ const document: ResourceTypes = .{ .document = true }; const frame: ResourceTypes = .{ .subdocument = true }; test "adblock.AdBlocker: the doubleclick rules from EasyList" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -523,7 +526,7 @@ test "adblock.AdBlocker: the doubleclick rules from EasyList" { } test "adblock.AdBlocker: the youtube rules from EasyList" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -559,7 +562,7 @@ test "adblock.AdBlocker: the youtube rules from EasyList" { } test "adblock.AdBlocker: parse accumulates across lists" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); var first: Io.Reader = .fixed( @@ -596,7 +599,7 @@ test "adblock.AdBlocker: parse accumulates across lists" { } test "adblock.AdBlocker: tokens past the request buffer still match" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -616,7 +619,7 @@ test "adblock.AdBlocker: tokens past the request buffer still match" { } test "adblock.AdBlocker: cosmetic-realm rules are not skipped rules" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -633,7 +636,7 @@ test "adblock.AdBlocker: cosmetic-realm rules are not skipped rules" { } test "adblock.AdBlocker: the regex rules from EasyList" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -666,7 +669,7 @@ test "adblock.AdBlocker: the regex rules from EasyList" { } test "adblock.AdBlocker: a regex is found under every token it may match" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -679,7 +682,7 @@ test "adblock.AdBlocker: a regex is found under every token it may match" { } test "adblock.AdBlocker: $badfilter removes a regex rule" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -692,7 +695,7 @@ test "adblock.AdBlocker: $badfilter removes a regex rule" { } test "adblock.AdBlocker: verdict precedence" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -717,7 +720,7 @@ test "adblock.AdBlocker: verdict precedence" { } test "adblock.AdBlocker: $badfilter removes its target" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -743,7 +746,7 @@ test "adblock.AdBlocker: $badfilter removes its target" { } test "adblock.AdBlocker: exceptions we cannot read suppress their hostname" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -769,7 +772,7 @@ test "adblock.AdBlocker: exceptions we cannot read suppress their hostname" { } test "adblock.AdBlocker: trie absorbs every pure-hostname form" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try testLoad(&blocker, @@ -795,7 +798,7 @@ test "adblock.AdBlocker: trie absorbs every pure-hostname form" { } test "adblock.AdBlocker: $important on an exception is invalid" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); // uBO rejects `@@…$important`, so nothing outranks a `$important` @@ -812,7 +815,7 @@ test "adblock.AdBlocker: $important on an exception is invalid" { } test "adblock.AdBlocker: an empty blocker decides nothing" { - var blocker: AdBlocker = try .init(testing.allocator); + var blocker: AdBlocker = try .init(testing.allocator, testing.test_app.regex_context); defer blocker.deinit(); try blocker.build(); diff --git a/src/network/adblock/NetworkFilter.zig b/src/network/adblock/NetworkFilter.zig index 6140c5832..9e23e2862 100644 --- a/src/network/adblock/NetworkFilter.zig +++ b/src/network/adblock/NetworkFilter.zig @@ -18,7 +18,7 @@ const std = @import("std"); const domain = @import("domain.zig"); -const Regex = @import("Regex.zig"); +const Regex = @import("../../Regex.zig"); const NetworkFilter = @This(); diff --git a/src/script/skill.zig b/src/script/skill.zig index 7ad4306c9..8de0e5329 100644 --- a/src/script/skill.zig +++ b/src/script/skill.zig @@ -203,7 +203,8 @@ fn note(tool: browser_tools.Tool) []const u8 { .screenshot => "`path` is required: writes a PNG of the text layout.", .press => "Selector first! `page.press(\"Enter\")` binds \"Enter\" to `selector` and fails — use `page.press(null, \"Enter\")` or `page.press({ key: \"Enter\" })`.", .click, .fill, .scroll, .hover, .selectOption, .setChecked => "", - .search, .markdown, .html, .links, .tree, .nodeDetails, .interactiveElements, .structuredData, .detectForms, .findElement, .consoleLogs, .getUrl, .getCookies, .getEnv => "", + .findElement => "`name` is a case-insensitive substring, or a JS regex literal like `/sign (in|up)/` (also case-insensitive).", + .search, .markdown, .html, .links, .tree, .nodeDetails, .interactiveElements, .structuredData, .detectForms, .consoleLogs, .getUrl, .getCookies, .getEnv => "", }; }