diff --git a/src/browser/Frame.zig b/src/browser/Frame.zig index 68d3475e1..738027197 100644 --- a/src/browser/Frame.zig +++ b/src/browser/Frame.zig @@ -4221,7 +4221,11 @@ pub fn handleClick(self: *Frame, target: *Node) !void { pub fn triggerKeyboard(self: *Frame, keyboard_event: *KeyboardEvent) !void { const event = keyboard_event.asEvent(); - const element = self.window._document._active_element orelse { + // Dispatch to the effective active element. When nothing is explicitly + // focused this resolves to (matching `document.activeElement`), so + // the keydown still fires and its default action — e.g. sequential focus + // navigation on Tab — can run. + const element = self.window._document.getActiveElement() orelse { event.deinit(self._page); return; }; @@ -4245,6 +4249,11 @@ pub fn handleKeydown(self: *Frame, target: *Node, event: *Event) !void { return; } + if (key == .Tab) { + // tab -> forward, shift+tab -> backwards + return self.moveFocus(keyboard_event.getShiftKey() == false); + } + if (target.is(Element.Html.Input)) |input| { if (key == .Enter) { return self.submitForm(input.asElement(), input.getForm(self), .{}); @@ -4275,6 +4284,114 @@ pub fn handleKeydown(self: *Frame, target: *Node, event: *Event) !void { } } +// Sequential focus navigation: move `document.activeElement` to the next (Tab) +// or previous (Shift+Tab) focusable element, firing the usual blur/focus events +// via `Element.focus`. The order is fully determined by tabindex + document +// position, so no layout is needed: +// 1. elements with a positive tabindex, in ascending tabindex order; +// 2. then elements with tabindex 0 (or a natively-focusable default), in +// document order. +// Ties within a group break on document order, and Tab wraps around at the ends. +// https://html.spec.whatwg.org/multipage/interaction.html#sequential-focus-navigation +fn moveFocus(self: *Frame, forward: bool) !void { + const document = self.document; + const current = document._active_element; + + const current_tab_index = blk: { + const cur = current orelse break :blk 0; + const current_html = cur.is(Element.Html) orelse break :blk 0; + break :blk current_html.getTabIndex(); + }; + + // Single document-order pass tracking two candidates: + // edge — the global first (forward) / last (backward) focusable element, + // used to wrap around when `current` is at an end, or as the + // landing spot when nothing is focused yet. + // chosen — the closest focusable element strictly past `current` in the + // travel direction. + var edge: ?*Element = null; + var edge_tab_index: i32 = 0; + + var chosen: ?*Element = null; + var chosen_tab_index: i32 = 0; + + var tw = @import("webapi/TreeWalker.zig").Full.Elements.init(document.asNode(), .{}); + while (tw.next()) |candidate| { + if (candidate.isDisabled()) { + continue; + } + if (candidate.is(Element.Html) == null) { + continue; + } + + const candidate_tab_index = blk: { + if (candidate.getAttributeSafe(comptime .wrap("tabindex"))) |attr| { + if (Element.Html.parseInteger(attr)) |tab_index| { + if (tab_index < 0) { + continue; + } + break :blk tab_index; + } + break :blk 0; + } + + // no tab index, maybe this item isn't focusable.. + const focusable = switch (candidate.getTag()) { + .button, .select, .textarea, .iframe => true, + .input => candidate.as(Element.Html.Input)._input_type != .hidden, + .anchor, .area => candidate.getAttributeSafe(comptime .wrap("href")) != null, + else => false, + }; + if (focusable == false) { + continue; + } + + break :blk 0; + }; + + if (edge == null or focusOrderBefore(candidate, candidate_tab_index, edge.?, edge_tab_index) == forward) { + edge = candidate; + edge_tab_index = candidate_tab_index; + } + + const cur = current orelse continue; + + if (candidate == cur) { + continue; + } + + const past = if (forward) focusOrderBefore(cur, current_tab_index, candidate, candidate_tab_index) else focusOrderBefore(candidate, candidate_tab_index, cur, current_tab_index); + if (!past) { + continue; + } + if (chosen == null or focusOrderBefore(candidate, candidate_tab_index, chosen.?, chosen_tab_index) == forward) { + chosen = candidate; + chosen_tab_index = candidate_tab_index; + } + } + + const next = chosen orelse edge orelse return; + try next.focus(self); +} + +// Orders two focusable elements by sequential focus navigation order: positive +// tabindex first (ascending), then tabindex 0, ties broken by document order. +fn focusOrderBefore(a: *Element, a_tab_index: i32, b: *Element, b_tab_index: i32) bool { + if (a_tab_index == b_tab_index) { + // Equal tabindex → document order: `a` precedes `b` when `b` follows `a`. + const FOLLOWING: u16 = 0x04; + return (a.asNode().compareDocumentPosition(b.asNode()) & FOLLOWING) != 0; + } + + const group_a: u8 = if (a_tab_index > 0) 0 else 1; + const group_b: u8 = if (b_tab_index > 0) 0 else 1; + if (group_a != group_b) { + return group_a < group_b; + } + + return a_tab_index < b_tab_index; +} + const SubmitFormOpts = struct { fire_event: bool = true, }; diff --git a/src/browser/structured_data.zig b/src/browser/structured_data.zig index 09cf72a2a..55043cfb6 100644 --- a/src/browser/structured_data.zig +++ b/src/browser/structured_data.zig @@ -39,6 +39,13 @@ pub const AlternateLink = struct { title: ?[]const u8, }; +/// A relation discovered in the HTTP `Link:` response header (RFC 8288), +/// restricted to the registered relations an agent can act on. +pub const LinkRel = struct { + rel: []const u8, + href: []const u8, +}; + pub const StructuredData = struct { json_ld: []const []const u8, open_graph: []const Property, @@ -46,6 +53,7 @@ pub const StructuredData = struct { meta: []const Property, links: []const Property, alternate: []const AlternateLink, + link_headers: []const LinkRel, pub fn jsonStringify(self: *const StructuredData, jw: anytype) !void { try jw.beginObject(); @@ -89,6 +97,23 @@ pub const StructuredData = struct { try jw.endArray(); } + // Emitted only when present so existing consumers of the CDP/MCP + // structured-data shape see no new key on sites without a `Link:` + // response header. + if (self.link_headers.len > 0) { + try jw.objectField("linkHeaders"); + try jw.beginArray(); + for (self.link_headers) |lh| { + try jw.beginObject(); + try jw.objectField("rel"); + try jw.write(lh.rel); + try jw.objectField("href"); + try jw.write(lh.href); + try jw.endObject(); + } + try jw.endArray(); + } + try jw.endObject(); } }; @@ -144,6 +169,7 @@ pub fn collectStructuredData( var meta: std.ArrayList(Property) = .empty; var links: std.ArrayList(Property) = .empty; var alternate: std.ArrayList(AlternateLink) = .empty; + var link_headers: std.ArrayList(LinkRel) = .empty; // Extract language from the root element. if (root.is(Element)) |root_el| { @@ -182,6 +208,10 @@ pub fn collectStructuredData( } } + // The `Link:` response header lives on the frame, not in the DOM, so it is + // collected separately from the navigation's retained headers. + try collectLinkHeaders(arena, frame, &link_headers); + return .{ .json_ld = json_ld.items, .open_graph = open_graph.items, @@ -189,9 +219,150 @@ pub fn collectStructuredData( .meta = meta.items, .links = links.items, .alternate = alternate.items, + .link_headers = link_headers.items, }; } +/// Parse every `Link:` response header retained on the frame, surfacing only the +/// registered, agent-useful relations. Relative targets are resolved against the +/// frame's base URL, mirroring `collectLink`. +fn collectLinkHeaders( + arena: Allocator, + frame: *Frame, + out: *std.ArrayList(LinkRel), +) !void { + // Registered link relations we surface from the HTTP `Link:` response header + // (RFC 8288). + const header_link_rels = [_][]const u8{ "service-doc", "service-desc", "api" }; + + for (frame._http_headers.items) |header| { + if (!std.ascii.eqlIgnoreCase(header.name, "link")) { + continue; + } + + var it: LinkHeaderIterator = .{ .value = header.value }; + while (it.next()) |entry| { + const rel_value = entry.rel orelse continue; + + // The `rel` parameter is a space-separated list of relation types. + var rel_it = std.mem.tokenizeAny(u8, rel_value, " \t"); + while (rel_it.next()) |rel_token| { + const known = for (header_link_rels) |candidate| { + if (std.ascii.eqlIgnoreCase(rel_token, candidate)) { + break candidate; + } + } else continue; + + const href = URL.resolve(arena, frame.base(), entry.target, .{ .encoding = frame.charset }) catch entry.target; + + // Drop duplicate (rel, href) pairs (it happens) + for (out.items) |existing| { + if (std.mem.eql(u8, existing.rel, known) and std.mem.eql(u8, existing.href, href)) { + break; + } + } else { + try out.append(arena, .{ .rel = known, .href = href }); + } + } + } + } +} + +/// One `< target >; param=value; ...` entry of an RFC 8288 `Link` header value. +const LinkHeaderEntry = struct { + target: []const u8, + rel: ?[]const u8, +}; + +/// Splits a `Link:` header field value into its comma-separated link-values, +/// skipping commas that fall inside the `<...>` target or a `"..."` quoted +/// parameter value (RFC 8288 §3). +const LinkHeaderIterator = struct { + i: usize = 0, + value: []const u8, + + fn next(self: *LinkHeaderIterator) ?LinkHeaderEntry { + const v = self.value; + + while (true) { + // Skip separators and whitespace between link-values. The comma that + // ended the previous link-value is consumed here. + for (v[self.i..]) |c| { + if (std.ascii.isWhitespace(c) == false and c != ',') { + break; + } + self.i += 1; + } else { + return null; + } + + // A link-value must begin with the angle-bracketed target. + if (v[self.i] != '<') { + self.skipToNextComma(); + continue; + } + + const target_start = self.i + 1; + const gt = std.mem.indexOfScalarPos(u8, v, target_start, '>') orelse { + self.i = v.len; + return null; + }; + const target = v[target_start..gt]; + self.i = gt + 1; + + // Parameters run to the next top-level comma, left at `self.i`. + const params_start = self.i; + self.skipToNextComma(); + return .{ .target = target, .rel = extractRel(v[params_start..self.i]) }; + } + } + + // Advance until `self.i` sits on the next top-level comma (one outside a + // `"..."` quoted-string, honouring RFC 7230 backslash escapes) or the end + // of the value. The comma itself is left for the leading-separator skip. + fn skipToNextComma(self: *LinkHeaderIterator) void { + const v = self.value; + var in_quotes = false; + while (self.i < v.len) : (self.i += 1) { + const c = v[self.i]; + + if (c == ',' and in_quotes == false) { + return; + } + + if (c == '\\' and in_quotes) { + // Skip the escaped byte; guard a trailing backslash. + if (self.i + 1 < v.len) { + self.i += 1; + } + } else if (c == '"') { + in_quotes = !in_quotes; + } + } + } +}; + +/// Return the (unquoted) value of the `rel` parameter from a link-value's +/// parameter list, or null when absent. +fn extractRel(params: []const u8) ?[]const u8 { + var it = std.mem.splitScalar(u8, params, ';'); + while (it.next()) |raw_param| { + const param = std.mem.trim(u8, raw_param, &std.ascii.whitespace); + const eq = std.mem.indexOfScalarPos(u8, param, 0, '=') orelse continue; + const name = std.mem.trim(u8, param[0..eq], &std.ascii.whitespace); + if (std.ascii.eqlIgnoreCase(name, "rel") == false) { + continue; + } + + var val = std.mem.trim(u8, param[eq + 1 ..], &std.ascii.whitespace); + if (val.len >= 2 and val[0] == '"' and val[val.len - 1] == '"') { + val = val[1 .. val.len - 1]; + } + return val; + } + return null; +} + fn collectJsonLd( el: *Element, arena: Allocator, @@ -484,6 +655,78 @@ test "structured_data: charset and http-equiv" { try testing.expectEqual("text/html; charset=utf-8", findProperty(data.meta, "Content-Type").?); } +test "structured_data: parseLinkHeader - single link with quoted rel" { + var it: LinkHeaderIterator = .{ .value = + \\; rel="service-desc" + }; + const entry = it.next().?; + try testing.expectString("https://api.example.com/openapi.json", entry.target); + try testing.expectString("service-desc", entry.rel.?); + try testing.expectEqual(null, it.next()); +} + +test "structured_data: parseLinkHeader - multiple links and unquoted rel" { + var it: LinkHeaderIterator = .{ .value = + \\; rel=service-doc, ; rel="stylesheet"; type="text/css" + }; + const first = it.next().?; + try testing.expectString("https://docs.example.com/", first.target); + try testing.expectString("service-doc", first.rel.?); + + const second = it.next().?; + try testing.expectString("/style.css", second.target); + // A comma inside the quoted type param must not split the link-value. + try testing.expectString("stylesheet", second.rel.?); + + try testing.expectEqual(null, it.next()); +} + +test "structured_data: parseLinkHeader - escaped quote in param does not split link" { + // A backslash-escaped quote inside an earlier param must not terminate the + // quoted-string, so the comma inside it is not treated as a separator. + var it: LinkHeaderIterator = .{ .value = + \\; title="x\",y"; rel="service-doc" + }; + const entry = it.next().?; + try testing.expectString("https://a/", entry.target); + try testing.expectString("service-doc", entry.rel.?); + try testing.expectEqual(null, it.next()); +} + +test "structured_data: link headers from response" { + const frame = try testing.test_session.createPage(); + defer testing.test_session.removePage(); + + // Stand in for what frameHeaderDoneCallback records from the navigation. + try frame._http_headers.append(frame.arena, .{ .name = "Link", .value = + \\; rel="service-doc" + }); + try frame._http_headers.append(frame.arena, .{ .name = "link", .value = + \\; rel="service-desc", ; rel="stylesheet" + }); + + const doc = frame.window._document; + const div = try doc.createElement("div", null, frame); + try frame.parseHtmlAsChildren(div.asNode(), "Example"); + + const data = try collectStructuredData(div.asNode(), frame.call_arena, frame); + + // service-doc + service-desc are surfaced; stylesheet is filtered out. + try testing.expectEqual(2, data.link_headers.len); + try testing.expectString("service-doc", data.link_headers[0].rel); + try testing.expectString("https://docs.example.com/", data.link_headers[0].href); + try testing.expectString("service-desc", data.link_headers[1].rel); + try testing.expectString("https://api.example.com/spec", data.link_headers[1].href); +} + +test "structured_data: no link header yields empty list" { + const data = try testStructuredData( + \\ + ); + defer testing.test_session.removePage(); + try testing.expectEqual(0, data.link_headers.len); +} + test "structured_data: mixed content" { const data = try testStructuredData( \\My Site diff --git a/src/browser/webapi/element/Html.zig b/src/browser/webapi/element/Html.zig index e124c156d..871a0882d 100644 --- a/src/browser/webapi/element/Html.zig +++ b/src/browser/webapi/element/Html.zig @@ -387,12 +387,16 @@ pub fn togglePopover(self: *HtmlElement, force: ?bool, frame: *Frame) !bool { } pub fn getTabIndex(self: *HtmlElement) i32 { - const default: i32 = switch (self._type) { + if (self.asElement().getAttributeSafe(comptime .wrap("tabindex"))) |attr| { + if (parseInteger(attr)) |tab_index| { + return tab_index; + } + } + + return switch (self._type) { .anchor, .area, .button, .input, .select, .textarea, .iframe => 0, else => -1, }; - const attr = self.asElement().getAttributeSafe(comptime .wrap("tabindex")) orelse return default; - return parseInteger(attr) orelse default; } pub fn setTabIndex(self: *HtmlElement, value: i32, frame: *Frame) !void { diff --git a/src/cdp/domains/input.zig b/src/cdp/domains/input.zig index a6772abd3..a865ee344 100644 --- a/src/cdp/domains/input.zig +++ b/src/cdp/domains/input.zig @@ -314,3 +314,59 @@ test "cdp.input: dispatchMouseEvent right button fires contextmenu, double-click const result = try ls.local.compileAndRun("window.downButton === 2 && window.ctxButton === 2 && window.dbl === true", null); try testing.expect(result.isTrue()); } + +test "cdp.input: dispatchKeyEvent Tab runs sequential focus navigation" { + var ctx = try testing.context(); + defer ctx.deinit(); + + const bc = try ctx.loadBrowserContext(.{}); + const frame = try bc.session.createPage(); + const url = "http://localhost:9582/src/browser/tests/mcp_actions.html"; + try frame.navigate(url, .{ .reason = .address_bar, .kind = .{ .push = null } }); + var runner = try bc.session.runner(.{}); + try runner.wait(.{ .ms = 2000 }); + + var ls: lp.js.Local.Scope = undefined; + frame.js.localScope(&ls); + defer ls.deinit(); + + var try_catch: lp.js.TryCatch = undefined; + try_catch.init(&ls.local); + defer try_catch.deinit(); + + // Three controls whose tabindex order (1, 2, 3) differs from document + // order (2, 1, 3): focus order must follow tabindex, not the tree. + _ = try ls.local.compileAndRun( + \\document.body.innerHTML = + \\ '' + + \\ '' + + \\ ''; + , null); + + // Nothing focused yet → activeElement is . + try testing.expect((try ls.local.compileAndRun("document.activeElement === document.body", null)).isTrue()); + + // First Tab → lowest positive tabindex (#b1), regardless of document order. + try ctx.processMessage(.{ + .id = 1, + .method = "Input.dispatchKeyEvent", + .params = .{ .type = "keyDown", .key = "Tab", .code = "Tab" }, + }); + try testing.expect((try ls.local.compileAndRun("document.activeElement.id === 'b1'", null)).isTrue()); + + // Second Tab → next in tabindex order (#i2). + try ctx.processMessage(.{ + .id = 2, + .method = "Input.dispatchKeyEvent", + .params = .{ .type = "keyDown", .key = "Tab", .code = "Tab" }, + }); + try testing.expect((try ls.local.compileAndRun("document.activeElement.id === 'i2'", null)).isTrue()); + + // Shift+Tab (modifiers bit 8) walks backward → back to #b1. + try ctx.processMessage(.{ + .id = 3, + .method = "Input.dispatchKeyEvent", + .params = .{ .type = "keyDown", .key = "Tab", .code = "Tab", .modifiers = 8 }, + }); + try testing.expect((try ls.local.compileAndRun("document.activeElement.id === 'b1'", null)).isTrue()); +} diff --git a/src/cdp/domains/lp.zig b/src/cdp/domains/lp.zig index b3b3aec0a..dd8fdd991 100644 --- a/src/cdp/domains/lp.zig +++ b/src/cdp/domains/lp.zig @@ -23,6 +23,7 @@ const CDP = @import("../CDP.zig"); const Node = @import("../Node.zig"); const DOMNode = @import("../../browser/webapi/Node.zig"); +const Robots = @import("../../network/Robots.zig"); const markdown = lp.markdown; const SemanticTree = lp.SemanticTree; @@ -36,6 +37,7 @@ pub fn processMessage(cmd: *CDP.Command) !void { getInteractiveElements, getNodeDetails, getStructuredData, + getContentSignal, detectForms, clickNode, fillNode, @@ -51,6 +53,7 @@ pub fn processMessage(cmd: *CDP.Command) !void { .getInteractiveElements => return getInteractiveElements(cmd), .getNodeDetails => return getNodeDetails(cmd), .getStructuredData => return getStructuredData(cmd), + .getContentSignal => return getContentSignal(cmd), .detectForms => return detectForms(cmd), .clickNode => return clickNode(cmd), .fillNode => return fillNode(cmd), @@ -201,6 +204,27 @@ fn getStructuredData(cmd: anytype) !void { }, .{}); } +// Advisory `Content-Signal` robots.txt preferences for the current document's +// host (https://contentsignals.org). `available` is false when no robots.txt +// has been fetched for the host — note this is the case unless `obey_robots` +// is enabled, since the robots layer is what populates the store. +fn getContentSignal(cmd: anytype) !void { + const bc = cmd.browser_context orelse return error.NoBrowserContext; + const frame = bc.session.currentFrame() orelse return error.FrameNotLoaded; + const network = bc.cdp.browser.http_client.network; + + const empty: []const Robots.ContentSignal = &.{}; + const robots_url = lp.URL.getRobotsUrl(cmd.arena, frame.url) catch { + return cmd.sendResult(.{ .available = false, .contentSignals = empty }, .{}); + }; + + const signals = network.robot_store.getContentSignals(robots_url); + return cmd.sendResult(.{ + .available = signals != null, + .contentSignals = signals orelse empty, + }, .{}); +} + fn detectForms(cmd: anytype) !void { const bc = cmd.browser_context orelse return error.NoBrowserContext; const frame = bc.session.currentFrame() orelse return error.FrameNotLoaded; @@ -410,6 +434,25 @@ test "cdp.lp: getStructuredData" { try testing.expect(result.get("structuredData") != null); } +test "cdp.lp: getContentSignal" { + var ctx = try testing.context(); + defer ctx.deinit(); + + const bc = try ctx.loadBrowserContext(.{}); + _ = try bc.session.createPage(); + + try ctx.processMessage(.{ + .id = 1, + .method = "LP.getContentSignal", + }); + + // No robots.txt fetched for the page, so the directive is reported absent + // rather than erroring — the command is still wired and returns the shape. + const result = (try ctx.getSentMessage(0)).?.object.get("result").?.object; + try testing.expectEqual(false, result.get("available").?.bool); + try testing.expect(result.get("contentSignals") != null); +} + test "cdp.lp: action tools" { var ctx = try testing.context(); defer ctx.deinit(); diff --git a/src/cdp/domains/network.zig b/src/cdp/domains/network.zig index dea679989..fcaa2c876 100644 --- a/src/cdp/domains/network.zig +++ b/src/cdp/domains/network.zig @@ -22,10 +22,12 @@ const lp = @import("lightpanda"); const id = @import("../id.zig"); const CDP = @import("../CDP.zig"); +const Config = @import("../../Config.zig"); const URL = @import("../../browser/URL.zig"); const Mime = @import("../../browser/Mime.zig"); const Notification = @import("../../Notification.zig"); const timestamp = @import("../../datetime.zig").timestamp; +const Headers = @import("../../browser/HttpClient.zig").Headers; const Transfer = @import("../../browser/HttpClient.zig").Transfer; const Response = @import("../../browser/HttpClient.zig").Response; @@ -108,7 +110,24 @@ fn setExtraHTTPHeaders(cmd: *CDP.Command) !void { try extra_headers.ensureTotalCapacity(arena, params.headers.map.count()); var it = params.headers.map.iterator(); while (it.next()) |header| { - const header_string = try std.fmt.allocPrintSentinel(arena, "{s}: {s}", .{ header.key_ptr.*, header.value_ptr.* }, 0); + const key = header.key_ptr.*; + const value = header.value_ptr.*; + + if (std.mem.indexOfAny(u8, key, "\r\n") != null or std.mem.indexOfAny(u8, value, "\r\n") != null) { + log.warn(.not_implemented, "network.setExtraHTTPHeaders", .{ .param = "header", .value = key, .info = "header name/value must not contain CR or LF" }); + continue; + } + + const header_string = try std.fmt.allocPrintSentinel(arena, "{s}: {s}", .{ key, value }, 0); + + if (Headers.parseHeader(header_string)) |parsed| { + if (std.ascii.eqlIgnoreCase(parsed.name, "user-agent")) { + Config.validateUserAgent(parsed.value) catch |err| { + log.warn(.not_implemented, "network.setExtraHTTPHeaders", .{ .param = "userAgent", .value = parsed.value, .err = err }); + continue; + }; + } + } extra_headers.appendAssumeCapacity(header_string); } @@ -307,9 +326,11 @@ pub fn httpRequestStart(bc: *CDP.BrowserContext, msg: *const Notification.Reques const frame_id = req.frame_id; const frame = bc.session.findFrameByFrameId(frame_id) orelse return; - // Modify request with extra CDP headers + // Modify request with extra CDP headers. Use set (replace by name) so a + // caller-supplied header overrides a built-in default of the same name + // (e.g. User-Agent) instead of producing a duplicate libcurl drops. for (bc.extra_headers.items) |extra| { - try req.headers.add(extra); + try req.headers.set(extra); } // We're missing a bunch of fields, but, for now, this eems like enough @@ -552,6 +573,109 @@ test "cdp.network setExtraHTTPHeaders" { try testing.expectEqual(bc.extra_headers.items.len, 1); } +test "cdp.network setExtraHTTPHeaders rejects non-printable User-Agent" { + const filter: testing.LogFilter = .init(&.{.not_implemented}); + defer filter.deinit(); + + var ctx = try testing.context(); + defer ctx.deinit(); + + const bc = try ctx.loadBrowserContext(.{ .id = "NID-UA1", .session_id = "NESI-UA1" }); + + try ctx.processMessage(.{ + .id = 3, + .method = "Network.setExtraHTTPHeaders", + .params = .{ .headers = .{ + .@"User-Agent" = "Bot/1.0\x01hidden", + .@"x-custom" = "hi", + } }, + }); + + try testing.expectEqual(bc.extra_headers.items.len, 1); + try testing.expectEqual("x-custom: hi", std.mem.span(bc.extra_headers.items[0])); +} + +test "cdp.network setExtraHTTPHeaders rejects a Mozilla User-Agent" { + const filter: testing.LogFilter = .init(&.{.not_implemented}); + defer filter.deinit(); + + var ctx = try testing.context(); + defer ctx.deinit(); + + _ = try ctx.loadBrowserContext(.{ .id = "NID-UA2", .session_id = "NESI-UA2" }); + + try ctx.processMessage(.{ + .id = 3, + .method = "Network.setExtraHTTPHeaders", + .params = .{ .headers = .{ .@"User-Agent" = "Mozilla/5.0" } }, + }); + + const bc = ctx.cdp().browser_context.?; + try testing.expectEqual(bc.extra_headers.items.len, 0); +} + +test "cdp.network setExtraHTTPHeaders accepts valid User-Agent" { + var ctx = try testing.context(); + defer ctx.deinit(); + + _ = try ctx.loadBrowserContext(.{ .id = "NID-UA3", .session_id = "NESI-UA3" }); + + try ctx.processMessage(.{ + .id = 3, + .method = "Network.setExtraHTTPHeaders", + .params = .{ .headers = .{ .@"User-Agent" = "CustomBot/2.0" } }, + }); + + const bc = ctx.cdp().browser_context.?; + try testing.expectEqual(bc.extra_headers.items.len, 1); +} + +test "cdp.network setExtraHTTPHeaders rejects a Mozilla User-Agent smuggled via a colon in the key" { + const filter: testing.LogFilter = .init(&.{.not_implemented}); + defer filter.deinit(); + + var ctx = try testing.context(); + defer ctx.deinit(); + + _ = try ctx.loadBrowserContext(.{ .id = "NID-UA4", .session_id = "NESI-UA4" }); + + // A colon in the key desyncs the raw key from the first-colon parse that + // req.headers.set/libcurl use: "User-Agent:Mozilla/5.0 (X: Y)" parses to + // name="User-Agent", value="Mozilla/5.0 (X: Y)" on the wire. + try ctx.processMessage(.{ + .id = 3, + .method = "Network.setExtraHTTPHeaders", + .params = .{ .headers = .{ .@"User-Agent:Mozilla/5.0 (X" = "Y)" } }, + }); + + const bc = ctx.cdp().browser_context.?; + try testing.expectEqual(bc.extra_headers.items.len, 0); +} + +test "cdp.network setExtraHTTPHeaders rejects a header that smuggles CRLF" { + const filter: testing.LogFilter = .init(&.{.not_implemented}); + defer filter.deinit(); + + var ctx = try testing.context(); + defer ctx.deinit(); + + const bc = try ctx.loadBrowserContext(.{ .id = "NID-UA5", .session_id = "NESI-UA5" }); + + // The CRLF in the value would inject a second User-Agent line that never + // saw validation; the whole header must be dropped. + try ctx.processMessage(.{ + .id = 3, + .method = "Network.setExtraHTTPHeaders", + .params = .{ .headers = .{ + .@"x-custom" = "bar\r\nUser-Agent: Mozilla/5.0", + .@"x-keep" = "ok", + } }, + }); + + try testing.expectEqual(bc.extra_headers.items.len, 1); + try testing.expectEqual("x-keep: ok", std.mem.span(bc.extra_headers.items[0])); +} + test "cdp.Network: cookies" { const ResCookie = CdpStorage.ResCookie; const CdpCookie = CdpStorage.CdpCookie; diff --git a/src/cdp/domains/runtime.zig b/src/cdp/domains/runtime.zig index 6228ee738..f2758cd92 100644 --- a/src/cdp/domains/runtime.zig +++ b/src/cdp/domains/runtime.zig @@ -167,6 +167,9 @@ pub fn consoleMessage(arena: Allocator, bc: *CDP.BrowserContext, event: *const N const testing = @import("../testing.zig"); test "cdp.runtime: consoleAPICalled type matches the console method" { + const filter: testing.LogFilter = .init(&.{.js}); + defer filter.deinit(); + // Wire types per the CDP protocol: console.log -> "log", // console.warn -> "warning" (not "warn"), console.info -> "info", // console.error -> "error", console.debug -> "debug". diff --git a/src/network/Robots.zig b/src/network/Robots.zig index 202e0a5c7..73038a460 100644 --- a/src/network/Robots.zig +++ b/src/network/Robots.zig @@ -72,12 +72,24 @@ pub const Key = enum { @"user-agent", allow, disallow, + @"content-signal", +}; + +/// A declared content-usage preference from a `Content-Signal:` robots.txt +/// directive (https://contentsignals.org), e.g. name="ai-train" value="no". +/// Advisory metadata surfaced to the agent — it never affects crawl gating. +pub const ContentSignal = struct { + name: []const u8, + value: []const u8, }; /// https://www.rfc-editor.org/rfc/rfc9309.html pub const Robots = @This(); -pub const empty: Robots = .{ .rules = &.{} }; +pub const empty: Robots = .{ .rules = &.{}, .content_signals = &.{} }; +// Think twice before deleting/freeing any entries from the map. Readers, e.g. +// get and getContentSignals, receive values from the map, and if another thread +// was to delete / free those values while in use, UAF. pub const RobotStore = struct { const RobotsEntry = union(enum) { present: Robots, @@ -156,6 +168,18 @@ pub const RobotStore = struct { try self.map.put(self.allocator, duped, .{ .present = robots }); } + // The returned slice is owned by the store + pub fn getContentSignals(self: *RobotStore, url: []const u8) ?[]const ContentSignal { + self.mutex.lock(); + defer self.mutex.unlock(); + + const entry = self.map.get(url) orelse return null; + return switch (entry) { + .present => |robots| robots.content_signals, + .absent => null, + }; + } + pub fn putAbsent(self: *RobotStore, url: []const u8) !void { self.mutex.lock(); defer self.mutex.unlock(); @@ -166,6 +190,15 @@ pub const RobotStore = struct { }; rules: []const Rule, +content_signals: []const ContentSignal = &.{}, + +/// Result of parsing a robots.txt for one user-agent: the applicable crawl +/// rules plus any advisory `Content-Signal` preferences. The two are kept +/// distinct so content signals can never reach `isAllowed`. +const ParseResult = struct { + rules: []Rule, + content_signals: []ContentSignal, +}; const State = struct { entry: enum { @@ -186,17 +219,59 @@ fn freeRulesInList(allocator: std.mem.Allocator, rules: []const Rule) void { } } +fn freeContentSignals(allocator: std.mem.Allocator, signals: []const ContentSignal) void { + for (signals) |signal| { + allocator.free(signal.name); + allocator.free(signal.value); + } +} + +/// Parse a `Content-Signal:` value — a comma-separated list of `name=value` +/// pairs (e.g. `ai-train=no, search=yes`) — into the given list. Names and +/// values are lower-cased and owned by `allocator`. +fn appendContentSignals( + allocator: std.mem.Allocator, + list: *std.ArrayList(ContentSignal), + value: []const u8, +) !void { + var iter = std.mem.splitScalar(u8, value, ','); + while (iter.next()) |raw_pair| { + const pair = std.mem.trim(u8, raw_pair, &std.ascii.whitespace); + const eq = std.mem.indexOfScalarPos(u8, pair, 0, '=') orelse continue; + + const name_raw = std.mem.trim(u8, pair[0..eq], &std.ascii.whitespace); + const value_raw = std.mem.trim(u8, pair[eq + 1 ..], &std.ascii.whitespace); + if (name_raw.len == 0) { + continue; + } + + const name = try std.ascii.allocLowerString(allocator, name_raw); + errdefer allocator.free(name); + + const val = try std.ascii.allocLowerString(allocator, value_raw); + errdefer allocator.free(val); + + try list.append(allocator, .{ .name = name, .value = val }); + } +} + fn parseRulesWithUserAgent( allocator: std.mem.Allocator, user_agent: []const u8, raw_bytes: []const u8, -) ![]Rule { +) !ParseResult { var rules: std.ArrayList(Rule) = .empty; defer rules.deinit(allocator); var wildcard_rules: std.ArrayList(Rule) = .empty; defer wildcard_rules.deinit(allocator); + var content_signals: std.ArrayList(ContentSignal) = .empty; + defer content_signals.deinit(allocator); + + var wildcard_content_signals: std.ArrayList(ContentSignal) = .empty; + defer wildcard_content_signals.deinit(allocator); + var state: State = .{ .entry = .not_in_entry, .has_rules = false }; // https://en.wikipedia.org/wiki/Byte_order_mark @@ -308,22 +383,51 @@ fn parseRulesWithUserAgent( }, } }, + .@"content-signal" => { + // A group member like allow/disallow, so it sets has_rules + // (a following `User-agent:` opens a new group), but it is + // captured into a separate list and never reaches isAllowed. + defer state.has_rules = true; + + switch (state.entry) { + .in_our_entry => try appendContentSignals(allocator, &content_signals, value), + .in_wildcard_entry => try appendContentSignals(allocator, &wildcard_content_signals, value), + .in_other_entry, .not_in_entry => {}, + } + }, } } // If we have rules for our specific User-Agent, we will use those rules. // If we don't have any rules, we fallback to using the wildcard ("*") rules. - if (rules.items.len > 0) { + const out_rules = if (rules.items.len > 0) blk: { freeRulesInList(allocator, wildcard_rules.items); - return try rules.toOwnedSlice(allocator); - } else { + break :blk try rules.toOwnedSlice(allocator); + } else blk: { freeRulesInList(allocator, rules.items); - return try wildcard_rules.toOwnedSlice(allocator); + break :blk try wildcard_rules.toOwnedSlice(allocator); + }; + errdefer { + freeRulesInList(allocator, out_rules); + allocator.free(out_rules); } + + // Content signals follow the same specific-else-wildcard precedence, + // chosen independently of the rule selection above. + const out_signals = if (content_signals.items.len > 0) blk: { + freeContentSignals(allocator, wildcard_content_signals.items); + break :blk try content_signals.toOwnedSlice(allocator); + } else blk: { + freeContentSignals(allocator, content_signals.items); + break :blk try wildcard_content_signals.toOwnedSlice(allocator); + }; + + return .{ .rules = out_rules, .content_signals = out_signals }; } pub fn fromBytes(allocator: std.mem.Allocator, user_agent: []const u8, bytes: []const u8) !Robots { - const rules = try parseRulesWithUserAgent(allocator, user_agent, bytes); + const parsed = try parseRulesWithUserAgent(allocator, user_agent, bytes); + const rules = parsed.rules; // sort by order once. std.mem.sort(Rule, rules, {}, struct { @@ -357,12 +461,14 @@ pub fn fromBytes(allocator: std.mem.Allocator, user_agent: []const u8, bytes: [] } }.lessThan); - return .{ .rules = rules }; + return .{ .rules = rules, .content_signals = parsed.content_signals }; } pub fn deinit(self: *Robots, allocator: std.mem.Allocator) void { freeRulesInList(allocator, self.rules); allocator.free(self.rules); + freeContentSignals(allocator, self.content_signals); + allocator.free(self.content_signals); } /// There are rules for how the pattern in robots.txt should be matched. @@ -476,10 +582,13 @@ test "Robots: simple robots.txt" { \\ ; - const rules = try parseRulesWithUserAgent(allocator, "GoogleBot", file); + const parsed = try parseRulesWithUserAgent(allocator, "GoogleBot", file); + const rules = parsed.rules; defer { freeRulesInList(allocator, rules); allocator.free(rules); + freeContentSignals(allocator, parsed.content_signals); + allocator.free(parsed.content_signals); } try std.testing.expectEqual(1, rules.len); @@ -1003,3 +1112,101 @@ test "Robots: blank lines don't end entries" { try std.testing.expect(robots.isAllowed("/admin/") == false); try std.testing.expect(robots.isAllowed("/public/") == true); } + +test "Robots: content-signal captured and gating unaffected" { + const allocator = std.testing.allocator; + + var robots = try Robots.fromBytes(allocator, "MyBot", + \\User-agent: * + \\Content-Signal: ai-train=no, search=yes, ai-input=yes + \\Disallow: /private/ + \\ + ); + defer robots.deinit(allocator); + + try std.testing.expectEqual(3, robots.content_signals.len); + try std.testing.expectEqualStrings("ai-train", robots.content_signals[0].name); + try std.testing.expectEqualStrings("no", robots.content_signals[0].value); + try std.testing.expectEqualStrings("search", robots.content_signals[1].name); + try std.testing.expectEqualStrings("yes", robots.content_signals[1].value); + try std.testing.expectEqualStrings("ai-input", robots.content_signals[2].name); + + // The whole point: Content-Signal is advisory metadata, it must not move + // the crawl gate. Disallow/Allow still decide isAllowed. + try std.testing.expect(robots.isAllowed("/private/page") == false); + try std.testing.expect(robots.isAllowed("/public/page") == true); +} + +test "Robots: content-signal mixed casing is normalized" { + const allocator = std.testing.allocator; + + var robots = try Robots.fromBytes(allocator, "MyBot", + \\User-agent: * + \\Content-Signal: AI-Train=NO + \\ + ); + defer robots.deinit(allocator); + + try std.testing.expectEqual(1, robots.content_signals.len); + try std.testing.expectEqualStrings("ai-train", robots.content_signals[0].name); + try std.testing.expectEqualStrings("no", robots.content_signals[0].value); +} + +test "Robots: content-signal absent yields empty list" { + const allocator = std.testing.allocator; + + var robots = try Robots.fromBytes(allocator, "MyBot", + \\User-agent: * + \\Disallow: /admin/ + \\ + ); + defer robots.deinit(allocator); + + try std.testing.expectEqual(0, robots.content_signals.len); + try std.testing.expect(robots.isAllowed("/admin/") == false); +} + +test "Robots: content-signal prefers specific user-agent over wildcard" { + const allocator = std.testing.allocator; + + const file = + \\User-agent: * + \\Content-Signal: ai-train=yes + \\ + \\User-agent: MyBot + \\Content-Signal: ai-train=no + \\ + ; + + var mine = try Robots.fromBytes(allocator, "MyBot", file); + defer mine.deinit(allocator); + try std.testing.expectEqual(1, mine.content_signals.len); + try std.testing.expectEqualStrings("no", mine.content_signals[0].value); + + var other = try Robots.fromBytes(allocator, "OtherBot", file); + defer other.deinit(allocator); + try std.testing.expectEqual(1, other.content_signals.len); + try std.testing.expectEqualStrings("yes", other.content_signals[0].value); +} + +test "Robots: RobotStore.getContentSignals round-trips" { + const allocator = std.testing.allocator; + + var store = RobotStore.init(allocator); + defer store.deinit(); + + const robots = try store.robotsFromBytes("MyBot", + \\User-agent: * + \\Content-Signal: ai-train=no + \\ + ); + try store.put("https://example.com/robots.txt", robots); + + const signals = store.getContentSignals("https://example.com/robots.txt").?; + try std.testing.expectEqual(1, signals.len); + try std.testing.expectEqualStrings("ai-train", signals[0].name); + try std.testing.expectEqualStrings("no", signals[0].value); + + // Unknown host has no stored robots. + try std.testing.expectEqual(null, store.getContentSignals("https://other.com/robots.txt")); +} diff --git a/src/network/http.zig b/src/network/http.zig index bec25af53..3664bac48 100644 --- a/src/network/http.zig +++ b/src/network/http.zig @@ -99,6 +99,34 @@ pub const Headers = struct { self.headers = updated_headers; } + // Adds `header` ("Name: Value"), replacing any existing header with the + // same case-insensitive name. Caller-supplied headers (CDP + // Network.setExtraHTTPHeaders) must override built-in defaults like + // User-Agent; a plain append produces a duplicate that libcurl silently + // collapses to the first occurrence, so the override would be dropped. + pub fn set(self: *Headers, header: [*c]const u8) !void { + const new = parseHeader(std.mem.span(@as([*:0]const u8, @ptrCast(header)))) orelse { + // No colon: nothing to match against, fall back to append. + return self.add(header); + }; + + var rebuilt: ?*libcurl.CurlSList = null; + errdefer libcurl.curl_slist_free_all(rebuilt); + + var node = self.headers; + while (node) |n| : (node = n.*.next) { + const data = @as([*:0]const u8, @ptrCast(n.*.data)); + if (parseHeader(std.mem.span(data))) |existing| { + if (std.ascii.eqlIgnoreCase(existing.name, new.name)) continue; + } + rebuilt = libcurl.curl_slist_append(rebuilt, data) orelse return error.OutOfMemory; + } + rebuilt = libcurl.curl_slist_append(rebuilt, header) orelse return error.OutOfMemory; + + libcurl.curl_slist_free_all(self.headers); + self.headers = rebuilt; + } + pub fn parseHeader(header_str: []const u8) ?Header { const colon_pos = std.mem.indexOfScalar(u8, header_str, ':') orelse return null; @@ -689,6 +717,53 @@ fn makeSockAddrV4(ip: [4]u8) libcurl.CurlSockAddr { } const testing = @import("../testing.zig"); + +fn findHeader(headers: Headers, name: []const u8) struct { count: usize, value: []const u8 } { + var count: usize = 0; + var value: []const u8 = ""; + var it = headers.iterator(); + while (it.next()) |h| { + if (std.ascii.eqlIgnoreCase(h.name, name)) { + count += 1; + value = h.value; + } + } + return .{ .count = count, .value = value }; +} + +test "Headers.set replaces an existing header instead of duplicating it" { + var headers = try Headers.init("User-Agent: Lightpanda/1.0"); + defer headers.deinit(); + + try headers.set("User-Agent: Custom/1.0"); + + const ua = findHeader(headers, "User-Agent"); + try testing.expectEqual(@as(usize, 1), ua.count); + try testing.expectString("Custom/1.0", ua.value); +} + +test "Headers.set matches header names case-insensitively" { + var headers = try Headers.init("User-Agent: Lightpanda/1.0"); + defer headers.deinit(); + + try headers.set("user-agent: Custom/1.0"); + + const ua = findHeader(headers, "User-Agent"); + try testing.expectEqual(@as(usize, 1), ua.count); + try testing.expectString("Custom/1.0", ua.value); +} + +test "Headers.set adds a new header and preserves defaults" { + var headers = try Headers.init("User-Agent: Lightpanda/1.0"); + defer headers.deinit(); + + try headers.set("X-Custom: yes"); + + try testing.expectEqual(@as(usize, 1), findHeader(headers, "X-Custom").count); + try testing.expectEqual(@as(usize, 1), findHeader(headers, "User-Agent").count); + try testing.expectEqual(@as(usize, 1), findHeader(headers, "Accept-Language").count); +} + test "opensocketCallback: private IPv4 returns CURL_SOCKET_BAD" { const lf: testing.LogFilter = .init(&.{.http}); defer lf.deinit();