From ff6a54ba22402a5bac528ec2eecd70ac9fbd6233 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A0=20Arrufat?= Date: Thu, 24 Sep 2026 20:38:39 +0200 Subject: [PATCH 1/3] agent: drop dated prompt patterns - Synthesis prompt: drop the one-word/one-phrase answer clamp (a benchmark format applied to every budget-exhausted REPL turn) and the no-more-tools line the request already enforces; keep the report-don't-confabulate rule. - Driver guidance: drop the HN/Reddit source heuristic, which also reached every MCP client. - Save prompt: state the comment rule once instead of repeating it in the output-format line. - goto: describe when it returns, what it returns, and when a url-taking read is the better call. --- src/agent/Agent.zig | 16 ++++++---------- src/browser/tools.zig | 11 ++++------- 2 files changed, 10 insertions(+), 17 deletions(-) diff --git a/src/agent/Agent.zig b/src/agent/Agent.zig index 5207605ec..14bbfbcbf 100644 --- a/src/agent/Agent.zig +++ b/src/agent/Agent.zig @@ -118,16 +118,12 @@ fn savePrompt(revision: bool) []const u8 { const synthesis_prompt = \\You have used your tool budget or cannot finish the exploration. - \\Give your best final answer NOW based ONLY on what you actually observed - \\via tool calls in this conversation. Do NOT fall back to prior knowledge — - \\if your snapshots show only cookie banners, 403/access-denied pages, - \\blocked search results, or empty bodies, say that explicitly - \\(e.g. "the page was blocked by a cookie wall and I could not extract X"). - \\Do not invent details that are not visible in the tool outputs above. - \\Do not call any more tools. - \\Respond with ONLY the answer — one word, one number, one short phrase, - \\or a brief honest explanation of why the page could not be read. - \\No prefix, no markdown. + \\Give your best final answer using only what the tool outputs above + \\show, not prior knowledge. If they show only cookie banners, + \\403/access-denied pages, blocked search results, or empty bodies, say + \\so plainly (e.g. "the page was blocked by a cookie wall and I could not + \\extract X") rather than filling the gap. Answer at the length the + \\question needs. ; allocator: std.mem.Allocator, diff --git a/src/browser/tools.zig b/src/browser/tools.zig index 9b8d31e52..87ca0cd7c 100644 --- a/src/browser/tools.zig +++ b/src/browser/tools.zig @@ -106,9 +106,7 @@ pub const driver_guidance = \\- Triage from `search` snippets before opening links; open only the few \\ most promising. Don't re-run a search you already ran, and skip \\ near-duplicate sources that repeat the same announcement verbatim. - \\- Stop once the gathered material answers the question. For opinion or - \\ discussion questions, a couple of high-signal threads (e.g. Hacker - \\ News, Reddit) usually beat scraping a dozen news sites. + \\- Stop once the gathered material answers the question. \\ \\Selector rules: \\- NEVER pass backendNodeId to click/fill/hover/selectOption/setChecked. @@ -179,9 +177,8 @@ pub const save_synthesis_prompt = \\list, fan out to detail pages, aggregate, return) stating what that block \\accomplishes toward the goal — NOT restating the API call. One comment per \\step, not per line; skip self-evident lines. - \\Output ONLY JavaScript source — no markdown fences and no prose outside the - \\code, but DO annotate the script with the `//` intent comments described - \\above. + \\Output the JavaScript source alone, with no markdown fences or prose + \\around it. ; /// Script-language rules for consumers that never see the full @@ -345,7 +342,7 @@ pub const Tool = enum { pub fn definition(self: Tool) Definition { return switch (self) { .goto => .{ - .description = "Navigate to a specified URL and load the page in memory so it can be reused later for info extraction.", + .description = "Navigate the current page to a URL. Returns a short status once `waitUntil` fires (default `load`), or a timeout notice; content rendered by post-load JavaScript may not be there yet (see `waitForState`). The page stays loaded for later reads and actions. To navigate and read in one call, pass `url` to `markdown`, `tree` or `html` instead; use `goto` when the next step is an action or `extract`.", .summary = "Open a URL and keep the page in memory", .input_schema = minify( \\{ From b5a30be0767ddb16f75d3a6023ed5ba8060aa26d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A0=20Arrufat?= Date: Thu, 24 Sep 2026 20:43:11 +0200 Subject: [PATCH 2/3] build: bump zenai Picks up effort=none sending thinking disabled to Anthropic (zenai#19) and the reasoning max_tokens floor on every provider (zenai#18). ErrorDetail now carries the parsed error message instead of the raw body. --- build.zig.zon | 4 ++-- src/browser/tools.zig | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/build.zig.zon b/build.zig.zon index e78035c35..fa9608e67 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -36,8 +36,8 @@ .hash = "sqlite3-3.53.2-DMxLWuAOAAA_Px0arJOIOaP4AKEu5prbsQgPMA35W1zz", }, .zenai = .{ - .url = "git+https://github.com/lightpanda-io/zenai.git#15d6e4c37b4508373ba7ae6a2f400717f4ff9d89", - .hash = "zenai-0.0.0-iOY_VLq7BgAZIQzUIZ4w8ikmwpuNaJdyJSvKYCtnVing", + .url = "git+https://github.com/lightpanda-io/zenai.git#43658858f6c677b107a0f4fb76dc92e5ac97d3a4", + .hash = "zenai-0.0.0-iOY_VKquBgBT36nAFRpM4tQki3JNZgFvTwQY6Pz-wiQx", }, .isocline = .{ .url = "git+https://github.com/arrufat/isocline#ec538faf435c616a6b38716f53980b5815c30f8a", diff --git a/src/browser/tools.zig b/src/browser/tools.zig index 87ca0cd7c..b31520d8d 100644 --- a/src/browser/tools.zig +++ b/src/browser/tools.zig @@ -1214,7 +1214,7 @@ fn apiSearch( if (client.last_error.status) |status| { log.warn(.browser, @tagName(engine.tag) ++ " non-2xx", .{ .status = status, - .body = client.last_error.body, + .message = client.last_error.message, }); } return err; From 6d6f19d6ed0f6130b4f24a7c751da3aa132c708d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A0=20Arrufat?= Date: Thu, 24 Sep 2026 20:47:29 +0200 Subject: [PATCH 3/3] tools: fill out under-described tool descriptions tree, interactiveElements, structuredData, detectForms and scroll were one or two sentences, and evaluate's script parameter had none. MCP clients that drop the server instructions only see these, so state what each returns (fields, shapes, empty cases), when to reach for it over its neighbours, and scroll's absolute-position semantics. --- src/browser/tools.zig | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/browser/tools.zig b/src/browser/tools.zig index b31520d8d..dab21d9e7 100644 --- a/src/browser/tools.zig +++ b/src/browser/tools.zig @@ -445,7 +445,7 @@ pub const Tool = enum { \\{ \\ "type": "object", \\ "properties": { - \\ "script": { "type": "string" }, + \\ "script": { "type": "string", "description": "JavaScript run in the page context. A bare trailing expression, or `return` with top-level `await`, is the result." }, \\ "url": { "type": "string", "description": "Optional URL to navigate to before evaluating." }, \\ "timeout": { "type": "integer", "description": "Optional timeout in milliseconds. Defaults to 10000." }, \\ "save": { "type": "string", "description": "Optional bridge-store key. The evaluate's return value is stored under this name and re-exposed as `lp.` to subsequent evaluates. Objects, arrays, and strings are serialized automatically — no JSON.stringify needed." } @@ -486,7 +486,7 @@ pub const Tool = enum { ), }, .tree => .{ - .description = "Simplified semantic DOM tree (role, name, value, backendNodeId per node). Pass `backendNodeId` to scope, `maxDepth` to limit depth.", + .description = "Semantic outline of the page as indented text: one node per line with its role, accessible name, value and backendNodeId, plus checked state and select options with the selected one marked. The default first read of an unfamiliar page; input and select values are already here, so no `nodeDetails` call is needed to read them. Pass `backendNodeId` to scope to a subtree and `maxDepth` to survey structure before going deeper. Read it again after any page-changing action, since the DOM it describes may have changed; use `nodeDetails` to turn a backendNodeId into a CSS selector for actions.", .summary = "Semantic DOM tree of the page", .input_schema = minify( \\{ @@ -514,17 +514,17 @@ pub const Tool = enum { ), }, .interactiveElements => .{ - .description = "Extract interactive elements from the opened page. If a url is provided, it navigates to that url first.", + .description = "List every visible interactive element on the page as a JSON array: native controls, ARIA widgets, contenteditable regions, elements with event listeners, and focusable elements. Each entry has `backendNodeId`, `tagName`, `role`, `name`, `type` (why it counts as interactive), `tabIndex`, and when present `listeners`, `disabled`, `id`, `class`, `href`, `inputType`, `value`, `elementName` and `placeholder`. Use it to survey what can be acted on; to locate one element by role or name, `findElement` is cheaper. If a url is provided, it navigates there first.", .summary = "List interactive elements on the page", .input_schema = url_params_schema, }, .structuredData => .{ - .description = "Extract structured data (like JSON-LD, OpenGraph, etc) from the opened page. If a url is provided, it navigates to that url first.", + .description = "Page metadata as JSON: `jsonLd` (each JSON-LD block as a string), `openGraph`, `twitterCard`, `meta` and `links` (key/value lists), plus `alternate` (hreflang variants) and `linkHeaders` (relations from the HTTP Link header) when present. Empty sections come back as empty arrays. Use it for publisher-declared facts such as product price, article author or canonical URL before scraping the visible text for them. If a url is provided, it navigates there first.", .summary = "Extract JSON-LD / OpenGraph data", .input_schema = url_params_schema, }, .detectForms => .{ - .description = "Detect all forms on the page and return their structure including fields, types, and required status. If a url is provided, it navigates to that url first.", + .description = "List the forms on the page as JSON: each form's `backendNodeId`, `action`, `method` and `fields`, where each field has `backendNodeId`, `tagName`, `name`, `inputType`, `required`, `disabled`, and when present `value`, `placeholder` and select `options`. Use it before filling a form to see every field it expects. It returns no CSS selectors; get one per field with `nodeDetails` so the fill calls stay replayable. If a url is provided, it navigates there first.", .summary = "List forms and their fields", .input_schema = url_params_schema, }, @@ -557,7 +557,7 @@ pub const Tool = enum { ), }, .scroll => .{ - .description = "Scroll the page or a specific element. Returns the scroll position and current page URL and title.", + .description = "Scroll the window, or an element's scroll container, to an absolute position; an omitted axis keeps its current offset. Page scripts receive a `scroll` event, so content that loads on scroll (infinite feeds, lazy lists) may appear: read the page again afterwards, with `waitForState` if it is still loading. Returns the final scroll position and the current page URL and title.", .summary = "Scroll the page or an element", .input_schema = minify( \\{