agent: add Google Vertex AI support

Supports both express mode (via VERTEX_API_KEY) and project mode
(via GOOGLE_CLOUD_PROJECT and gcloud auth). Handles automatic
OAuth token retrieval and refresh.
This commit is contained in:
Adrià Arrufat
2026-07-03 20:23:51 +02:00
parent 933dc61bf6
commit ff7545a8c3
4 changed files with 147 additions and 24 deletions

View File

@@ -168,10 +168,11 @@ Run `/save` to export one from your current session, then replay it with
you can prototype with the LLM and ship the output to production without a
model at runtime.
It supports Anthropic, OpenAI, Gemini, Hugging Face, and local models via
Ollama. You can also run without an LLM using `--no-llm`, which drops you into
the REPL. See the [agent documentation](https://lightpanda.io/docs/usage/agent)
for the full reference.
It supports Anthropic, OpenAI, Gemini, Google Vertex AI, Hugging Face, and
local models via Ollama. You can also run without an LLM using `--no-llm`,
which drops you into the REPL. See the
[agent documentation](https://lightpanda.io/docs/usage/agent) for the full
reference.
```console
./lightpanda agent # auto-detects API key from env
@@ -179,6 +180,8 @@ for the full reference.
./lightpanda agent --no-llm # basic REPL, no LLM
./lightpanda agent session.js # run a recorded script
./lightpanda agent --provider gemini --task "..." # force a specific provider
VERTEX_API_KEY=... ./lightpanda agent --provider vertex # Vertex AI, express mode
GOOGLE_CLOUD_PROJECT=my-proj ./lightpanda agent --provider vertex # Vertex AI, token via gcloud auth
```
### Native MCP and skill

View File

@@ -35,8 +35,8 @@
.hash = "sqlite3-3.51.0-DMxLWssOAABZ8cAvU_LfBIbp0kZjm824PU8sSLXpEDdr",
},
.zenai = .{
.url = "git+https://github.com/lightpanda-io/zenai.git#a8c031222f7545c363b91197394b50024d370a7e",
.hash = "zenai-0.0.0-iOY_VB6zBABxSlIgM38xQQv53QEzy8PuDgHopcQ4mwbV",
.url = "git+https://github.com/lightpanda-io/zenai.git#e58635146733157eb4936388b21a071fcabfa687",
.hash = "zenai-0.0.0-iOY_VIf0BADH7PvSOlnVf564krXnAlns_1IIhh9ZDdwz",
},
.isocline = .{
.url = "git+https://github.com/arrufat/isocline?ref=lightpanda#832a9fe25f5f4458fcc47b5acc7c21db669c2f47",

View File

@@ -222,6 +222,9 @@ const synthesis_prompt =
allocator: std.mem.Allocator,
ai_client: ?zenai.provider.Client,
model_credentials: ?Credentials,
/// Allocated credentials key (Vertex gcloud token) — other keys are unowned
/// env pointers. The AI client references it: free only after client deinit.
owned_key: ?[:0]const u8,
/// True when the no-LLM state is a persisted preference (remembered null
/// provider or runtime `/provider null`), so `reportSaved` writes
/// `provider = null`. A transient `--no-llm` run leaves it false so saving
@@ -340,6 +343,8 @@ pub fn init(allocator: std.mem.Allocator, app: *App, opts: Config.Agent) !*Agent
if (banner_before) welcome.print(resolve);
const resolved: ?settings.ResolvedProvider = if (resolve) try settings.resolveCredentials(allocator, opts, remembered, will_repl) else null;
// Before the ai_client errdefer, so on unwind the client goes first.
errdefer if (resolved) |r| if (r.key_owned) allocator.free(r.credentials.key);
if (will_repl and !banner_before and resolved != null) welcome.print(resolve);
const llm: ?Credentials = if (resolved) |r| r.credentials else null;
@@ -394,6 +399,7 @@ pub fn init(allocator: std.mem.Allocator, app: *App, opts: Config.Agent) !*Agent
.allocator = allocator,
.ai_client = null,
.model_credentials = llm,
.owned_key = if (resolved) |r| (if (r.key_owned) r.credentials.key else null) else null,
.no_llm_persisted = remembered_no_llm,
.model_base_url = opts.base_url,
.model_completions = null,
@@ -455,6 +461,7 @@ pub fn deinit(self: *Agent) void {
self.browser.deinit();
self.notification.deinit();
if (self.ai_client) |ai_client| ai_client.deinit(self.allocator);
if (self.owned_key) |k| self.allocator.free(k);
self.allocator.free(self.model);
for (self.available_providers) |p| self.allocator.free(p);
self.allocator.free(self.available_providers);
@@ -915,11 +922,28 @@ fn handleProvider(self: *Agent, _: std.mem.Allocator, rest: []const u8) void {
self.terminal.printError("unknown provider: {s} (or 'null' to disable the LLM)", .{trimmed});
return;
};
if (self.model_credentials) |current| if (provider == current.provider) {
// Re-selecting vertex falls through — that's the token-refresh path.
const vertex_project = provider == .vertex and settings.vertexProjectMode();
if (self.model_credentials) |current| if (provider == current.provider and !vertex_project) {
self.terminal.printInfo("provider: {s}", .{@tagName(provider)});
return;
};
if (vertex_project) {
const token = settings.gcloudAccessToken(self.allocator) catch |err| {
self.terminal.printError("could not obtain a Vertex access token: {s} (details above)", .{@errorName(err)});
return;
};
self.setProvider(.{ .provider = .vertex, .key = token }, token) catch |err| {
self.allocator.free(token);
self.terminal.printError("failed to set provider: {s}", .{@errorName(err)});
};
return;
}
const key = zenai.provider.envApiKey(provider) orelse {
if (provider == .vertex) {
self.terminal.printError("vertex needs VERTEX_API_KEY (express mode) or GOOGLE_CLOUD_PROJECT (project mode, token via gcloud)", .{});
return;
}
self.terminal.printError("no API key for {s}; set {s}", .{ @tagName(provider), zenai.provider.envVarName(provider) });
return;
};
@@ -932,7 +956,7 @@ fn handleProvider(self: *Agent, _: std.mem.Allocator, rest: []const u8) void {
self.terminal.printError("no llama.cpp server with a loaded model at {s}", .{self.model_base_url orelse zenai.provider.llama_cpp_default_base_url});
return;
}
self.setProvider(.{ .provider = provider, .key = key }) catch |err| {
self.setProvider(.{ .provider = provider, .key = key }, null) catch |err| {
self.terminal.printError("failed to set provider: {s}", .{@errorName(err)});
};
}
@@ -942,6 +966,8 @@ fn handleProvider(self: *Agent, _: std.mem.Allocator, rest: []const u8) void {
fn disableProvider(self: *Agent) void {
if (self.ai_client) |client| client.deinit(self.allocator);
self.ai_client = null;
if (self.owned_key) |k| self.allocator.free(k);
self.owned_key = null;
self.model_credentials = null;
self.model_completions = null;
self.no_llm_persisted = true;
@@ -955,12 +981,18 @@ fn hfBillTo(provider: Config.AiProvider) ?[]const u8 {
return std.posix.getenv("HF_BILL_TO");
}
fn setProvider(self: *Agent, credentials: Credentials) !void {
/// `owned_key` transfers ownership of an allocated `credentials.key` (Vertex
/// gcloud token) on success; on error the caller still owns it.
fn setProvider(self: *Agent, credentials: Credentials, owned_key: ?[:0]const u8) !void {
const new_client = try zenai.provider.Client.init(self.allocator, credentials, .{ .base_url = self.model_base_url, .retry_policy = .long_running, .bill_to = hfBillTo(credentials.provider) });
errdefer new_client.deinit(self.allocator);
const new_model = try self.allocator.dupe(u8, zenai.provider.defaultModel(credentials.provider));
// A same-provider re-select (vertex token refresh) must not reset the model.
const same_provider = if (self.model_credentials) |c| c.provider == credentials.provider else false;
const new_model = try self.allocator.dupe(u8, if (same_provider) self.model else zenai.provider.defaultModel(credentials.provider));
if (self.ai_client) |client| client.deinit(self.allocator);
if (self.owned_key) |k| self.allocator.free(k);
self.owned_key = owned_key;
new_client.setInterrupt(&self.http_interrupt);
self.ai_client = new_client;
self.model_credentials = credentials;
@@ -1527,10 +1559,17 @@ fn recordSlashToolCall(
fn formatApiError(self: *Agent, client: zenai.provider.Client, err: anyerror) []const u8 {
const e = client.lastError();
const status = e.status orelse return @errorName(err);
const hint = if (status == 401 and client == .vertex)
if (self.owned_key != null)
" (Vertex token may have expired; run /provider vertex to refresh)"
else
" (Vertex express mode needs an express API key — a Gemini Developer key won't work)"
else
"";
if (e.message) |m| {
if (std.fmt.bufPrint(&self.api_error_buf, "HTTP {d} — {s}", .{ status, m })) |s| return s else |_| {}
if (std.fmt.bufPrint(&self.api_error_buf, "HTTP {d} — {s}{s}", .{ status, m, hint })) |s| return s else |_| {}
}
return std.fmt.bufPrint(&self.api_error_buf, "HTTP {d}", .{status}) catch @errorName(err);
return std.fmt.bufPrint(&self.api_error_buf, "HTTP {d}{s}", .{ status, hint }) catch @errorName(err);
}
/// Returned text lives in `conversation.arena`, valid only until the next prune.
@@ -1797,10 +1836,16 @@ pub fn listModels(allocator: std.mem.Allocator, opts: Config.Agent) !void {
}
const resolved = (try settings.resolveCredentials(allocator, opts, null, false)) orelse return error.MissingProvider;
const llm = resolved.credentials;
defer if (resolved.key_owned) allocator.free(llm.key);
var arena: std.heap.ArenaAllocator = .init(allocator);
defer arena.deinit();
const ids = try zenai.provider.listChatModelIds(allocator, arena.allocator(), llm.provider, llm.key, opts.base_url);
const ids = zenai.provider.listChatModelIds(allocator, arena.allocator(), llm.provider, llm.key, opts.base_url) catch |err| {
if (llm.provider == .vertex and !settings.vertexProjectMode()) {
std.debug.print("Vertex express mode cannot list models (the endpoint requires OAuth); set GOOGLE_CLOUD_PROJECT for project mode.\n", .{});
}
return err;
};
var stdout_file = std.fs.File.stdout().writer(&.{});
const w = &stdout_file.interface;

View File

@@ -30,7 +30,7 @@ const Terminal = @import("Terminal.zig");
const string = @import("../string.zig");
const Credentials = zenai.provider.Credentials;
pub const api_keys_hint = "ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, HF_TOKEN, AI_GATEWAY_API_KEY, or MISTRAL_API_KEY";
pub const api_keys_hint = "ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, HF_TOKEN, AI_GATEWAY_API_KEY, or MISTRAL_API_KEY (Vertex AI: VERTEX_API_KEY, or GOOGLE_CLOUD_PROJECT via gcloud)";
/// Determine which provider to use and read its env key. Returns null
/// only when no `--provider` was given AND no env key exists (the caller
@@ -38,6 +38,9 @@ pub const api_keys_hint = "ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, HF
pub const ResolvedProvider = struct {
credentials: Credentials,
source: enum { flag, remembered, detected, picked },
/// Key allocated (Vertex gcloud token) rather than an env pointer; the
/// caller frees it, only after the client that references it is gone.
key_owned: bool = false,
};
/// Probe a keyless local provider (Ollama, llama.cpp): its env key is a
@@ -52,21 +55,69 @@ pub fn detectLocalProvider(allocator: std.mem.Allocator, tag: Config.AiProvider,
return .{ .provider = tag, .key = key };
}
/// With GOOGLE_CLOUD_PROJECT set, zenai's client always sends Bearer auth —
/// an API key can never work, so the credential must be an OAuth token.
pub fn vertexProjectMode() bool {
return std.posix.getenv("GOOGLE_CLOUD_PROJECT") != null;
}
/// Caller owns the result. Failure prints gcloud's own stderr so the real
/// cause (not logged in, missing SDK) reaches the user.
pub fn gcloudAccessToken(allocator: std.mem.Allocator) ![:0]const u8 {
const result = std.process.Child.run(.{
.allocator = allocator,
.argv = &.{ "gcloud", "auth", "print-access-token" },
.max_output_bytes = 64 * 1024,
}) catch |err| {
if (err == error.FileNotFound) {
std.debug.print("gcloud not found on PATH; install the Google Cloud SDK, or unset GOOGLE_CLOUD_PROJECT to use Vertex express mode with GOOGLE_API_KEY.\n", .{});
return error.GcloudNotFound;
}
return err;
};
defer allocator.free(result.stdout);
defer allocator.free(result.stderr);
const failed = switch (result.term) {
.Exited => |code| code != 0,
else => true,
};
const token = std.mem.trim(u8, result.stdout, &std.ascii.whitespace);
if (failed or token.len == 0) {
std.debug.print("`gcloud auth print-access-token` failed:\n{s}", .{result.stderr});
return error.GcloudTokenFailed;
}
return allocator.dupeZ(u8, token);
}
/// True when a non-Ollama provider key is available (flag, remembered, or
/// env-detected). Skips the Ollama probe so it isn't run twice at startup; the
/// interactive picker only fires on detected keys, which this still catches.
pub fn hasDetectableKey(opts: Config.Agent, remembered: ?Remembered) bool {
if (opts.provider) |p| return zenai.provider.envApiKey(p) != null;
if (remembered) |r| if (r.provider) |p| if (zenai.provider.envApiKey(p)) |_| return true;
if (opts.provider) |p| return zenai.provider.envApiKey(p) != null or (p == .vertex and vertexProjectMode());
if (remembered) |r| if (r.provider) |p| {
if (zenai.provider.envApiKey(p) != null) return true;
if (p == .vertex and vertexProjectMode()) return true;
};
var buf: [zenai.provider.default_candidates.len]Credentials = undefined;
return zenai.provider.detectKeys(&buf, zenai.provider.default_candidates).len > 0;
return availableProviders(&buf).len > 0;
}
/// Precedence: `--provider` > remembered (if its key is still set) > first
/// detected. Null means no key at all (the reason is already printed).
pub fn resolveCredentials(allocator: std.mem.Allocator, opts: Config.Agent, remembered: ?Remembered, allow_pick: bool) !?ResolvedProvider {
if (opts.provider) |p| {
if (p == .vertex and vertexProjectMode()) {
const token = try gcloudAccessToken(allocator);
return .{ .credentials = .{ .provider = p, .key = token }, .source = .flag, .key_owned = true };
}
const key = zenai.provider.envApiKey(p) orelse {
if (p == .vertex) {
std.debug.print(
"Vertex needs VERTEX_API_KEY (express mode) or GOOGLE_CLOUD_PROJECT (project mode, token via gcloud) — or pass --no-llm for the basic REPL.\n",
.{},
);
return error.MissingApiKey;
}
std.debug.print(
"Missing API key for --provider {s}: set {s} — or pass --no-llm for the basic REPL.\n",
.{ @tagName(p), zenai.provider.envVarName(p) },
@@ -76,12 +127,19 @@ pub fn resolveCredentials(allocator: std.mem.Allocator, opts: Config.Agent, reme
return .{ .credentials = .{ .provider = p, .key = key }, .source = .flag };
}
if (remembered) |r| if (r.provider) |p| if (zenai.provider.envApiKey(p)) |key| {
return .{ .credentials = .{ .provider = p, .key = key }, .source = .remembered };
if (remembered) |r| if (r.provider) |p| {
if (p == .vertex and vertexProjectMode()) {
// On failure the reason is already printed; fall through to detection.
if (gcloudAccessToken(allocator)) |token| {
return .{ .credentials = .{ .provider = p, .key = token }, .source = .remembered, .key_owned = true };
} else |_| {}
} else if (zenai.provider.envApiKey(p)) |key| {
return .{ .credentials = .{ .provider = p, .key = key }, .source = .remembered };
}
};
var buf: [zenai.provider.default_candidates.len]Credentials = undefined;
const found = zenai.provider.detectKeys(&buf, zenai.provider.default_candidates);
const found = availableProviders(&buf);
if (found.len == 0) {
if (detectLocalProvider(allocator, .ollama, opts.base_url)) |creds| {
return .{ .credentials = creds, .source = .detected };
@@ -99,16 +157,26 @@ pub fn resolveCredentials(allocator: std.mem.Allocator, opts: Config.Agent, reme
// A single key needs no choice; non-interactive callers (--list-models,
// one-shot tasks, pipes) must not block on a prompt — take the first.
if (!allow_pick or found.len == 1 or !Terminal.interactiveTty()) {
return .{ .credentials = found[0], .source = .detected };
return try finishResolved(allocator, found[0], .detected);
}
var names: [zenai.provider.default_candidates.len][:0]const u8 = undefined;
for (found, 0..) |cred, i| names[i] = @tagName(cred.provider);
std.debug.print("\n", .{});
const idx = Terminal.promptNumberedChoice(" Select a provider:", names[0..found.len], 0) catch {
return .{ .credentials = found[0], .source = .detected };
return try finishResolved(allocator, found[0], .detected);
};
return .{ .credentials = found[idx], .source = .picked };
return try finishResolved(allocator, found[idx], .picked);
}
/// Swaps the placeholder key of a detected project-mode Vertex for a real
/// gcloud token.
fn finishResolved(allocator: std.mem.Allocator, credentials: Credentials, source: @FieldType(ResolvedProvider, "source")) !ResolvedProvider {
if (credentials.provider == .vertex and vertexProjectMode()) {
const token = try gcloudAccessToken(allocator);
return .{ .credentials = .{ .provider = .vertex, .key = token }, .source = source, .key_owned = true };
}
return .{ .credentials = credentials, .source = source };
}
pub const remembered_path = ".lp-agent.zon";
@@ -157,8 +225,15 @@ pub fn saveRemembered(remembered: Remembered) !void {
/// Cloud providers with a key set. Ollama is excluded — its availability needs
/// a live probe (`detectLocalProvider`), too costly for an unconditional startup scan.
/// Vertex project mode joins with a placeholder key — no subprocess during a
/// scan; the gcloud token is fetched on selection (`finishResolved`).
pub fn availableProviders(buf: []Credentials) []Credentials {
return zenai.provider.detectKeys(buf, zenai.provider.default_candidates);
const found = zenai.provider.detectKeys(buf, zenai.provider.default_candidates);
if (zenai.provider.useVertex() and vertexProjectMode() and found.len < buf.len) {
buf[found.len] = .{ .provider = .vertex, .key = "" };
return buf[0 .. found.len + 1];
}
return found;
}
pub fn resolveModelName(opts: Config.Agent, resolved: ?ResolvedProvider, remembered: ?Remembered) []const u8 {