mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-30 18:14:32 -04:00
A local-ai worker no longer opens a bus connection. connectNATS and its
spec are gone; Run registers once, starts its tunnel, arms /readyz on that
tunnel, and heartbeats. The worker's bus credential flags (--nats-jwt,
--nats-user-seed, --nats-require-auth, the three TLS flags) and
Config.NatsAuthRequired go with it. --nats-url stays, accepted and
ignored, so an existing worker command line still parses.
/readyz was the thing most likely to wedge a tunnel-only worker: it
required a live NATS link, so a worker with no bus would have reported
itself unready forever. nodes.NATSReadiness becomes nodes.TunnelReadiness
over a local interface{ Connected() bool }, and worker.Tunnel gains
Connected(), backed by a mutex-guarded session field the loop publishes
and clears. A closed-but-not-yet-cleared session reads as disconnected:
the loop waits for every in-flight stream before it clears the field, and
the probe must answer not-ready through that wait.
The heartbeat gate is DELETED rather than re-pointed at the tunnel. The
heartbeat is the worker's own answer that its process is alive; whether
the frontend can reach it is a separate fact the frontend already holds
and ages against LOCALAI_WORKER_RECONNECT_GRACE. Withholding the
heartbeat would report an unreachable worker as an absent one on the one
path with no grace, where the health monitor marks it offline and its
pending backend ops are deleted behind it. heartbeatLoop is given no view
of the tunnel, so a gate cannot be added back without changing its
signature.
Removing the NATS credential manager from this path also removes a defect
it carried: its refresh loop re-registered on a timer to renew a JWT, and
Register CLEARS a node's NodeModel rows. Any backend worker running on
frontend-minted credentials had its replica rows deleted roughly every
18 hours.
Of core/cli/workerregistry, everything survives. The manager is still
used in full by core/cli/agent_worker.go, which still needs NATS: Acquire,
Provider, RefreshLoop, HasCredentials and TunnelToken are all untouched.
The backend worker simply calls RegisterFullWithRetry directly now.
WorkerPermissions is documented as serving agent nodes, and its non-agent
branch narrowed to _INBOX.> on both sides. It is NOT deleted: NATS reads
an empty allow list as no restriction, so returning nil would upgrade
every JWT the frontend still mints for a backend node from its own inbox
to the whole account.
Agent workers keep the bus everywhere: their CLI flags, their
subscriptions, the agent branch of WorkerPermissions, and the compose
service with its LOCALAI_NATS_URL and depends_on: nats.
Also corrected two flags the Nodes page advertised that do not exist
(--distributed-nats, --distributed-db), and a log line plus several
comments that still named a bus the code no longer touches.
Assisted-by: Claude Opus 5 [claude-code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
91 lines
4.9 KiB
JavaScript
91 lines
4.9 KiB
JavaScript
import { test, expect } from './coverage-fixtures.js'
|
|
|
|
async function mockCluster(page, nodes) {
|
|
await page.route('**/api/nodes', route => route.fulfill({ status: 200, contentType: 'application/json', body: JSON.stringify(nodes) }))
|
|
}
|
|
|
|
test.describe('Nodes fleet roster', () => {
|
|
test('uses the fleet response without prefetching models or backends', async ({ page }) => {
|
|
const requests = []
|
|
page.on('request', request => requests.push(request.url()))
|
|
await mockCluster(page, [
|
|
{ id: 'n1', name: 'alpha', node_type: 'backend', address: '10.0.0.1:50051', status: 'healthy', model_count: 3 },
|
|
{ id: 'a1', name: 'agent-1', node_type: 'agent', address: '10.0.0.9:50051', status: 'draining', model_count: 0 },
|
|
])
|
|
await page.goto('/app/nodes')
|
|
await expect(page.getByRole('table', { name: 'Fleet nodes' })).toBeVisible({ timeout: 15_000 })
|
|
await expect(page.getByRole('tab', { name: 'Nodes' })).toHaveAttribute('aria-selected', 'true')
|
|
await page.getByRole('tab', { name: 'Nodes' }).click()
|
|
await expect(page.getByRole('row', { name: /alpha/ })).toContainText('3')
|
|
expect(requests.some(url => url.includes('/api/nodes/models'))).toBe(false)
|
|
expect(requests.some(url => /\/api\/nodes\/[^/]+\/backends/.test(url))).toBe(false)
|
|
})
|
|
|
|
test('preserves the empty worker setup experience', async ({ page }) => {
|
|
await mockCluster(page, [])
|
|
await page.goto('/app/nodes')
|
|
await expect(page.getByText('No workers registered yet')).toBeVisible({ timeout: 15_000 })
|
|
})
|
|
|
|
// A single-node server does not register the cluster routes at all, so the
|
|
// real answer is 404; 503 is what they say when mounted without a registry.
|
|
// Either way the page is about this host, with the distributed setup one
|
|
// click away rather than the whole page.
|
|
for (const status of [404, 503]) {
|
|
test(`shows this machine when the cluster API answers ${status}`, async ({ page }) => {
|
|
await page.route('**/api/nodes', route => route.fulfill({ status, body: 'unavailable' }))
|
|
await page.goto('/app/nodes')
|
|
await expect(page.getByTestId('local-machine')).toBeVisible({ timeout: 15_000 })
|
|
await expect(page.getByText('No workers registered yet')).toHaveCount(0)
|
|
await expect(page.getByTestId('scale-out')).toHaveCount(0)
|
|
await page.getByRole('button', { name: 'Add machines' }).click()
|
|
await expect(page.getByTestId('scale-out')).toContainText('Distributed mode is not enabled')
|
|
})
|
|
}
|
|
})
|
|
|
|
test.describe('Nodes join command', () => {
|
|
// The panel emits BOTH the backend and the agent join command from one
|
|
// component, so the bus flag has to differ per tab rather than be deleted.
|
|
// Backend workers connect to no NATS server; agent workers still do.
|
|
test('omits the NATS flag for a backend worker and keeps it for an agent worker', async ({ page }) => {
|
|
await mockCluster(page, [])
|
|
await page.goto('/app/nodes')
|
|
|
|
await page.getByRole('radio', { name: /^Backend$/ }).click()
|
|
const backendCli = page.locator('.p2p-cmd pre').first()
|
|
await expect(backendCli).toContainText('local-ai worker', { timeout: 15_000 })
|
|
await expect(backendCli).not.toContainText('--nats-url')
|
|
const backendDocker = page.locator('.p2p-cmd pre').nth(1)
|
|
await expect(backendDocker).not.toContainText('LOCALAI_NATS_URL')
|
|
|
|
await page.getByRole('radio', { name: /^Agent$/ }).click()
|
|
const agentCli = page.locator('.p2p-cmd pre').first()
|
|
await expect(agentCli).toContainText('local-ai agent-worker', { timeout: 15_000 })
|
|
await expect(agentCli).toContainText('--nats-url')
|
|
const agentDocker = page.locator('.p2p-cmd pre').nth(1)
|
|
await expect(agentDocker).toContainText('LOCALAI_NATS_URL')
|
|
})
|
|
|
|
test('does not advertise flags the CLI does not have', async ({ page }) => {
|
|
// The "How to Enable Distributed Mode" card renders ONLY on the disabled
|
|
// state, which the page enters when /api/nodes answers 503. Mocking a
|
|
// healthy cluster here would assert absence against a card that was never
|
|
// on the page.
|
|
await page.route('**/api/nodes', r => r.fulfill({ status: 503, contentType: 'application/json', body: '{}' }))
|
|
await page.route('**/api/nodes/models', r => r.fulfill({ status: 503, contentType: 'application/json', body: '{}' }))
|
|
await page.route('**/api/nodes/scheduling', r => r.fulfill({ status: 503, contentType: 'application/json', body: '{}' }))
|
|
await page.goto('/app/nodes')
|
|
|
|
const card = page.locator('.p2p-enable')
|
|
await expect(card).toBeVisible({ timeout: 15_000 })
|
|
// --distributed-nats and --distributed-db were never real flags; a copied
|
|
// command carrying them fails at kong before LocalAI does anything.
|
|
await expect(card).not.toContainText('--distributed-nats')
|
|
await expect(card).not.toContainText('--distributed-db')
|
|
// And the worker step no longer tells an operator to point a backend
|
|
// worker at a bus it does not dial.
|
|
await expect(card.locator('.p2p-cmd pre').nth(1)).not.toContainText('--nats-url')
|
|
})
|
|
})
|