Files
LocalAI/tests/e2e/distributed/testhelpers_test.go
T
Ettore Di Giacinto 6e2be0fd5a refactor(distributed): move job and agent fan-out onto the PostgreSQL carrier
Five of the six families whose subscriber is an open HTTP response rather
than a process-lifetime cache now travel on pgbus: jobs.<id>.progress,
jobs.<id>.result, jobs.<id>.cancel, agent.<name>.events.<user> and
responses.<id>.cancel. Both ends of each move together, so there is no
state where a publisher is on one carrier and its subscriber on the other.

agent.<name>.cancel does NOT move, and the plan was wrong about why. Its
only subscriber in the tree is the agent worker, which has no database and
so cannot join the PostgreSQL carrier at all. Publishing that cancel on
pgbus would have lost every cancel of a worker-run agent while returning
nil, which reports a cancel that reached nobody as a cancel that was sent.
EventBridge now names its cancel carrier separately, a frontend replica
sets it to the carrier the worker reads, and it stays there until a cancel
rides the worker's tunnel like every other verb addressed to a worker.

The carrier drops at 256 rather than blocking, which is not safe on its own
for a result: a lost result has no successor message. It is not the only
path. The claiming replica persists the terminal line before it releases
the claim, and an open progress stream re-reads the job row once after
subscribing and then periodically, so a dropped terminal broadcast costs
promptness and never the answer.

Both per-request subscriptions close in a defer instead of on one return
path, and pgbus grows Subscribers() so the leak they would otherwise cause
can be asserted. It has no other symptom: only the first subscriber of a
channel issues a LISTEN, so a leaked filter just adds one closure per
notification for every stream the replica has ever served. Subscribe now
issues its LISTEN before it registers, which makes that count a readiness
signal rather than a figure to compare against itself.

Two rules that were stated at several sites and pinned at none are now one
each. The re-broadcaster is built beside the dispatcher and the bridge and
handed to the dispatch loop, so no line is left that can point it at a
carrier nobody subscribes to while every spec stays green. The set of
statuses a job never leaves is one exported set that the SSE bridge and the
store both read. The last hand-written subject filter in production code
became messaging.SubjectAgentEventsWildcard.

Assisted-by: Claude Opus 5 [claude-code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
2026-09-27 03:05:13 +00:00

265 lines
8.1 KiB
Go

package distributed_test
import (
"context"
"fmt"
"net/url"
"strings"
"sync/atomic"
"time"
"github.com/mudler/LocalAI/core/services/messaging"
"github.com/mudler/LocalAI/core/services/pgbus"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/testcontainers/testcontainers-go"
tcnats "github.com/testcontainers/testcontainers-go/modules/nats"
tcpostgres "github.com/testcontainers/testcontainers-go/modules/postgres"
"github.com/testcontainers/testcontainers-go/wait"
"gorm.io/driver/postgres"
"gorm.io/gorm"
gormlogger "gorm.io/gorm/logger"
)
// TestInfra holds shared test containers and connection strings.
//
// PGContainer and NATSContainer are the SUITE-WIDE containers, shared by every
// spec. Never call Terminate or Stop on them from a spec: it ends the run for
// everything after it. They are exposed only because nats_jwt_helpers_test.go
// builds its own TestInfra around a dedicated NATS container.
type TestInfra struct {
Ctx context.Context
PGContainer *tcpostgres.PostgresContainer
NATSContainer *tcnats.NATSContainer
PGURL string
NatsURL string
NC *messaging.Client
}
// Containers are suite-scoped, not spec-scoped. Starting a Postgres (~10s) and a
// NATS (~3.5s) per spec cost roughly 48 minutes of pure startup across the 213
// specs behind SetupInfra, which is why this suite was never wired into CI.
// Isolation now comes from a database per spec (~67ms), which is what the dbName
// argument was always describing.
//
// Plain BeforeSuite rather than SynchronizedBeforeSuite is deliberate: under
// `ginkgo -p` each process gets its own container pair, which keeps NATS subjects
// isolated per process. A single shared NATS across parallel processes would let
// specs on different processes see each other's messages on the same subject.
var (
suitePG *tcpostgres.PostgresContainer
suiteNATS *tcnats.NATSContainer
suitePGDSN string
suiteNatsURL string
dbCounter atomic.Int64
)
var _ = BeforeSuite(func() {
ctx := context.Background()
var err error
suitePG, err = tcpostgres.Run(ctx, "postgres:16-alpine",
tcpostgres.WithDatabase("localai_suite"),
tcpostgres.WithUsername("test"),
tcpostgres.WithPassword("test"),
testcontainers.WithWaitStrategy(
wait.ForLog("database system is ready to accept connections").
WithOccurrence(2).
WithStartupTimeout(90*time.Second),
),
)
Expect(err).ToNot(HaveOccurred())
suitePGDSN, err = suitePG.ConnectionString(ctx, "sslmode=disable")
Expect(err).ToNot(HaveOccurred())
suiteNATS, err = tcnats.Run(ctx, "nats:2-alpine")
Expect(err).ToNot(HaveOccurred())
suiteNatsURL, err = suiteNATS.ConnectionString(ctx)
Expect(err).ToNot(HaveOccurred())
})
var _ = AfterSuite(func() {
ctx := context.Background()
if suitePG != nil {
_ = suitePG.Terminate(ctx)
}
if suiteNATS != nil {
_ = suiteNATS.Terminate(ctx)
}
})
// sanitizeDBName maps a spec-supplied label onto a legal unquoted Postgres
// identifier, leaving headroom for the uniqueness suffix appended by SetupInfra.
func sanitizeDBName(name string) string {
var b strings.Builder
for _, r := range strings.ToLower(name) {
switch {
case r >= 'a' && r <= 'z', r >= '0' && r <= '9', r == '_':
b.WriteRune(r)
default:
b.WriteRune('_')
}
}
out := strings.Trim(b.String(), "_")
if out == "" {
out = "spec"
}
// Postgres identifiers cap at 63 bytes; reserve the rest for "_<counter>".
if len(out) > 50 {
out = out[:50]
}
return out
}
// replaceDBName swaps the database component of a DSN, preserving credentials,
// host, port and query parameters.
func replaceDBName(dsn, name string) string {
GinkgoHelper()
u, err := url.Parse(dsn)
Expect(err).ToNot(HaveOccurred())
u.Path = "/" + name
return u.String()
}
// tryAdminDB opens a short-lived connection to the suite's maintenance
// database. Cleanup paths use this rather than adminDB: once connections are
// scarce, a fatal assertion here would convert one Postgres hiccup into a
// suite-wide cascade that buries the original failure.
//
// CREATE/DROP DATABASE cannot run inside a transaction or against the target
// database itself, so every call gets its own connection and closes it.
func tryAdminDB() (*gorm.DB, error) {
db, err := gorm.Open(postgres.Open(suitePGDSN), &gorm.Config{Logger: gormlogger.Discard})
if err != nil {
return nil, fmt.Errorf("connecting to the suite maintenance database: %w", err)
}
return db, nil
}
func adminDB() *gorm.DB {
GinkgoHelper()
db, err := tryAdminDB()
Expect(err).ToNot(HaveOccurred())
return db
}
func closeDB(db *gorm.DB) {
if db == nil {
return
}
if sqlDB, err := db.DB(); err == nil {
_ = sqlDB.Close()
}
}
// SetupInfra provisions a dedicated database on the suite-scoped Postgres and
// returns a client connected to the suite-scoped NATS. Call in BeforeEach;
// cleanup is registered with DeferCleanup.
func SetupInfra(dbName string) *TestInfra {
GinkgoHelper()
Expect(suitePG).ToNot(BeNil(), "SetupInfra called before BeforeSuite started the shared containers")
infra := &TestInfra{
Ctx: context.Background(),
PGContainer: suitePG,
NATSContainer: suiteNATS,
NatsURL: suiteNatsURL,
}
db := fmt.Sprintf("%s_%d", sanitizeDBName(dbName), dbCounter.Add(1))
// Scoped so a failed CREATE cannot leak the pool: the assertion panics, and a
// leaked pgx pool per failing spec exhausts the server's connection limit.
func() {
admin := adminDB()
defer closeDB(admin)
Expect(admin.Exec(fmt.Sprintf("CREATE DATABASE %q", db)).Error).To(Succeed())
}()
// Registered before anything else can fail: a NATS connect error below would
// otherwise leave the database behind for the rest of the suite.
DeferCleanup(func() {
if infra.NC != nil {
infra.NC.Close()
}
drop, err := tryAdminDB()
if err != nil {
AddReportEntry("drop database skipped", fmt.Sprintf("%s: %v", db, err))
return
}
defer closeDB(drop)
// FORCE terminates any connection the spec left open (Postgres 13+).
if err := drop.Exec(fmt.Sprintf("DROP DATABASE IF EXISTS %q WITH (FORCE)", db)).Error; err != nil {
AddReportEntry("drop database failed", fmt.Sprintf("%s: %v", db, err))
}
})
infra.PGURL = replaceDBName(suitePGDSN, db)
var err error
infra.NC, err = messaging.New(infra.NatsURL)
Expect(err).ToNot(HaveOccurred())
return infra
}
// SetupNATSOnly returns a client on the suite-scoped NATS for specs that need no
// database.
func SetupNATSOnly() *TestInfra {
GinkgoHelper()
Expect(suiteNATS).ToNot(BeNil(), "SetupNATSOnly called before BeforeSuite started the shared containers")
infra := &TestInfra{
Ctx: context.Background(),
NATSContainer: suiteNATS,
NatsURL: suiteNatsURL,
}
var err error
infra.NC, err = messaging.New(infra.NatsURL)
Expect(err).ToNot(HaveOccurred())
DeferCleanup(func() {
if infra.NC != nil {
infra.NC.Close()
}
})
return infra
}
// FlushNATS ensures all subscriptions are registered server-side before publishing.
func FlushNATS(nc *messaging.Client) {
GinkgoHelper()
Expect(nc.Conn().Flush()).To(Succeed())
}
// Bus opens a broadcast carrier on THIS spec's database.
//
// It is the carrier the job, agent and response families travel on, and it is
// what these specs must build their dispatchers and bridges with. Publishing on
// one carrier while the subscriber reads another is a defect with no error
// anywhere: the publish succeeds and the SSE stream is simply empty, so a spec
// that used the NATS client here would keep passing after production had gone
// silent.
//
// Every call returns a SEPARATE carrier on the same database, so a spec can
// build two and assert across them, which is the shape a deployment has.
func (i *TestInfra) Bus() *pgbus.Bus {
GinkgoHelper()
Expect(i.PGURL).ToNot(BeEmpty(), "Bus needs a database; use SetupInfra rather than SetupNATSOnly")
db, err := gorm.Open(postgres.Open(i.PGURL), &gorm.Config{Logger: gormlogger.Default.LogMode(gormlogger.Silent)})
Expect(err).ToNot(HaveOccurred())
Expect(pgbus.Migrate(i.Ctx, db)).To(Succeed())
bus, err := pgbus.New(i.Ctx, pgbus.Config{DSN: i.PGURL, DB: db})
Expect(err).ToNot(HaveOccurred())
DeferCleanup(bus.Close)
return bus
}