mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 17:44:30 -04:00
Five of the six families whose subscriber is an open HTTP response rather than a process-lifetime cache now travel on pgbus: jobs.<id>.progress, jobs.<id>.result, jobs.<id>.cancel, agent.<name>.events.<user> and responses.<id>.cancel. Both ends of each move together, so there is no state where a publisher is on one carrier and its subscriber on the other. agent.<name>.cancel does NOT move, and the plan was wrong about why. Its only subscriber in the tree is the agent worker, which has no database and so cannot join the PostgreSQL carrier at all. Publishing that cancel on pgbus would have lost every cancel of a worker-run agent while returning nil, which reports a cancel that reached nobody as a cancel that was sent. EventBridge now names its cancel carrier separately, a frontend replica sets it to the carrier the worker reads, and it stays there until a cancel rides the worker's tunnel like every other verb addressed to a worker. The carrier drops at 256 rather than blocking, which is not safe on its own for a result: a lost result has no successor message. It is not the only path. The claiming replica persists the terminal line before it releases the claim, and an open progress stream re-reads the job row once after subscribing and then periodically, so a dropped terminal broadcast costs promptness and never the answer. Both per-request subscriptions close in a defer instead of on one return path, and pgbus grows Subscribers() so the leak they would otherwise cause can be asserted. It has no other symptom: only the first subscriber of a channel issues a LISTEN, so a leaked filter just adds one closure per notification for every stream the replica has ever served. Subscribe now issues its LISTEN before it registers, which makes that count a readiness signal rather than a figure to compare against itself. Two rules that were stated at several sites and pinned at none are now one each. The re-broadcaster is built beside the dispatcher and the bridge and handed to the dispatch loop, so no line is left that can point it at a carrier nobody subscribes to while every spec stays green. The set of statuses a job never leaves is one exported set that the SSE bridge and the store both read. The last hand-written subject filter in production code became messaging.SubjectAgentEventsWildcard. Assisted-by: Claude Opus 5 [claude-code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
265 lines
8.1 KiB
Go
265 lines
8.1 KiB
Go
package distributed_test
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"net/url"
|
|
"strings"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"github.com/mudler/LocalAI/core/services/messaging"
|
|
"github.com/mudler/LocalAI/core/services/pgbus"
|
|
|
|
. "github.com/onsi/ginkgo/v2"
|
|
. "github.com/onsi/gomega"
|
|
|
|
"github.com/testcontainers/testcontainers-go"
|
|
tcnats "github.com/testcontainers/testcontainers-go/modules/nats"
|
|
tcpostgres "github.com/testcontainers/testcontainers-go/modules/postgres"
|
|
"github.com/testcontainers/testcontainers-go/wait"
|
|
"gorm.io/driver/postgres"
|
|
"gorm.io/gorm"
|
|
gormlogger "gorm.io/gorm/logger"
|
|
)
|
|
|
|
// TestInfra holds shared test containers and connection strings.
|
|
//
|
|
// PGContainer and NATSContainer are the SUITE-WIDE containers, shared by every
|
|
// spec. Never call Terminate or Stop on them from a spec: it ends the run for
|
|
// everything after it. They are exposed only because nats_jwt_helpers_test.go
|
|
// builds its own TestInfra around a dedicated NATS container.
|
|
type TestInfra struct {
|
|
Ctx context.Context
|
|
PGContainer *tcpostgres.PostgresContainer
|
|
NATSContainer *tcnats.NATSContainer
|
|
PGURL string
|
|
NatsURL string
|
|
NC *messaging.Client
|
|
}
|
|
|
|
// Containers are suite-scoped, not spec-scoped. Starting a Postgres (~10s) and a
|
|
// NATS (~3.5s) per spec cost roughly 48 minutes of pure startup across the 213
|
|
// specs behind SetupInfra, which is why this suite was never wired into CI.
|
|
// Isolation now comes from a database per spec (~67ms), which is what the dbName
|
|
// argument was always describing.
|
|
//
|
|
// Plain BeforeSuite rather than SynchronizedBeforeSuite is deliberate: under
|
|
// `ginkgo -p` each process gets its own container pair, which keeps NATS subjects
|
|
// isolated per process. A single shared NATS across parallel processes would let
|
|
// specs on different processes see each other's messages on the same subject.
|
|
var (
|
|
suitePG *tcpostgres.PostgresContainer
|
|
suiteNATS *tcnats.NATSContainer
|
|
suitePGDSN string
|
|
suiteNatsURL string
|
|
dbCounter atomic.Int64
|
|
)
|
|
|
|
var _ = BeforeSuite(func() {
|
|
ctx := context.Background()
|
|
var err error
|
|
|
|
suitePG, err = tcpostgres.Run(ctx, "postgres:16-alpine",
|
|
tcpostgres.WithDatabase("localai_suite"),
|
|
tcpostgres.WithUsername("test"),
|
|
tcpostgres.WithPassword("test"),
|
|
testcontainers.WithWaitStrategy(
|
|
wait.ForLog("database system is ready to accept connections").
|
|
WithOccurrence(2).
|
|
WithStartupTimeout(90*time.Second),
|
|
),
|
|
)
|
|
Expect(err).ToNot(HaveOccurred())
|
|
|
|
suitePGDSN, err = suitePG.ConnectionString(ctx, "sslmode=disable")
|
|
Expect(err).ToNot(HaveOccurred())
|
|
|
|
suiteNATS, err = tcnats.Run(ctx, "nats:2-alpine")
|
|
Expect(err).ToNot(HaveOccurred())
|
|
|
|
suiteNatsURL, err = suiteNATS.ConnectionString(ctx)
|
|
Expect(err).ToNot(HaveOccurred())
|
|
})
|
|
|
|
var _ = AfterSuite(func() {
|
|
ctx := context.Background()
|
|
if suitePG != nil {
|
|
_ = suitePG.Terminate(ctx)
|
|
}
|
|
if suiteNATS != nil {
|
|
_ = suiteNATS.Terminate(ctx)
|
|
}
|
|
})
|
|
|
|
// sanitizeDBName maps a spec-supplied label onto a legal unquoted Postgres
|
|
// identifier, leaving headroom for the uniqueness suffix appended by SetupInfra.
|
|
func sanitizeDBName(name string) string {
|
|
var b strings.Builder
|
|
for _, r := range strings.ToLower(name) {
|
|
switch {
|
|
case r >= 'a' && r <= 'z', r >= '0' && r <= '9', r == '_':
|
|
b.WriteRune(r)
|
|
default:
|
|
b.WriteRune('_')
|
|
}
|
|
}
|
|
out := strings.Trim(b.String(), "_")
|
|
if out == "" {
|
|
out = "spec"
|
|
}
|
|
// Postgres identifiers cap at 63 bytes; reserve the rest for "_<counter>".
|
|
if len(out) > 50 {
|
|
out = out[:50]
|
|
}
|
|
return out
|
|
}
|
|
|
|
// replaceDBName swaps the database component of a DSN, preserving credentials,
|
|
// host, port and query parameters.
|
|
func replaceDBName(dsn, name string) string {
|
|
GinkgoHelper()
|
|
u, err := url.Parse(dsn)
|
|
Expect(err).ToNot(HaveOccurred())
|
|
u.Path = "/" + name
|
|
return u.String()
|
|
}
|
|
|
|
// tryAdminDB opens a short-lived connection to the suite's maintenance
|
|
// database. Cleanup paths use this rather than adminDB: once connections are
|
|
// scarce, a fatal assertion here would convert one Postgres hiccup into a
|
|
// suite-wide cascade that buries the original failure.
|
|
//
|
|
// CREATE/DROP DATABASE cannot run inside a transaction or against the target
|
|
// database itself, so every call gets its own connection and closes it.
|
|
func tryAdminDB() (*gorm.DB, error) {
|
|
db, err := gorm.Open(postgres.Open(suitePGDSN), &gorm.Config{Logger: gormlogger.Discard})
|
|
if err != nil {
|
|
return nil, fmt.Errorf("connecting to the suite maintenance database: %w", err)
|
|
}
|
|
return db, nil
|
|
}
|
|
|
|
func adminDB() *gorm.DB {
|
|
GinkgoHelper()
|
|
db, err := tryAdminDB()
|
|
Expect(err).ToNot(HaveOccurred())
|
|
return db
|
|
}
|
|
|
|
func closeDB(db *gorm.DB) {
|
|
if db == nil {
|
|
return
|
|
}
|
|
if sqlDB, err := db.DB(); err == nil {
|
|
_ = sqlDB.Close()
|
|
}
|
|
}
|
|
|
|
// SetupInfra provisions a dedicated database on the suite-scoped Postgres and
|
|
// returns a client connected to the suite-scoped NATS. Call in BeforeEach;
|
|
// cleanup is registered with DeferCleanup.
|
|
func SetupInfra(dbName string) *TestInfra {
|
|
GinkgoHelper()
|
|
Expect(suitePG).ToNot(BeNil(), "SetupInfra called before BeforeSuite started the shared containers")
|
|
|
|
infra := &TestInfra{
|
|
Ctx: context.Background(),
|
|
PGContainer: suitePG,
|
|
NATSContainer: suiteNATS,
|
|
NatsURL: suiteNatsURL,
|
|
}
|
|
|
|
db := fmt.Sprintf("%s_%d", sanitizeDBName(dbName), dbCounter.Add(1))
|
|
|
|
// Scoped so a failed CREATE cannot leak the pool: the assertion panics, and a
|
|
// leaked pgx pool per failing spec exhausts the server's connection limit.
|
|
func() {
|
|
admin := adminDB()
|
|
defer closeDB(admin)
|
|
Expect(admin.Exec(fmt.Sprintf("CREATE DATABASE %q", db)).Error).To(Succeed())
|
|
}()
|
|
|
|
// Registered before anything else can fail: a NATS connect error below would
|
|
// otherwise leave the database behind for the rest of the suite.
|
|
DeferCleanup(func() {
|
|
if infra.NC != nil {
|
|
infra.NC.Close()
|
|
}
|
|
drop, err := tryAdminDB()
|
|
if err != nil {
|
|
AddReportEntry("drop database skipped", fmt.Sprintf("%s: %v", db, err))
|
|
return
|
|
}
|
|
defer closeDB(drop)
|
|
// FORCE terminates any connection the spec left open (Postgres 13+).
|
|
if err := drop.Exec(fmt.Sprintf("DROP DATABASE IF EXISTS %q WITH (FORCE)", db)).Error; err != nil {
|
|
AddReportEntry("drop database failed", fmt.Sprintf("%s: %v", db, err))
|
|
}
|
|
})
|
|
|
|
infra.PGURL = replaceDBName(suitePGDSN, db)
|
|
|
|
var err error
|
|
infra.NC, err = messaging.New(infra.NatsURL)
|
|
Expect(err).ToNot(HaveOccurred())
|
|
|
|
return infra
|
|
}
|
|
|
|
// SetupNATSOnly returns a client on the suite-scoped NATS for specs that need no
|
|
// database.
|
|
func SetupNATSOnly() *TestInfra {
|
|
GinkgoHelper()
|
|
Expect(suiteNATS).ToNot(BeNil(), "SetupNATSOnly called before BeforeSuite started the shared containers")
|
|
|
|
infra := &TestInfra{
|
|
Ctx: context.Background(),
|
|
NATSContainer: suiteNATS,
|
|
NatsURL: suiteNatsURL,
|
|
}
|
|
|
|
var err error
|
|
infra.NC, err = messaging.New(infra.NatsURL)
|
|
Expect(err).ToNot(HaveOccurred())
|
|
|
|
DeferCleanup(func() {
|
|
if infra.NC != nil {
|
|
infra.NC.Close()
|
|
}
|
|
})
|
|
|
|
return infra
|
|
}
|
|
|
|
// FlushNATS ensures all subscriptions are registered server-side before publishing.
|
|
func FlushNATS(nc *messaging.Client) {
|
|
GinkgoHelper()
|
|
Expect(nc.Conn().Flush()).To(Succeed())
|
|
}
|
|
|
|
// Bus opens a broadcast carrier on THIS spec's database.
|
|
//
|
|
// It is the carrier the job, agent and response families travel on, and it is
|
|
// what these specs must build their dispatchers and bridges with. Publishing on
|
|
// one carrier while the subscriber reads another is a defect with no error
|
|
// anywhere: the publish succeeds and the SSE stream is simply empty, so a spec
|
|
// that used the NATS client here would keep passing after production had gone
|
|
// silent.
|
|
//
|
|
// Every call returns a SEPARATE carrier on the same database, so a spec can
|
|
// build two and assert across them, which is the shape a deployment has.
|
|
func (i *TestInfra) Bus() *pgbus.Bus {
|
|
GinkgoHelper()
|
|
Expect(i.PGURL).ToNot(BeEmpty(), "Bus needs a database; use SetupInfra rather than SetupNATSOnly")
|
|
|
|
db, err := gorm.Open(postgres.Open(i.PGURL), &gorm.Config{Logger: gormlogger.Default.LogMode(gormlogger.Silent)})
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(pgbus.Migrate(i.Ctx, db)).To(Succeed())
|
|
|
|
bus, err := pgbus.New(i.Ctx, pgbus.Config{DSN: i.PGURL, DB: db})
|
|
Expect(err).ToNot(HaveOccurred())
|
|
DeferCleanup(bus.Close)
|
|
return bus
|
|
}
|