fix(advisorylock): bound HeldLock.TryAcquire on a hung database

The failover prober gate calls TryAcquire with the application context,
which never ends. A database that stopped answering blocked the
scheduler goroutine for good, with the lock's mutex held. Bound the
session open and the lock query with the same timeout as Verify.

Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
Ettore Di Giacinto committed 2026-09-27 07:42:21 +00:00
1 parent 5ecaf8f792
commit 4a7c96f533
2 files changed
+70 -3

No files matched your search

+10 -3
View File
@@ -12,9 +12,10 @@ import (
"gorm.io/gorm"
)
// heldLockCheckTimeout bounds the liveness check and the unlock, so a hung
// database cannot stall the caller's loop.
const heldLockCheckTimeout = 5 * time.Second
// heldLockCheckTimeout bounds an acquire, the liveness check and the unlock,
// so a hung database cannot stall the caller's loop. A var so tests can
// shorten it.
var heldLockCheckTimeout = 5 * time.Second
// heldLockSessionSettings make the server notice a lock holder whose host
// died without closing the connection (crash, power loss, partition) within
@@ -80,6 +81,12 @@ func (l *HeldLock) TryAcquire(ctx context.Context) (bool, error) {
}
}
// Callers pass long-lived contexts (the application's, which never ends),
// and l.mu is held throughout: a hung database would otherwise block this
// caller and every other user of the lock for good. The deadline only
// covers obtaining the session, which stays usable afterwards.
ctx, cancel := context.WithTimeout(ctx, heldLockCheckTimeout)
defer cancel()
if l.conn == nil {
conn, err := l.openSession(ctx)
if err != nil {
@@ -0,0 +1,60 @@
package advisorylock
import (
"context"
"database/sql"
"database/sql/driver"
"errors"
"time"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"gorm.io/driver/postgres"
"gorm.io/gorm"
)
// hungConnector opens connections that accept session settings but never
// answer a query, like a database host that stopped responding mid-session.
type hungConnector struct{}
func (hungConnector) Connect(context.Context) (driver.Conn, error) { return hungConn{}, nil }
func (hungConnector) Driver() driver.Driver { return hungDriver{} }
type hungDriver struct{}
func (hungDriver) Open(string) (driver.Conn, error) { return hungConn{}, nil }
type hungConn struct{}
func (hungConn) Prepare(string) (driver.Stmt, error) { return nil, errors.New("not supported") }
func (hungConn) Close() error { return nil }
func (hungConn) Begin() (driver.Tx, error) { return nil, errors.New("not supported") }
func (hungConn) ExecContext(context.Context, string, []driver.NamedValue) (driver.Result, error) {
return driver.RowsAffected(0), nil
}
func (hungConn) QueryContext(ctx context.Context, _ string, _ []driver.NamedValue) (driver.Rows, error) {
<-ctx.Done()
return nil, ctx.Err()
}
var _ = Describe("HeldLock against a hung database", func() {
It("gives up an acquire after a bounded time, even with a context that never ends", func() {
prev := heldLockCheckTimeout
heldLockCheckTimeout = 200 * time.Millisecond
DeferCleanup(func() { heldLockCheckTimeout = prev })
db, err := gorm.Open(postgres.New(postgres.Config{Conn: sql.OpenDB(hungConnector{})}), &gorm.Config{DisableAutomaticPing: true})
Expect(err).ToNot(HaveOccurred())
l := NewHeldLock(db, 4242)
done := make(chan error, 1)
go func() {
_, err := l.TryAcquire(context.Background())
done <- err
}()
var got error
Eventually(done, 5*time.Second).Should(Receive(&got))
Expect(got).To(HaveOccurred())
Expect(l.Held()).To(BeFalse())
})
})