mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-30 18:09:05 -04:00
#10970 gave the four PredictOptions RPCs a model-identity check so a backend reached through a stale distributed route rejects the request instead of answering from whatever model it holds (#10952). Every other modality shares that exposure: the route is cached by host:port, a worker can recycle a stopped backend's port for another model's backend, and a liveness-only probe cannot tell a stale row from a valid one. Extends the same mechanism to the 21 remaining request messages that reach a backend through the router, using the pattern #10970 established rather than a parallel one: - proto: ModelIdentity on each modality request message. - controller: populated from ModelConfig.Model at the call site that also builds ModelOptions, so load-time and request-time values are equal by construction. - backends: one generic guard in pkg/grpc/server.go (27 Go backends), the method set in backend/python/common (36 Python backends), llama-cpp (AudioTranscription/Stream, Rerank, Score) and privacy-filter (TokenClassify). - reconcile already drops the stale row on IsModelMismatch; no change. TTSRequest and SoundGenerationRequest get a SEPARATE ModelIdentity field rather than reusing their existing `model`: FileStagingClient rewrites `model` to a worker-local path, so comparing it would reject valid requests in exactly the configuration this guards. AudioEncode/AudioDecode are deliberately left unguarded: the opus codec backend is loaded from a literal rather than a ModelConfig, so no value carries the equality guarantee the comparison depends on. The four bidirectional stream RPCs are out of scope; they bypass reconcile. Empty means skip on both sides, so an old controller, an old backend, and the bare request structs in tests/e2e-backends all keep working. Assisted-by: Claude Code:claude-opus-4-8 [Read] [Edit] [Bash] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
111 lines
3.5 KiB
Go
111 lines
3.5 KiB
Go
package backend
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"time"
|
|
|
|
"github.com/mudler/LocalAI/core/config"
|
|
"github.com/mudler/LocalAI/core/trace"
|
|
"github.com/mudler/LocalAI/pkg/grpc/proto"
|
|
model "github.com/mudler/LocalAI/pkg/model"
|
|
)
|
|
|
|
// RerankResult is the per-document score returned to consumers,
|
|
// narrowed from proto.RerankResult so callers don't need to depend on
|
|
// the proto package.
|
|
type RerankResult struct {
|
|
Index int
|
|
RelevanceScore float32
|
|
}
|
|
|
|
// Reranker scores a list of candidate documents against a query.
|
|
// Returns one RerankResult per input document (no top-N truncation -
|
|
// callers that need it can sort and slice).
|
|
type Reranker interface {
|
|
Rerank(ctx context.Context, query string, documents []string) ([]RerankResult, error)
|
|
}
|
|
|
|
// NewReranker binds (loader, modelConfig, appConfig) into a Reranker.
|
|
func NewReranker(loader *model.ModelLoader, modelConfig config.ModelConfig, appConfig *config.ApplicationConfig) Reranker {
|
|
return &modelReranker{loader: loader, modelConfig: modelConfig, appConfig: appConfig}
|
|
}
|
|
|
|
type modelReranker struct {
|
|
loader *model.ModelLoader
|
|
modelConfig config.ModelConfig
|
|
appConfig *config.ApplicationConfig
|
|
}
|
|
|
|
func (r *modelReranker) Rerank(ctx context.Context, query string, documents []string) ([]RerankResult, error) {
|
|
req := &proto.RerankRequest{
|
|
Query: query,
|
|
Documents: documents,
|
|
// TopN=0: backend returns scores for every document. Truncating
|
|
// here would silently zero out labels the reranker considered
|
|
// unlikely, which the router classifier needs.
|
|
}
|
|
res, err := Rerank(ctx, req, r.loader, r.appConfig, r.modelConfig)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := make([]RerankResult, 0, len(res.GetResults()))
|
|
for _, dr := range res.GetResults() {
|
|
out = append(out, RerankResult{Index: int(dr.GetIndex()), RelevanceScore: dr.GetRelevanceScore()})
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
func Rerank(ctx context.Context, request *proto.RerankRequest, loader *model.ModelLoader, appConfig *config.ApplicationConfig, modelConfig config.ModelConfig) (*proto.RerankResult, error) {
|
|
// model.WithContext(ctx) overrides the app-context default set in
|
|
// ModelOptions so distributed routing decisions reach the request's
|
|
// X-LocalAI-Node holder via distributedhdr.Stamp.
|
|
opts := ModelOptions(modelConfig, appConfig, model.WithContext(ctx))
|
|
rerankModel, err := loader.Load(opts...)
|
|
if err != nil {
|
|
recordModelLoadFailure(appConfig, modelConfig.Name, modelConfig.Backend, err, nil)
|
|
return nil, err
|
|
}
|
|
|
|
if rerankModel == nil {
|
|
return nil, fmt.Errorf("could not load rerank model")
|
|
}
|
|
|
|
var startTime time.Time
|
|
if appConfig.EnableTracing {
|
|
trace.InitBackendTracingIfEnabled(appConfig.TracingMaxItems, appConfig.TracingMaxBodyBytes)
|
|
startTime = time.Now()
|
|
}
|
|
|
|
// Stamped here, not at the HTTP handler: this is the function that also
|
|
// builds ModelOptions from the same config, so the two values are equal by
|
|
// construction (#10952).
|
|
request.ModelIdentity = modelConfig.Model
|
|
|
|
res, err := rerankModel.Rerank(ctx, request)
|
|
|
|
if appConfig.EnableTracing {
|
|
errStr := ""
|
|
if err != nil {
|
|
errStr = err.Error()
|
|
}
|
|
|
|
trace.RecordBackendTrace(trace.BackendTrace{
|
|
Timestamp: startTime,
|
|
Duration: time.Since(startTime),
|
|
Type: trace.BackendTraceRerank,
|
|
ModelName: modelConfig.Name,
|
|
Backend: modelConfig.Backend,
|
|
Summary: trace.TruncateString(request.Query, 200),
|
|
Error: errStr,
|
|
Data: map[string]any{
|
|
"query": request.Query,
|
|
"documents_count": len(request.Documents),
|
|
"top_n": request.TopN,
|
|
},
|
|
})
|
|
}
|
|
|
|
return res, err
|
|
}
|