Files
LocalAI/backend/go/localai-proxy/proxy.go
T
Ettore Di Giacinto 53ce024efc fix(localai-proxy): clear the gosec findings
Code scanning flagged seven issues in the new backend:

- G115 text.go: tool-call indexes and tokenize lengths come from the
  upstream server as int and were cast straight to int32. Add clampInt32 so
  an absurd upstream value saturates instead of wrapping.
- G115 live.go: the int16 -> uint16 cast in PCM16 encoding is a deliberate
  two's-complement reinterpretation of an already clamped sample; mark it
  with #nosec and say so.
- G304 proxy.go, media.go, client.go: api_key_file comes from the model
  config, and the media input and output paths are files core staged or
  chose for the call. None are caller-supplied. Clean the paths and add
  #nosec with that reason, as core/gallery and the sound classification
  endpoint already do.
- G306 media.go: write generated media 0o600. Core runs as the same user
  and serves the file itself.

gosec reports 0 issues for backend/go/localai-proxy and
core/services/failover. The G104 once reported for failover/prober.go is
no longer present.

Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
2026-09-27 07:42:21 +00:00

240 lines
7.9 KiB
Go

package main
import (
"context"
"errors"
"fmt"
"net/http"
"net/url"
"os"
"path/filepath"
"strings"
"sync/atomic"
"time"
"github.com/mudler/xlog"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
"github.com/mudler/LocalAI/pkg/grpc/base"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/LocalAI/pkg/httpclient"
)
const (
backendName = "localai-proxy"
// realtimePipelineOption names the upstream realtime pipeline that serves
// live transcription sessions (options: ["realtime_pipeline:<name>"]).
realtimePipelineOption = "realtime_pipeline:"
)
// LocalAIProxy serves backend methods by calling a remote LocalAI's REST API.
// base.SingleThread is not embedded: every call is an independent HTTP
// request, so serialising them would only add latency.
type LocalAIProxy struct {
base.Base
cfg atomic.Pointer[proxyConfig]
client *http.Client
}
type proxyConfig struct {
base string // upstream base URL without a trailing slash
upstreamModel string // model name sent upstream
apiKey string
realtimePipeline string
timeout time.Duration // per-request limit for non-streaming calls; 0 = none
}
func NewLocalAIProxy() *LocalAIProxy {
// httpclient.New refuses redirects: the upstream is one configured
// LocalAI, so a 3xx means misconfiguration or a hijacked host, and
// following it would replay the bearer key to an unvetted host. It also
// sets no body deadline, so long SSE streams are not cut short.
return &LocalAIProxy{client: httpclient.New()}
}
// Load refuses a model without proxy options so greedy backend probing,
// which tries every installed backend on a model file, never selects it.
func (p *LocalAIProxy) Load(opts *pb.ModelOptions) error {
po := opts.GetProxy()
if po == nil {
return errors.New("localai-proxy: Load requires proxy options (proxy.upstream_url)")
}
raw := po.GetUpstreamUrl()
if raw == "" {
return errors.New("localai-proxy: proxy.upstream_url is required")
}
u, err := url.ParseRequestURI(raw)
if err != nil {
return fmt.Errorf("localai-proxy: proxy.upstream_url %q invalid: %w", raw, err)
}
if (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" {
return fmt.Errorf("localai-proxy: proxy.upstream_url %q must be an http(s) URL with a host", raw)
}
// Every request path starts with /v1, so upstream_url is the server's
// root. A URL copied from an OpenAI-style config ends in /v1 (or a full
// endpoint): cut the path at /v1 as the failover prober does, or the
// prober reports the target healthy while every request 404s.
base := strings.TrimRight(raw, "/")
if i := strings.Index(u.Path, "/v1"); i >= 0 {
base = strings.TrimRight(u.Scheme+"://"+u.Host+u.Path[:i], "/")
xlog.Warn("localai-proxy: proxy.upstream_url should be the server root; ignoring its /v1 path",
"upstream_url", raw, "using", base)
}
// There is no translate mode: the upstream always speaks LocalAI's API.
if po.GetMode() != "" || po.GetProvider() != "" {
xlog.Warn("localai-proxy: proxy.mode and proxy.provider are ignored",
"mode", po.GetMode(), "provider", po.GetProvider())
}
key, err := resolveAPIKey(po.GetApiKeyEnv(), po.GetApiKeyFile())
if err != nil {
return err
}
model := po.GetUpstreamModel()
if model == "" {
model = opts.GetModel()
}
if model == "" {
xlog.Warn("localai-proxy: no upstream model name; set proxy.upstream_model")
}
var pipeline string
for _, o := range opts.GetOptions() {
if v, ok := strings.CutPrefix(o, realtimePipelineOption); ok {
pipeline = strings.TrimSpace(v)
}
}
var timeout time.Duration
if s := po.GetRequestTimeoutSeconds(); s > 0 {
timeout = time.Duration(s) * time.Second
}
p.cfg.Store(&proxyConfig{
base: base,
upstreamModel: model,
apiKey: key,
realtimePipeline: pipeline,
timeout: timeout,
})
xlog.Info("localai-proxy: ready", "upstream", base, "upstream_model", model,
"has_key", key != "", "realtime_pipeline", pipeline)
return nil
}
// config returns the loaded configuration, or the typed not-loaded error so
// callers see FailedPrecondition instead of a nil dereference.
func (p *LocalAIProxy) config() (*proxyConfig, error) {
cfg := p.cfg.Load()
if cfg == nil {
return nil, grpcerrors.ModelNotLoaded(backendName)
}
return cfg, nil
}
// model returns the model name to send upstream. The configured name wins so
// every method targets the same upstream model; req (a model named by the
// request itself) is only a fallback for configs that resolved no name.
func (p *LocalAIProxy) model(req string) string {
if cfg := p.cfg.Load(); cfg != nil && cfg.upstreamModel != "" {
return cfg.upstreamModel
}
return req
}
// resolveAPIKey mirrors config.ProxyConfig.ResolveAPIKey (and cloud-proxy's
// copy). Duplicated so the backend binary does not depend on core's layout.
func resolveAPIKey(envName, filePath string) (string, error) {
if envName != "" {
v := os.Getenv(envName)
if v == "" {
return "", fmt.Errorf("localai-proxy: api_key_env %q is unset", envName)
}
return v, nil
}
if filePath != "" {
// #nosec G304 -- api_key_file comes from the operator's model config (passed by core as a backend option), not from a request
b, err := os.ReadFile(filepath.Clean(filePath))
if err != nil {
return "", fmt.Errorf("localai-proxy: read api_key_file %q: %w", filePath, err)
}
return strings.TrimSpace(string(b)), nil
}
return "", nil
}
// unimplemented is the error for methods LocalAI's REST API cannot serve.
// Failover reads gRPC Unimplemented as a capability gap and moves to the next
// target without marking this one unhealthy.
func unimplemented(method string) error {
return status.Errorf(codes.Unimplemented, "localai-proxy: %s has no upstream counterpart", method)
}
func (p *LocalAIProxy) AudioEncode(*pb.AudioEncodeRequest) (*pb.AudioEncodeResult, error) {
return nil, unimplemented("AudioEncode")
}
func (p *LocalAIProxy) AudioDecode(*pb.AudioDecodeRequest) (*pb.AudioDecodeResult, error) {
return nil, unimplemented("AudioDecode")
}
// AudioToAudioStream closes out because the gRPC server drains it until
// closed; leaving it open would hang the call.
func (p *LocalAIProxy) AudioToAudioStream(_ <-chan *pb.AudioToAudioRequest, out chan<- *pb.AudioToAudioResponse) error {
close(out)
return unimplemented("AudioToAudioStream")
}
func (p *LocalAIProxy) TokenClassify(context.Context, *pb.TokenClassifyRequest) (*pb.TokenClassifyResponse, error) {
return nil, unimplemented("TokenClassify")
}
func (p *LocalAIProxy) ModelMetadata(*pb.ModelOptions) (*pb.ModelMetadataResponse, error) {
return nil, unimplemented("ModelMetadata")
}
func (p *LocalAIProxy) StartFineTune(*pb.FineTuneRequest) (*pb.FineTuneJobResult, error) {
return nil, unimplemented("StartFineTune")
}
// FineTuneProgress closes the channel: the gRPC server waits for it to close
// before returning, and base.Base leaves it open.
func (p *LocalAIProxy) FineTuneProgress(_ *pb.FineTuneProgressRequest, updates chan *pb.FineTuneProgressUpdate) error {
close(updates)
return unimplemented("FineTuneProgress")
}
func (p *LocalAIProxy) StopFineTune(*pb.FineTuneStopRequest) error {
return unimplemented("StopFineTune")
}
func (p *LocalAIProxy) ListCheckpoints(*pb.ListCheckpointsRequest) (*pb.ListCheckpointsResponse, error) {
return nil, unimplemented("ListCheckpoints")
}
func (p *LocalAIProxy) ExportModel(*pb.ExportModelRequest) error {
return unimplemented("ExportModel")
}
func (p *LocalAIProxy) StartQuantization(*pb.QuantizationRequest) (*pb.QuantizationJobResult, error) {
return nil, unimplemented("StartQuantization")
}
// QuantizationProgress closes the channel for the same reason as
// FineTuneProgress.
func (p *LocalAIProxy) QuantizationProgress(_ *pb.QuantizationProgressRequest, updates chan *pb.QuantizationProgressUpdate) error {
close(updates)
return unimplemented("QuantizationProgress")
}
func (p *LocalAIProxy) StopQuantization(*pb.QuantizationStopRequest) error {
return unimplemented("StopQuantization")
}