mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-30 18:09:05 -04:00
SetDefaults injected the llama.cpp server options cache_reuse (ApplyServingDefaults) and parallel (ApplyHardwareDefaults, re-applied per selected node by the distributed router) onto every model config regardless of backend. Every other backend ignores options it does not understand, so this was harmless until longcat-video, which strictly validates its options and fails LoadModel with "unknown model option(s): cache_reuse, parallel". Gate both injections behind a new UsesLlamaCppServingOptions allow-list (llama-cpp plus the empty/auto-detect case that resolves to llama.cpp from a GGUF file, mirroring how llamaCppDefaults is registered). This follows the existing UsesLlamaSamplerDefaults precedent for llama-only defaults. The typed NBatch field is deliberately left alone: it is a proto field every backend simply ignores, which is why batch never triggered the error. Also harden the longcat-video backend to warn-and-ignore unknown model options and request params through a testable select_known_options helper, matching the other LocalAI Python backends, so a future server-injected option cannot break loading again. Assisted-by: Claude:claude-opus-4-8 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
62 lines
2.1 KiB
Go
62 lines
2.1 KiB
Go
package config
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
|
|
"github.com/mudler/xlog"
|
|
)
|
|
|
|
// Serving-policy model-config defaults.
|
|
//
|
|
// Sibling to hardware_defaults.go: those fill values driven by the target
|
|
// *device* (Blackwell batch, VRAM-scaled parallel slots); these fill values
|
|
// that improve multi-request / multi-user *serving* regardless of the GPU. They
|
|
// run together from SetDefaults and only ever fill values the user left unset.
|
|
|
|
// DefaultCacheReuse is the minimum shared-prefix chunk (in tokens) the backend
|
|
// reuses across requests via KV-cache shifting. The llama.cpp backend ships this
|
|
// disabled (n_cache_reuse = 0); we enable it so repeated prefixes (system
|
|
// prompts, RAG context, agent scaffolds, multi-turn chat) are not recomputed.
|
|
// This is the universally-useful part of "paged attention" (cross-request prefix
|
|
// sharing) and needs none of the block-KV machinery.
|
|
const DefaultCacheReuse = 256
|
|
|
|
// ApplyServingDefaults fills serving-policy ModelConfig values the user left
|
|
// unset. Currently: enable cross-request prefix caching. Explicit
|
|
// cache_reuse/n_cache_reuse in the model options always wins.
|
|
func ApplyServingDefaults(cfg *ModelConfig) {
|
|
if cfg == nil {
|
|
return
|
|
}
|
|
// cache_reuse is a llama.cpp server option; a backend that strictly
|
|
// validates its options rejects it. Only inject it on the llama.cpp path.
|
|
if !UsesLlamaCppServingOptions(cfg.Backend) {
|
|
return
|
|
}
|
|
if !backendOptionSet(cfg.Options, "cache_reuse", "n_cache_reuse") {
|
|
cfg.Options = append(cfg.Options, fmt.Sprintf("cache_reuse:%d", DefaultCacheReuse))
|
|
xlog.Debug("[serving_defaults] enabling cross-request prefix cache",
|
|
"cache_reuse", DefaultCacheReuse)
|
|
}
|
|
}
|
|
|
|
// backendOptionSet reports whether the backend options already set any of names.
|
|
// Options are "name:value" strings (or bare "name"); used so we never override
|
|
// an explicit value. Shared with hardware_defaults.go.
|
|
func backendOptionSet(opts []string, names ...string) bool {
|
|
for _, o := range opts {
|
|
name := o
|
|
if i := strings.IndexByte(o, ':'); i >= 0 {
|
|
name = o[:i]
|
|
}
|
|
name = strings.TrimSpace(strings.ToLower(name))
|
|
for _, n := range names {
|
|
if name == n {
|
|
return true
|
|
}
|
|
}
|
|
}
|
|
return false
|
|
}
|