mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 17:15:02 -04:00
fix(models): fallback to application config default context size in /v1/models/capabilities (#12202) (#12216)
* fix(models): fallback to application config default context size (#12202) Honor appConfig.ContextSize in /v1/models/capabilities when model context_size is unset. * docs(models): explain context size fallback Describe the application default used by capability discovery and preserve the distinction between total context and per-request limits. Assisted-by: Codex:GPT-6 * fix(models): apply the default context size only when context_size is unset The request path applies the application default context size only when a model leaves context_size unset. An explicit 0 or -1 falls through to the backend fallback. The capabilities endpoint now does the same, so it reports the value the backend uses. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
3 files changed
+38
No files matched your search
@@ -36,6 +36,12 @@ func ListModelCapabilitiesEndpoint(bcl *config.ModelConfigLoader, ml *model.Mode
|
||||
for _, m := range modelNames {
|
||||
entry := schema.ModelCapabilities{ID: m, Object: "model"}
|
||||
if cfg, ok := modelConfigFor(bcl, m); ok {
|
||||
// Mirror the request path: SetDefaults applies the application
|
||||
// default only when the model leaves context_size unset. An
|
||||
// explicit 0 or -1 falls through to the backend fallback there.
|
||||
if cfg.ContextSize == nil && appConfig != nil && appConfig.ContextSize > 0 {
|
||||
cfg.ContextSize = &appConfig.ContextSize
|
||||
}
|
||||
entry.Capabilities = cfg.Capabilities()
|
||||
entry.ThreeDOperations = cfg.ThreeDOperations()
|
||||
entry.InputModalities = cfg.InputModalities()
|
||||
|
||||
@@ -160,6 +160,33 @@ parameters:
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
|
||||
})
|
||||
|
||||
It("uses application config context size when model context_size is unset", func() {
|
||||
writeConfig("llm-app-default", `
|
||||
name: llm-app-default
|
||||
backend: llama-cpp
|
||||
parameters:
|
||||
model: model.gguf
|
||||
`)
|
||||
appConf.ContextSize = 8192
|
||||
entry := entryFor(call(), "llm-app-default")
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(8192))
|
||||
})
|
||||
|
||||
It("keeps the backend fallback when the model sets a non-positive context_size", func() {
|
||||
writeConfig("llm-explicit-zero", `
|
||||
name: llm-explicit-zero
|
||||
backend: llama-cpp
|
||||
context_size: 0
|
||||
parameters:
|
||||
model: model.gguf
|
||||
`)
|
||||
appConf.ContextSize = 8192
|
||||
entry := entryFor(call(), "llm-explicit-zero")
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
|
||||
})
|
||||
It("reports an alias with its target's capabilities and context_size", func() {
|
||||
writeConfig("real-llm", `
|
||||
name: real-llm
|
||||
|
||||
@@ -141,6 +141,11 @@ curl http://localhost:8080/api/instructions/config-management?format=json
|
||||
|
||||
An additive, LocalAI-specific superset of `/v1/models`. It returns the same set of models but enriches each entry with the **capabilities** the model supports and the **input/output modalities** it accepts and produces. Use it to decide, before sending a request, whether a given model can take an image, audio, or video attachment directly - or whether the input needs converting/transcribing first.
|
||||
|
||||
The reported `context_size` uses a positive model-level value first.
|
||||
If the model does not set `context_size`, it uses **Settings → Performance → Default Context Size** when positive.
|
||||
Otherwise, it uses the backend fallback of 4096 tokens.
|
||||
For llama.cpp with separate KV caches, the reported value accounts for the number of parallel slots.
|
||||
|
||||
Because it is purely additive, clients that only understand `/v1/models` keep working unchanged; they simply never call this route.
|
||||
|
||||
```bash
|
||||
|
||||
Reference in new issue
Block a user