diff --git a/core/http/endpoints/openai/list_capabilities.go b/core/http/endpoints/openai/list_capabilities.go index 72a645ef6..27583385a 100644 --- a/core/http/endpoints/openai/list_capabilities.go +++ b/core/http/endpoints/openai/list_capabilities.go @@ -36,6 +36,12 @@ func ListModelCapabilitiesEndpoint(bcl *config.ModelConfigLoader, ml *model.Mode for _, m := range modelNames { entry := schema.ModelCapabilities{ID: m, Object: "model"} if cfg, ok := modelConfigFor(bcl, m); ok { + // Mirror the request path: SetDefaults applies the application + // default only when the model leaves context_size unset. An + // explicit 0 or -1 falls through to the backend fallback there. + if cfg.ContextSize == nil && appConfig != nil && appConfig.ContextSize > 0 { + cfg.ContextSize = &appConfig.ContextSize + } entry.Capabilities = cfg.Capabilities() entry.ThreeDOperations = cfg.ThreeDOperations() entry.InputModalities = cfg.InputModalities() diff --git a/core/http/endpoints/openai/list_capabilities_test.go b/core/http/endpoints/openai/list_capabilities_test.go index 0c8a5d4ff..ecfc85f81 100644 --- a/core/http/endpoints/openai/list_capabilities_test.go +++ b/core/http/endpoints/openai/list_capabilities_test.go @@ -160,6 +160,33 @@ parameters: Expect(entry).NotTo(BeNil()) Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize)) }) + + It("uses application config context size when model context_size is unset", func() { + writeConfig("llm-app-default", ` +name: llm-app-default +backend: llama-cpp +parameters: + model: model.gguf +`) + appConf.ContextSize = 8192 + entry := entryFor(call(), "llm-app-default") + Expect(entry).NotTo(BeNil()) + Expect(entry.ContextSize).To(Equal(8192)) + }) + + It("keeps the backend fallback when the model sets a non-positive context_size", func() { + writeConfig("llm-explicit-zero", ` +name: llm-explicit-zero +backend: llama-cpp +context_size: 0 +parameters: + model: model.gguf +`) + appConf.ContextSize = 8192 + entry := entryFor(call(), "llm-explicit-zero") + Expect(entry).NotTo(BeNil()) + Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize)) + }) It("reports an alias with its target's capabilities and context_size", func() { writeConfig("real-llm", ` name: real-llm diff --git a/docs/content/features/api-discovery.md b/docs/content/features/api-discovery.md index f45524423..932634a39 100644 --- a/docs/content/features/api-discovery.md +++ b/docs/content/features/api-discovery.md @@ -141,6 +141,11 @@ curl http://localhost:8080/api/instructions/config-management?format=json An additive, LocalAI-specific superset of `/v1/models`. It returns the same set of models but enriches each entry with the **capabilities** the model supports and the **input/output modalities** it accepts and produces. Use it to decide, before sending a request, whether a given model can take an image, audio, or video attachment directly - or whether the input needs converting/transcribing first. +The reported `context_size` uses a positive model-level value first. +If the model does not set `context_size`, it uses **Settings → Performance → Default Context Size** when positive. +Otherwise, it uses the backend fallback of 4096 tokens. +For llama.cpp with separate KV caches, the reported value accounts for the number of parallel slots. + Because it is purely additive, clients that only understand `/v1/models` keep working unchanged; they simply never call this route. ```bash