From 4eff44a03e12fb1c4a809e95a545bdba7fca5138 Mon Sep 17 00:00:00 2001 From: yzxcj797 <1784931579@qq.com> Date: Sun, 16 Aug 2026 11:07:11 +0800 Subject: [PATCH 01/49] docs: replace dead chatbot-ui example link with repo root --- docs/content/advanced/advanced-usage.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md index f7d9546cf..7850151ae 100644 --- a/docs/content/advanced/advanced-usage.md +++ b/docs/content/advanced/advanced-usage.md @@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master ``` -See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files. +See also [chatbot-ui](https://github.com/mudler/LocalAI-examples) as an example on how to use config files. ### Prompt templates From 6393efc12b87a00e9273dd76dd7404fb58f0ce0e Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:57:53 +0000 Subject: [PATCH 02/49] chore(model-gallery): propose variant groupings Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index 04fa669b6..ee86c092a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -44011,6 +44011,12 @@ sha256: "" uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors - name: kimodo-soma-rp + variants: + - model: kimodo-soma-rp-bf16 + - model: kimodo-soma-rp-q4_k + - model: kimodo-soma-rp-q4_k_m + - model: kimodo-soma-rp-q5_k + - model: kimodo-soma-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44198,6 +44204,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-soma-seed + variants: + - model: kimodo-soma-seed-bf16 + - model: kimodo-soma-seed-q4_k + - model: kimodo-soma-seed-q4_k_m + - model: kimodo-soma-seed-q5_k + - model: kimodo-soma-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44385,6 +44397,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-rp + variants: + - model: kimodo-g1-rp-bf16 + - model: kimodo-g1-rp-q4_k + - model: kimodo-g1-rp-q4_k_m + - model: kimodo-g1-rp-q5_k + - model: kimodo-g1-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44572,6 +44590,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-seed + variants: + - model: kimodo-g1-seed-bf16 + - model: kimodo-g1-seed-q4_k + - model: kimodo-g1-seed-q4_k_m + - model: kimodo-g1-seed-q5_k + - model: kimodo-g1-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: From e17039fb30972e00287e7a5ad58c2fb550c7698b Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:49:43 +0000 Subject: [PATCH 03/49] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 72 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 72 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..929c337bb 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,76 @@ --- +- name: "swift-qwen3.8-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF + description: | + Website  •  + Learn more  •  + GGUF  •  + Enterprise licensing + + # Swift-Qwen3.8-27B + + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, + using **58.3% fewer thinking tokens** while maintaining near-identical performance + (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. + + The prompt is a sample from LiveCodeBench v6 + + ## Training approach + + We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s + reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. + + Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. + + For maximum gains, Swift also includes a transfer component derived from + BottleCap AI's ThinkingCap-Qwen3.6-27B. + + ## Evaluation scope + + > All results below compare the Qwen3.8-27B BF16 base with the same base plus the + > Swift adapter. + + ## Benchmarks + + ... + license: "other" + tags: + - llm + - gguf + - reasoning + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + min_p: 0 + model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + presence_penalty: 1.5 + repeat_penalty: 1 + temperature: 0.7 + top_k: 20 + top_p: 0.8 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf + - filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From 3c9047654e3d12aecd9467245902446912e56faf Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:58:03 +0000 Subject: [PATCH 04/49] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 46 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..cceda1b77 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,50 @@ --- +- name: "ternary-bonsai-2-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + description: | + # Qwen3.8-27B + + > [!Note] + > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. + > + > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. + + > [!Tip] + > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. + > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. + + Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. + + ... + license: "apache-2.0" + tags: + - llm + - gguf + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf + - filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From 6f1b3d3fa9f812103bab8ddbc37d24cfcc62f360 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:03:15 +0000 Subject: [PATCH 05/49] docs(proxy): clarify optional upstream API keys Document the existing no-auth upstream configuration and distinguish upstream credentials from LocalAI client authentication. Closes #12264 Assisted-by: Codex:gpt-6 --- docs/content/operations/cloud-proxy.md | 39 +++++++++++++++++++++++--- 1 file changed, 35 insertions(+), 4 deletions(-) diff --git a/docs/content/operations/cloud-proxy.md b/docs/content/operations/cloud-proxy.md index 02af25bd0..a258ece14 100644 --- a/docs/content/operations/cloud-proxy.md +++ b/docs/content/operations/cloud-proxy.md @@ -60,9 +60,9 @@ against - and two modes: `proxy.provider` selects the auth scheme and (in translate mode) the wire format. Supported values: `openai`, `anthropic`. -API keys are loaded from either an environment variable (`api_key_env`) or a -file (`api_key_file`). The key never appears in the config file or the admin -UI; pick whichever fits your secret-management setup. +If the upstream requires an API key, configure either an environment variable +(`api_key_env`) or a file (`api_key_file`). The key never appears in the config +file or the admin UI. If the upstream requires no API key, omit both fields. ### OpenAI passthrough @@ -126,7 +126,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak the OpenAI chat-completions wire format. Use `provider: openai` with the -provider's URL and API key: +provider's URL and, if required, its API key: ```yaml name: llama-3-70b-via-together @@ -140,6 +140,37 @@ proxy: upstream_model: meta-llama/Llama-3-70b-chat-hf ``` +### Upstreams without an API key + +For an OpenAI-compatible upstream that accepts requests without authentication, +omit both `api_key_env` and `api_key_file`: + +```yaml +name: internal-chat-proxy +backend: cloud-proxy + +proxy: + mode: passthrough + provider: openai + upstream_url: http://inference.internal:8000/v1/chat/completions + upstream_model: my-model +``` + +Replace the example URL and model name with your upstream's values. LocalAI +loads this configuration without resolving a key and adds no upstream +`Authorization` header. This also applies to OpenAI-compatible upstreams in +translate mode. + +Omitting both fields differs from setting `api_key_env` to an empty or unset +environment variable: the latter causes a backend load error. + +LocalAI's client authentication is separate. Clients must still authenticate +to LocalAI when its authentication is enabled. LocalAI does not forward their +`Authorization` header to the upstream. + +An upstream without API keys can still require another authentication or +payment protocol. Omitting these fields does not implement that protocol. + ### Translate mode In translate mode the cloud-proxy backend converts LocalAI's internal proto From e35a9679703c437405b5a3fbe3a0515d21e21932 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:05:14 +0000 Subject: [PATCH 06/49] chore(gallery): add Sharp-Spark 4B variants Add Q4, Q5, and Q6 builds with the embedded chat template. Pin artifact revisions and document installation. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 14 +++ gallery/index.yaml | 114 +++++++++++++++++++++++++ 2 files changed, 128 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..810c82d56 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,20 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Sharp-Spark-X2.5-4B + +Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp. +The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants. +To select the publisher's recommended Q6 build, run: + +```bash +local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6 +``` + +All builds use a 32,768-token default context and the embedded Sharp-Spark chat template. +That template adds a terseness instruction to the system prompt. +See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..c5ffc5b9a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5725,6 +5725,120 @@ - filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442 +- name: sharp-spark-x2.5-4b + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: sharp-spark-x2.5-4b-q5 + - model: sharp-spark-x2.5-4b-q6 + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q4_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837 + +- name: sharp-spark-x2.5-4b-q5 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q5_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26 + +- name: sharp-spark-x2.5-4b-q6 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q6_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc + - &spark-x2-5-4b name: "spark-x2.5-4b-q4" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" From dc4db4b119a2991b10283e49d6c55016bb20d2c3 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:07:50 +0000 Subject: [PATCH 07/49] chore(gallery): add Swift 1.5 GSQ-RCO variants Add four text-only llama.cpp builds with pinned download URLs and verified checksums. Document variant selection and the model license. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 15 +++ gallery/index.yaml | 172 +++++++++++++++++++++++++ 2 files changed, 187 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..6eb034ba2 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,21 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Swift 1.5 Qwen3.8-27B GSQ-RCO + +Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp. +The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model. +To select IQ3_S explicitly, run: + +```bash +local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s +``` + +The configurations use the embedded chat template and default to 32,768 context tokens. +These builds support text chat only: the publisher has no verified vision projector for this release. +They use standard GGUF files without MTP decoding. +See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..47fb9b4f7 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5533,6 +5533,178 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2 +- name: "swift-1.5-qwen3.8-27b-gsq-rco" + variants: + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf - !!merge <<: *qwen3-8-27b name: "qwen3.8-27b-gsq-rco-iq2-xs" variants: [] From 7460312d23dce57f026025343e0557be1594236b Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 04:04:34 +0000 Subject: [PATCH 08/49] chore(gallery): add ThinkingCap Qwen3.8 variants Add Q4_K_M and Q8_0 builds with the F16 vision projector and install docs. Pin artifact revisions and verify SHA256 against HF LFS metadata and HTTP headers. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 15 ++++ gallery/index.yaml | 96 ++++++++++++++++++++++++++ 2 files changed, 111 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..fca3a2c96 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -48,6 +48,21 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9 The configurations default to 32,768 context tokens and use the model's embedded chat template. See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. +## ThinkingCap Qwen3.8-27B + +Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input. +The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector. +LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly: + +```bash +local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8 +``` + +Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings. +MTP speculative decoding is not enabled by these entries. +The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE). +Review that license for permitted use. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..dc1f0a7d6 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -388,6 +388,102 @@ - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: thinkingcap-qwen3.8-27b + variants: + - model: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf +- name: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf - name: hemmingway-1 variants: - model: hemmingway-1-q8 From dcddb641f0564ce3d328748581655b6b1b4a32d0 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 08:05:46 +0000 Subject: [PATCH 09/49] chore(gallery): add Agention Qwen3.8 variants Add IQ4_XS and Q4_K_M GGUF builds with a BF16 vision projector. Pin verified artifacts and document installation and variant selection. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 18 +++++ gallery/index.yaml | 98 ++++++++++++++++++++++++++ 2 files changed, 116 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..bf83b6300 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,24 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Qwen3.8-27B Agention Precision + +The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of +Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image +input and use a 32,768-token context by default. + +Install with automatic variant selection: + +```bash +local-ai models install qwen3.8-27b-agention-iq4-xs +``` + +To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs` +or `--variant qwen3.8-27b-agention-q4-k-m` to the same command. +The files use standard llama.cpp quantization types and the Apache-2.0 license. +See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF) +for quantization details. These entries do not enable MTP speculative decoding. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..0cab8c751 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5247,6 +5247,104 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545 +- name: qwen3.8-27b-agention-iq4-xs + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: qwen3.8-27b-agention-q4-k-m + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf + sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67 + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 +- name: qwen3.8-27b-agention-q4-k-m + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf + sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 - &qwen3-8-27b name: "qwen3.8-27b-q4" variants: From 9ea9277ee69be910faee05bce7cbe9bf9cfdc67a Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:03 +0200 Subject: [PATCH 10/49] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `cdf232cc17e410e60c1bc3b85516c4a41199b662` (#12288) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index 0f2a84e7d..d6bfd7490 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc +IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662 LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From 01017dcdd631baca0886cd8ad7e6ca48d06594b8 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:25 +0200 Subject: [PATCH 11/49] chore: :arrow_up: Update CrispStrobe/CrispASR to `013ae1624dc40ecf059065d577180722439f804e` (#12292) :arrow_up: Update CrispStrobe/CrispASR Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/crispasr/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index 65a4a784d..f2155ffb9 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee +CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From fc6df9efc33626b6a4354fba32b00c9d6b270cca Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:52 +0200 Subject: [PATCH 12/49] chore: :arrow_up: Update mudler/parakeet.cpp to `2bf88954dc628b32835734e2e9159550a75a1dc6` (#12291) :arrow_up: Update mudler/parakeet.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/parakeet-cpp/Makefile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index 8fc14bcb8..e288f6fcc 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go From 4524765b9f6a83234415ba1f8b2177f4e9b3ac84 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:25:09 +0200 Subject: [PATCH 13/49] chore: :arrow_up: Update 0xShug0/audio.cpp to `94bd4656399180befc141b17bd6696bf84df0a9f` (#12289) :arrow_up: Update 0xShug0/audio.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/audio-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index 85da233bd..e8d27eb52 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22 +AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) From c9e822215a2cda5362fdd8f679571dc28ccfc32c Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:25:27 +0200 Subject: [PATCH 14/49] chore(model-gallery): :arrow_up: update checksum (#12290) :arrow_up: Checksum updates in gallery/index.yaml Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- gallery/index.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..5b1df136a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -205,7 +205,7 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7 + sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf - name: "qwopus3.8-27b-flash" variants: - model: qwopus3.8-27b-flash-q8 From 6043e5e0cb61db03e06be7cb59f299348cc17fd9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:04:40 +0000 Subject: [PATCH 15/49] chore(gallery): add Qwopus Flash V2 variants Add Q4_K_M and Q8_0 builds with vision and MTP decoding. Pin the weights and projector to a verified Hugging Face revision. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 14 ++++ gallery/index.yaml | 94 ++++++++++++++++++++++++++ 2 files changed, 108 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..4839b88ab 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -48,6 +48,20 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9 The configurations default to 32,768 context tokens and use the model's embedded chat template. See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. +## Qwopus3.8 Flash V2 + +Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection: + +```bash +local-ai models install qwopus3.8-27b-flash-v2 +local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8 +``` + +Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector. +Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256. +This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks. +See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index 5b1df136a..d5e56d590 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -206,6 +206,100 @@ - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf +- name: "qwopus3.8-27b-flash-v2" + variants: + - model: qwopus3.8-27b-flash-v2-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q4_K_M GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 +- name: "qwopus3.8-27b-flash-v2-q8" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q8_0 GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78 + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 - name: "qwopus3.8-27b-flash" variants: - model: qwopus3.8-27b-flash-q8 From 065f9691fa27fdfbd4362f582c426412234d0b1d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:05:55 +0000 Subject: [PATCH 16/49] chore(gallery): add Cyber-Tiel-Coder variants Add Q4 and Q8 MTP builds with a shared vision projector and installation docs. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 8 ++ gallery/index.yaml | 104 +++++++++++++++++++++++++ 2 files changed, 112 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..e0a5c60bc 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,14 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Cyber-Tiel-Coder + +Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp. +The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector. +To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`. +Both configurations use the embedded chat template and default to 32,768 context tokens. +The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 5b1df136a..18c2dd662 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -4767,6 +4767,110 @@ - filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41 +- name: cyber-tiel-coder-35b-a3b-q4-mtp + variants: + - model: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 +- name: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961 + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 - &tiel-coder-35b-a3b name: "tiel-coder-35b-a3b-q4" variants: From 7690b06789c3b4a39f48a53f2ac084bb29268138 Mon Sep 17 00:00:00 2001 From: Stefan Walcz Date: Sun, 27 Sep 2026 20:58:01 +0200 Subject: [PATCH 17/49] chore(deps): bump LocalAGI to 8253de9 (re-dial dropped MCP sessions) (#12299) Picks up mudler/LocalAGI 3ce0a08 "fix(mcp): re-dial an MCP session the server has dropped". Agents open their MCP sessions once, when they are created; when the MCP server restarts it forgets them and the go-sdk client does not reconnect by itself, so the agent kept a dead session - or, behind a server that revives unknown session IDs, a stale tool list - until LocalAI restarted. Only go.mod/go.sum change; core/services/agentpool builds and vets against the new version. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz --- go.mod | 4 ++-- go.sum | 2 ++ 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/go.mod b/go.mod index 1d4025b89..c81fbd82e 100644 --- a/go.mod +++ b/go.mod @@ -259,7 +259,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 + github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect @@ -535,7 +535,7 @@ require ( golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect golang.org/x/mod v0.36.0 // indirect golang.org/x/sync v0.20.0 - golang.org/x/sys v0.45.0 // indirect + golang.org/x/sys v0.45.0 golang.org/x/term v0.43.0 golang.org/x/text v0.37.0 golang.org/x/tools v0.45.0 // indirect diff --git a/go.sum b/go.sum index 891a60d05..863118db1 100644 --- a/go.sum +++ b/go.sum @@ -1032,6 +1032,8 @@ github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxF github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= From a592e23778666b95c8ebfd5cefcd45decd4f6f0e Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:40 +0200 Subject: [PATCH 18/49] chore: :arrow_up: Update ggml-org/llama.cpp to `95887577ab5fead779581a7030a83c7752ff3234` (#12272) :arrow_up: Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index b672f2d31..8bf0ff5c0 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4 +LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= From 08827cfd5e60a97f107681a638bde9f0cd6e75d9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:45 +0200 Subject: [PATCH 19/49] chore: :arrow_up: Update mudler/vllm.cpp to `c3bebc357385990f721af66a3a6c69328dd4fc6c` (#12252) :arrow_up: Update mudler/vllm.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/vllm-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index d19d103b1..56a32dbaf 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb +VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun From 4bc5f292fe12f6bff847defe5f6f63b86d6b2c66 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:49 +0200 Subject: [PATCH 20/49] chore: :arrow_up: Update TheTom/llama-cpp-turboquant to `a3d5603d110bda29222d2011596cdc84d7fa532d` (#12232) * :arrow_up: Update TheTom/llama-cpp-turboquant Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(turboquant): drop the upstreamed D512 patch Upstream c0e227c guards D512 declarations, dispatch, and instances with GGML_USE_HIP. This prevents the CUDA shared-memory overflow that our patch addressed. The old patch now rejects the guarded source. Remove the obsolete patch for the pinned a3d5603d revision. The remaining patch series applies successfully, and the build-target test passes. Assisted-by: Codex:gpt-6 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/cpp/turboquant/Makefile | 2 +- ...emove-d512-turbo-shared-mem-overflow.patch | 52 ------------------- 2 files changed, 1 insertion(+), 53 deletions(-) delete mode 100644 backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch diff --git a/backend/cpp/turboquant/Makefile b/backend/cpp/turboquant/Makefile index e3482d8db..7a8024ebf 100644 --- a/backend/cpp/turboquant/Makefile +++ b/backend/cpp/turboquant/Makefile @@ -1,7 +1,7 @@ # Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant. # Auto-bumped nightly by .github/workflows/bump_deps.yaml. -TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64 +TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant CMAKE_ARGS?= diff --git a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch b/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch deleted file mode 100644 index 7dfb385c3..000000000 --- a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch +++ /dev/null @@ -1,52 +0,0 @@ -diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh -index 680fd12..ffd6604 100644 ---- a/ggml/src/ggml-cuda/fattn-vec.cuh -+++ b/ggml/src/ggml-cuda/fattn-vec.cuh -@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); -diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu -index 5c614a9..d765cfc 100644 ---- a/ggml/src/ggml-cuda/fattn.cu -+++ b/ggml/src/ggml-cuda/fattn.cu -@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0) - - #ifdef GGML_CUDA_FA_ALL_QUANTS - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16) -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -index a93be56..3630d87 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -index 3c806c2..c8a4d9f 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -index 180902f..1646ef0 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); From 13a01e657a1a389bad564c57c6fb3e82dee55835 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:05:55 +0200 Subject: [PATCH 21/49] chore(deps): bump sentence-transformers from 5.7.0 to 6.1.0 in /backend/python/transformers (#12245) chore(deps): bump sentence-transformers in /backend/python/transformers Bumps [sentence-transformers](https://github.com/huggingface/sentence-transformers) from 5.7.0 to 6.1.0. - [Release notes](https://github.com/huggingface/sentence-transformers/releases) - [Commits](https://github.com/huggingface/sentence-transformers/compare/v5.7.0...v6.1.0) --- updated-dependencies: - dependency-name: sentence-transformers dependency-version: 6.1.0 dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements-cpu.txt | 2 +- backend/python/transformers/requirements-cublas12.txt | 2 +- backend/python/transformers/requirements-cublas13.txt | 2 +- backend/python/transformers/requirements-hipblas.txt | 2 +- backend/python/transformers/requirements-intel.txt | 2 +- backend/python/transformers/requirements-mps.txt | 2 +- 6 files changed, 6 insertions(+), 6 deletions(-) diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt index 3e3206912..6cd324d8d 100644 --- a/backend/python/transformers/requirements-cpu.txt +++ b/backend/python/transformers/requirements-cpu.txt @@ -4,7 +4,7 @@ numba==0.67.0 accelerate transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt index 40bf331d4..632330714 100644 --- a/backend/python/transformers/requirements-cublas12.txt +++ b/backend/python/transformers/requirements-cublas12.txt @@ -4,7 +4,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt index f394f98b1..e91307ce8 100644 --- a/backend/python/transformers/requirements-cublas13.txt +++ b/backend/python/transformers/requirements-cublas13.txt @@ -4,7 +4,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt index e4b1bba11..c97e4d27b 100644 --- a/backend/python/transformers/requirements-hipblas.txt +++ b/backend/python/transformers/requirements-hipblas.txt @@ -5,7 +5,7 @@ transformers>=5.15.1 llvmlite==0.49.0 numba==0.67.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt index 54ee6ce67..0ae7eb9fe 100644 --- a/backend/python/transformers/requirements-intel.txt +++ b/backend/python/transformers/requirements-intel.txt @@ -5,7 +5,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt index ea8ba5ab0..ee8431b28 100644 --- a/backend/python/transformers/requirements-mps.txt +++ b/backend/python/transformers/requirements-mps.txt @@ -4,7 +4,7 @@ numba==0.67.0 accelerate transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 From 1cacecc460b6e85e3c5f172addf4d7158a9e1332 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:05:59 +0200 Subject: [PATCH 22/49] chore(deps): update numpy requirement from >=2.5.2 to >=2.5.3 in /backend/python/transformers (#12244) chore(deps): update numpy requirement in /backend/python/transformers Updates the requirements on [numpy](https://github.com/numpy/numpy) to permit the latest version. - [Release notes](https://github.com/numpy/numpy/releases) - [Changelog](https://github.com/numpy/numpy/blob/main/doc/RELEASE_WALKTHROUGH.rst) - [Commits](https://github.com/numpy/numpy/compare/v2.5.2...v2.5.3) --- updated-dependencies: - dependency-name: numpy dependency-version: 2.5.3 dependency-type: direct:production ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt index d85aca02a..63043dc5e 100644 --- a/backend/python/transformers/requirements.txt +++ b/backend/python/transformers/requirements.txt @@ -3,4 +3,4 @@ protobuf==7.36.1 certifi setuptools scipy==1.18.0 -numpy>=2.5.2 \ No newline at end of file +numpy>=2.5.3 \ No newline at end of file From de203c2fb5ef00594a869d935d8b583b34d46888 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:06:03 +0200 Subject: [PATCH 23/49] chore(deps): update transformers requirement from >=5.15.1 to >=5.17.0 in /backend/python/transformers (#12105) chore(deps): update transformers requirement Updates the requirements on [transformers](https://github.com/huggingface/transformers) to permit the latest version. - [Release notes](https://github.com/huggingface/transformers/releases) - [Commits](https://github.com/huggingface/transformers/compare/v5.15.1...v5.17.0) --- updated-dependencies: - dependency-name: transformers dependency-version: 5.17.0 dependency-type: direct:production ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements-cpu.txt | 2 +- backend/python/transformers/requirements-cublas12.txt | 2 +- backend/python/transformers/requirements-cublas13.txt | 2 +- backend/python/transformers/requirements-hipblas.txt | 2 +- backend/python/transformers/requirements-intel.txt | 2 +- backend/python/transformers/requirements-mps.txt | 2 +- 6 files changed, 6 insertions(+), 6 deletions(-) diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt index 6cd324d8d..8c5a025e2 100644 --- a/backend/python/transformers/requirements-cpu.txt +++ b/backend/python/transformers/requirements-cpu.txt @@ -2,7 +2,7 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt index 632330714..388a2334c 100644 --- a/backend/python/transformers/requirements-cublas12.txt +++ b/backend/python/transformers/requirements-cublas12.txt @@ -2,7 +2,7 @@ torch==2.7.1 accelerate llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt index e91307ce8..aa49676b1 100644 --- a/backend/python/transformers/requirements-cublas13.txt +++ b/backend/python/transformers/requirements-cublas13.txt @@ -2,7 +2,7 @@ torch==2.9.0 llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt index c97e4d27b..84b62042b 100644 --- a/backend/python/transformers/requirements-hipblas.txt +++ b/backend/python/transformers/requirements-hipblas.txt @@ -1,7 +1,7 @@ --extra-index-url https://download.pytorch.org/whl/rocm7.0 torch==2.10.0+rocm7.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 llvmlite==0.49.0 numba==0.67.0 bitsandbytes diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt index 0ae7eb9fe..b0b565bf1 100644 --- a/backend/python/transformers/requirements-intel.txt +++ b/backend/python/transformers/requirements-intel.txt @@ -3,7 +3,7 @@ torch optimum[openvino] llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt index ee8431b28..9659c943f 100644 --- a/backend/python/transformers/requirements-mps.txt +++ b/backend/python/transformers/requirements-mps.txt @@ -2,7 +2,7 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers From e6b2309b3fe5546833d094483e84437d75400acb Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:06:07 +0200 Subject: [PATCH 24/49] chore(deps): bump grpcio from 1.83.0 to 1.84.0 in /backend/python/transformers (#12102) chore(deps): bump grpcio in /backend/python/transformers Bumps [grpcio](https://github.com/grpc/grpc) from 1.83.0 to 1.84.0. - [Release notes](https://github.com/grpc/grpc/releases) - [Commits](https://github.com/grpc/grpc/compare/v1.83.0...v1.84.0) --- updated-dependencies: - dependency-name: grpcio dependency-version: 1.84.0 dependency-type: direct:production update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt index 63043dc5e..af5027f10 100644 --- a/backend/python/transformers/requirements.txt +++ b/backend/python/transformers/requirements.txt @@ -1,4 +1,4 @@ -grpcio==1.83.0 +grpcio==1.84.0 protobuf==7.36.1 certifi setuptools From 7340970ae798a35e1e877e18f06c8a4791160d0e Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:21 +0000 Subject: [PATCH 25/49] chore(gallery): tag swift-qwen3.8-27b as mtp and vision, fix license The entry enables spec_type:draft-mtp, so variant ranking needs the mtp tag. Replace the scraped model-card description, set the Swift Open License v1.0 and link the base model repo. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 42 ++++++++++-------------------------------- 1 file changed, 10 insertions(+), 32 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 929c337bb..eb9e5e15c 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -2,44 +2,21 @@ - name: "swift-qwen3.8-27b" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27b - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF description: | - Website  •  - Learn more  •  - GGUF  •  - Enterprise licensing - - # Swift-Qwen3.8-27B - - Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, - using **58.3% fewer thinking tokens** while maintaining near-identical performance - (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. - - The prompt is a sample from LiveCodeBench v6 - - ## Training approach - - We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s - reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. - - Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. - - For maximum gains, Swift also includes a transfer component derived from - BottleCap AI's ThinkingCap-Qwen3.6-27B. - - ## Evaluation scope - - > All results below compare the Qwen3.8-27B BF16 base with the same base plus the - > Swift adapter. - - ## Benchmarks - - ... - license: "other" + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B. + The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss. + This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding. + The weights use the Swift Open License v1.0. + license: "swift-open-license-1.0" tags: - llm - gguf - reasoning + - vision + - multimodal + - mtp overrides: backend: llama-cpp function: @@ -48,6 +25,7 @@ disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf options: - use_jinja:true From 40d37330bcbdeaac3c124abe7307b427a603646f Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:33 +0000 Subject: [PATCH 26/49] docs: point the config example link at the configurations directory The link text still said chatbot-ui, but it now pointed at the examples repository root. Link the configurations directory, which holds the example model config files, and describe it as such. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- docs/content/advanced/advanced-usage.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md index 7850151ae..8580aa499 100644 --- a/docs/content/advanced/advanced-usage.md +++ b/docs/content/advanced/advanced-usage.md @@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master ``` -See also [chatbot-ui](https://github.com/mudler/LocalAI-examples) as an example on how to use config files. +See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files. ### Prompt templates From a7a6bc2963bc660e07446b00687f2dec77b8a88e Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:50 +0200 Subject: [PATCH 27/49] fix(ci): use Go 1.27 for Darwin backends (#12284) Older Go linkers stamp pure-Go hosts with SDK metadata that disables modern Metal APIs. Select Go 1.27 for Darwin builds and document the backend rebuild requirement. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/backend.yml | 2 +- .github/workflows/backend_build_darwin.yml | 3 ++- .github/workflows/backend_pr.yml | 2 +- docs/content/getting-started/build.md | 4 ++++ 4 files changed, 8 insertions(+), 3 deletions(-) diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 13e67c6fe..d4bb0bf59 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -355,7 +355,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/.github/workflows/backend_build_darwin.yml b/.github/workflows/backend_build_darwin.yml index 6b8b2a89b..19952a5ef 100644 --- a/.github/workflows/backend_build_darwin.yml +++ b/.github/workflows/backend_build_darwin.yml @@ -22,7 +22,8 @@ on: type: string go-version: description: 'Go version to use' - default: '1.24.x' + # Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs. + default: '1.27.x' type: string tag-suffix: description: 'Tag suffix for the built image' diff --git a/.github/workflows/backend_pr.yml b/.github/workflows/backend_pr.yml index c13c444c4..2626f87e8 100644 --- a/.github/workflows/backend_pr.yml +++ b/.github/workflows/backend_pr.yml @@ -281,7 +281,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/docs/content/getting-started/build.md b/docs/content/getting-started/build.md index 77884a412..42e64928d 100644 --- a/docs/content/getting-started/build.md +++ b/docs/content/getting-started/build.md @@ -30,6 +30,10 @@ To install the dependencies follow the instructions below: {{< tabs >}} {{% tab title="Apple" %}} +To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later. +Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts. +Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables. + Install `xcode` from the App Store ```bash From 4a691099abeda53aeb16fdac1596926006c98f31 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:53 +0000 Subject: [PATCH 28/49] chore(gallery): serve ternary-bonsai-2-27b with the bonsai backend PTQ1_0 is a Prism-private GGUF type (GGML_TYPE_PTQ1_0 = 143 in the PrismML llama.cpp fork), so stock llama-cpp cannot load it. Switch to the bonsai backend like the existing ternary-bonsai-27b entries, and replace the scraped Qwen3.8 description and icon. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 28 ++++++++++++---------------- 1 file changed, 12 insertions(+), 16 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index cceda1b77..a348c60dc 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -3,34 +3,30 @@ url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + - https://github.com/PrismML-Eng/llama.cpp description: | - # Qwen3.8-27B - - > [!Note] - > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. - > - > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. - - > [!Tip] - > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. - > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. - - Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. - - ... + Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary + transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits + per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a + Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's + llama.cpp fork) instead of stock llama.cpp. license: "apache-2.0" tags: - llm - gguf - icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + - reasoning + - vision + - multimodal + icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg overrides: - backend: llama-cpp + backend: bonsai function: automatic_tool_parsing_fallback: true grammar: disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf options: - use_jinja:true From 4e94c914c945fdf2f4a19adc4db2f1dc4c8d7c07 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:54 +0200 Subject: [PATCH 29/49] fix(swagger): describe backend metadata as an object (#12178) Swag cannot resolve json.RawMessage in OpenAIResponse and aborts the daily schema generation. Set its Swagger type without changing JSON encoding, and regenerate the checked-in specifications. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/schema/openai.go | 2 +- swagger/docs.go | 3 +++ swagger/swagger.json | 3 +++ swagger/swagger.yaml | 2 ++ 4 files changed, 9 insertions(+), 1 deletion(-) diff --git a/core/schema/openai.go b/core/schema/openai.go index 2aa69969b..6f3717256 100644 --- a/core/schema/openai.go +++ b/core/schema/openai.go @@ -99,7 +99,7 @@ type OpenAIResponse struct { // OpenAI-SDK consumers that filter on a truthy `result.usage` // (continuedev/continue, Kilo Code, Roo Code, etc.). Usage *OpenAIUsage `json:"usage,omitempty"` - Metadata json.RawMessage `json:"metadata,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"` } // StreamOptions mirrors OpenAI's `stream_options` request field. The only diff --git a/swagger/docs.go b/swagger/docs.go index 6ae74b94a..dc2fcf063 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -7313,6 +7313,9 @@ const docTemplate = `{ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, diff --git a/swagger/swagger.json b/swagger/swagger.json index c19487f6c..a1e71cbdb 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -7310,6 +7310,9 @@ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 6f9f6b1b7..12de303a0 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2266,6 +2266,8 @@ definitions: type: array id: type: string + metadata: + type: object model: type: string object: From 657cf9ca838e3acea2c242d10918631eca05cbd9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:59 +0200 Subject: [PATCH 30/49] fix(ci): retain backend digests for release retries (#12160) The v4.10.0 ace-step and VibeVoice merge jobs started just after their digest artifacts expired. Keep the small digest artifacts for seven days so a multi-day release matrix can finish publishing its images. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/backend_build.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.github/workflows/backend_build.yml b/.github/workflows/backend_build.yml index 05d50cf82..3e3af89f0 100644 --- a/.github/workflows/backend_build.yml +++ b/.github/workflows/backend_build.yml @@ -252,7 +252,8 @@ jobs: name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }} path: /tmp/digests/* if-no-files-found: error - retention-days: 1 + # Release matrices and their retries can outlive a one-day artifact. + retention-days: 7 - name: Build (PR) uses: docker/build-push-action@v7 From c6f1e96a7d9f515c95844afb6e1217ba380e9f54 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:07:04 +0200 Subject: [PATCH 31/49] chore(website): refresh the counters (#12039) Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- website/data/stats.yaml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/website/data/stats.yaml b/website/data/stats.yaml index a54e240e8..253b5444d 100644 --- a/website/data/stats.yaml +++ b/website/data/stats.yaml @@ -3,10 +3,10 @@ # The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh, # which runs weekly from .github/workflows/refresh-site-counters.yml. Editing # them by hand works but will be overwritten on the next run. -stars: 48949 -forks: 4430 -contributors: 237 -releases: 135 +stars: 49204 +forks: 4459 +contributors: 245 +releases: 136 # The GitHub API cannot answer for this one, so it is maintained by hand and # the refresh script carries it through untouched. From 1b1bd0f0694411910fb84ec0079fad6bf2c3cd34 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:07:08 +0200 Subject: [PATCH 32/49] fix(compose): request NVIDIA compute capability (#11990) The legacy NVIDIA device reservation requests utility without compute. Docker derives driver capabilities from that list, leaving CUDA libraries unavailable even when monitoring works. Include compute in the legacy example and clarify the matching docs. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- docker-compose.yaml | 3 ++- docs/content/features/distributed-mode.md | 8 ++++++-- docs/content/reference/nvidia-l4t.md | 6 ++++-- 3 files changed, 12 insertions(+), 5 deletions(-) diff --git a/docker-compose.yaml b/docker-compose.yaml index ee137e83c..82b3c18b6 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -59,6 +59,7 @@ services: # capabilities: [gpu, utility] # # For legacy NVIDIA driver (for older NVIDIA Container Toolkit): + # Request compute for CUDA libraries (libcuda.so.1) and utility for NVML. # environment: # NVIDIA_DRIVER_CAPABILITIES: "compute,utility" # init: true @@ -68,7 +69,7 @@ services: # devices: # - driver: nvidia # count: 1 - # capabilities: [gpu, utility] + # capabilities: [gpu, compute, utility] ## Uncomment for PostgreSQL-backed knowledge base (see Agents docs) # postgres: diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md index f8a06539a..86ad8b14e 100644 --- a/docs/content/features/distributed-mode.md +++ b/docs/content/features/distributed-mode.md @@ -417,8 +417,12 @@ usage is reported back to the frontend: NVML library (and therefore `nvidia-smi`) is not available inside the container. CUDA compute still works, but the worker cannot query free VRAM and the Nodes page will show the node as fully used. Set - `NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI - runtime, list `capabilities: [gpu, utility]` on the device reservation). + `NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime. + For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Docker derives driver capabilities from this reservation, so include `compute` + for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone + enables monitoring but does not provide CUDA libraries. - **Run the container with `init: true` (or `docker run --init`).** The worker process becomes PID 1 in the container and cannot reap zombies on diff --git a/docs/content/reference/nvidia-l4t.md b/docs/content/reference/nvidia-l4t.md index 2adac3a84..e3b54020a 100644 --- a/docs/content/reference/nvidia-l4t.md +++ b/docs/content/reference/nvidia-l4t.md @@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things: NVML work inside the container. With `--gpus all` alone (or `--runtime nvidia` without extra flags) only `compute` is wired in on some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility` - to your `docker run`, or `capabilities: [gpu, utility]` in compose / - Kubernetes device reservations. + to your `docker run`. For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility` + alone only provides monitoring libraries and tools. 2. Pass `--init` to `docker run` (or `init: true` in compose) so the container has a proper PID 1 reaper - otherwise short-lived child processes like `nvidia-smi` can intermittently fail with From b9634e0339451318828868c79a788c6e26b2e711 Mon Sep 17 00:00:00 2001 From: Leoy Date: Mon, 28 Sep 2026 03:10:35 +0800 Subject: [PATCH 33/49] fix(modelartifacts): reuse committed sibling files for narrowed allow_patterns (#11484) A request with narrower allow_patterns hashes to a different CacheKey than an already-committed broader sibling, so committedResult misses and materializeLocked re-fetches files the sibling already holds. After the own-tree reuseMaterializedFile miss, consult committed sibling trees for the same Source (type+endpoint+repo+revision), re-verify the file through verifyDownloadedFile (full SHA-256, never size-only), and hard-link it into the writer's staging snapshot (copy fallback only on EXDEV). Each file is matched individually against the sibling's manifest, so a broader request can never inherit a narrower sibling's gaps as if complete. The sibling manifest set is loaded and source-matched once per materialization (files indexed by path) instead of once per staged file, so a models volume with 20 committed artifacts and a 300-file snapshot does one manifest pass rather than ~6000 reads and JSON parses. The sibling-reuse behavior cases live in the package's registered Ginkgo suite so repository test conventions apply. Refs #11047 Signed-off-by: supermario_leo --- pkg/modelartifacts/materializer.go | 141 +++++++++++ .../materializer_sibling_reuse_test.go | 230 ++++++++++++++++++ 2 files changed, 371 insertions(+) create mode 100644 pkg/modelartifacts/materializer_sibling_reuse_test.go diff --git a/pkg/modelartifacts/materializer.go b/pkg/modelartifacts/materializer.go index a3e2a9e8e..87036f756 100644 --- a/pkg/modelartifacts/materializer.go +++ b/pkg/modelartifacts/materializer.go @@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedFiles := 0 skippedBytes := int64(0) tasks := make([]downloader.FileTask, 0, len(snapshot.Files)) + // Sibling manifests are read once, before the staging loop, so the + // per-file reuse lookups below never re-read or re-parse them. + siblings := loadSiblingCandidates(modelsPath, spec, layout) for index, file := range snapshot.Files { if err := ctx.Err(); err != nil { return Result{}, err @@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedBytes += file.Size continue } + // Before reaching for the network, consult committed sibling trees for the + // same Source (type+endpoint+repo+revision). A narrower allow_patterns + // request gets a different CacheKey, so committedResult misses even though a + // broader sibling already holds this exact file; reusing it avoids a + // redundant re-download of tens of gigabytes. The match is re-verified + // through verifyDownloadedFile (full SHA-256), never size-only, and a broader + // request can never inherit a narrower sibling's gaps because each file is + // matched individually against the sibling's manifest. + if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok { + manifest.Files[taskIndex] = entry + completedBytes.Add(file.Size) + skippedFiles++ + skippedBytes += file.Size + continue + } nameSum := sha256.Sum256([]byte(file.Path)) blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:])) blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel)) @@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest return entry, true } +// siblingCandidate is one committed sibling artifact tree that shares this +// request's Source (type+endpoint+repo+revision), with its manifest files +// indexed by path. +type siblingCandidate struct { + final string + filesByPath map[string][]ManifestFile +} + +// loadSiblingCandidates reads the committed sibling manifest set once, before +// the staging loop. Doing it per file instead would re-read and re-parse every +// sibling manifest for every file — 20 committed siblings and a 300-file +// snapshot means 6000 manifest reads before the first byte is fetched. +// +// The current artifact's own committed tree is excluded: it is either absent +// (the reason materializeLocked is running) or already handled by +// committedResult's exact-key fast path. +func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate { + if spec.Resolved == nil || layout.Final == "" { + return nil + } + siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface") + entries, err := os.ReadDir(siblingsRoot) + if err != nil { + return nil + } + var candidates []siblingCandidate + for _, entry := range entries { + if !entry.IsDir() { + continue + } + siblingFinal := filepath.Join(siblingsRoot, entry.Name()) + if siblingFinal == layout.Final { + continue + } + siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json")) + if err != nil { + continue + } + siblingArtifact := siblingManifest.Artifact + if siblingArtifact.Resolved == nil || + siblingArtifact.Source.Type != spec.Source.Type || + siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint || + siblingArtifact.Source.Repo != spec.Source.Repo || + siblingArtifact.Resolved.Revision != spec.Resolved.Revision { + continue + } + byPath := make(map[string][]ManifestFile, len(siblingManifest.Files)) + for _, f := range siblingManifest.Files { + byPath[f.Path] = append(byPath[f.Path], f) + } + candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath}) + } + return candidates +} + +// reuseFromCommittedSibling looks for a file already committed under a sibling +// artifact tree — same Source (type+endpoint+repo+revision), different +// allow/ignore patterns — and stages it for this writer instead of fetching. +// A narrower allow_patterns request gets a different CacheKey (path.go:62), so +// committedResult misses and materializeLocked would otherwise re-download +// files an already-committed broader sibling already holds. +// +// The match is never size-only: the sibling file is re-hashed through the +// shared verifyDownloadedFile against the current request's SnapshotFile (its +// LFS or git blob OID), so the staged entry is byte-for-byte identical to a +// fresh download. A broader request can never stand in for files a narrower +// sibling lacks, because each requested file is matched individually against +// the sibling's manifest file set. Hard-link keeps the shared models volume +// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the +// kernel cannot hard-link. +func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) { + snapshotRel := path.Join("snapshot", file.Path) + snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel)) + for _, sibling := range candidates { + for _, siblingFile := range sibling.filesByPath[file.Path] { + if siblingFile.Size != file.Size { + continue + } + siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path)) + verified, err := verifyDownloadedFile(siblingPath, file) + if err != nil { + continue + } + if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil { + return ManifestFile{}, false + } + _ = root.Remove(snapshotRel) + if err := linkOrCopy(siblingPath, snapshotAbs); err != nil { + return ManifestFile{}, false + } + return verified, true + } + } + return ManifestFile{}, false +} + +// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only +// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking +// keeps the shared models volume neutral — a narrowed request does not double +// the storage of a broad sibling's files. +func linkOrCopy(src, dst string) error { + if err := os.Link(src, dst); err == nil { + return nil + } else if !errors.Is(err, syscall.EXDEV) { + return err + } + in, err := os.Open(src) + if err != nil { + return err + } + defer func() { _ = in.Close() }() + out, err := os.Create(dst) + if err != nil { + return err + } + if _, err := io.Copy(out, in); err != nil { + _ = out.Close() + _ = os.Remove(dst) + return err + } + return out.Close() +} + func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) { file, err := os.Open(fileName) if err != nil { diff --git a/pkg/modelartifacts/materializer_sibling_reuse_test.go b/pkg/modelartifacts/materializer_sibling_reuse_test.go new file mode 100644 index 000000000..572b255d8 --- /dev/null +++ b/pkg/modelartifacts/materializer_sibling_reuse_test.go @@ -0,0 +1,230 @@ +package modelartifacts_test + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strconv" + "strings" + "sync" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + hfapi "github.com/mudler/LocalAI/pkg/huggingface-api" + "github.com/mudler/LocalAI/pkg/modelartifacts" +) + +const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567" + +// recordingResolver serves a fixed full file set filtered by each request's +// allow/ignore patterns, so a narrower request genuinely resolves to a strict +// subset of a broader sibling's files. The HTTP server behind it records every +// fetch, which is the signal the sibling-reuse fix is verified through. The +// function under fix is never mocked: a real Manager drives the real staging + +// commit path against this stub collaborator. +type recordingResolver struct { + endpoint string + repo string + files []hfapi.SnapshotFile + server *httptest.Server + + mu sync.Mutex + fetched map[string]int +} + +func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver { + r := &recordingResolver{ + endpoint: "https://huggingface.co", + repo: "owner/repo", + files: files, + fetched: map[string]int{}, + } + r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + name := strings.TrimPrefix(req.URL.Path, "/file/") + body, ok := contents[name] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + r.mu.Lock() + r.fetched[name]++ + r.mu.Unlock() + w.Header().Set("Content-Length", strconv.Itoa(len(body))) + _, _ = w.Write(body) + })) + return r +} + +func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) { + files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns) + if err != nil { + return hfapi.Snapshot{}, err + } + out := make([]hfapi.SnapshotFile, len(files)) + for i, f := range files { + f.URL = r.server.URL + "/file/" + f.Path + out[i] = f + } + return hfapi.Snapshot{ + Endpoint: r.endpoint, Repo: r.repo, + RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out, + }, nil +} + +func (r *recordingResolver) fetchCount(path string) int { + r.mu.Lock() + defer r.mu.Unlock() + return r.fetched[path] +} + +func (r *recordingResolver) resetFetches() { + r.mu.Lock() + defer r.mu.Unlock() + r.fetched = map[string]int{} +} + +func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile { + paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"} + files := make([]hfapi.SnapshotFile, 0, len(paths)) + for _, p := range paths { + sum := sha256.Sum256(contents[p]) + files = append(files, hfapi.SnapshotFile{ + Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]), + }) + } + return files +} + +// The narrow-request case proves the fix for #11047: +// a request with narrower allow_patterns (a strict subset) reuses files an +// already-committed broader sibling holds, hard-linking instead of re-fetching. +// +// On master this is RED: a narrower allow_patterns set hashes to a different +// CacheKey (path.go:62), so committedResult misses and materializeLocked +// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On +// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling, +// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0, +// os.SameFile true). +var _ = Describe("committed sibling reuse", func() { + It("reuses files from a broader committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit the broad sibling: all three files, fetched from the resolver. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(broad.CacheHit).To(BeFalse()) + Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0), + "the broad sibling must have fetched a/first.bin to commit it") + + resolver.resetFetches() + + // Narrowed request: a strict subset of the broad sibling's file set. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(narrow.CacheHit).To(BeFalse()) + + // (b) The sibling-present file must NOT be re-fetched: zero fetches. This is + // the assertion that is RED on master (one fetch) and GREEN on the branch. + Expect(resolver.fetchCount("a/first.bin")).To(Equal(0), + "a/first.bin must be reused from the committed broad sibling, not re-fetched") + + // (a) The narrowed tree's staged file is the same inode as the broad + // sibling's file (hard-link), not a freshly downloaded second copy. RED on + // master (separate file), GREEN on the branch (hard-link). + broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin") + narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin") + broadInfo, err := os.Stat(broadFile) + Expect(err).NotTo(HaveOccurred()) + narrowInfo, err := os.Stat(narrowFile) + Expect(err).NotTo(HaveOccurred()) + Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(), + "the narrowed request must hard-link the broad sibling's file rather than store a second copy") + + // The reused bytes are intact end to end. + Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"])) + }) + + // The broader-request case is the manifest file-set guard: a broader request + // against a narrower committed sibling must still fetch the files the sibling + // lacks and commit a complete tree. Sibling-reuse can never serve an incomplete + // model as complete, because each requested file is matched individually against + // the sibling's manifest. + It("fetches files missing from a narrower committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit a NARROW sibling first: only a/first.bin and b/second.bin. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin", "b/second.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + narrowPaths := make([]string, 0, len(narrow.Manifest.Files)) + for _, f := range narrow.Manifest.Files { + narrowPaths = append(narrowPaths, f.Path) + } + Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"})) + + resolver.resetFetches() + + // A BROADER request asks for all three files, including c/third.bin which the + // narrow sibling does not hold. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + + // The file the narrow sibling lacks MUST be fetched: sibling-reuse must not + // inherit a narrower tree's gaps as if the broad request were complete. + Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0), + "c/third.bin is absent from the narrow sibling and must be fetched, not served as complete") + + // The broad tree's manifest file set is exactly the full set — never the + // narrow sibling's subset. This file-set comparison proves no incomplete model + // is ever served as complete via sibling-reuse. + broadPaths := make([]string, 0, len(broad.Manifest.Files)) + for _, f := range broad.Manifest.Files { + broadPaths = append(broadPaths, f.Path) + } + Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"})) + + // Every file is present on disk with the right bytes after commit. + for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} { + Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))). + To(Equal(contents[p])) + } + }) +}) From 0e52bb657e8c58f68fba24c62589f46466958103 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:24 +0200 Subject: [PATCH 34/49] fix(responses): wait for complete JSON tool calls (#12001) Partial JSON parsing heals a name-only chunk into a tool call. The stream emits that call with empty arguments and skips later chunks. Require complete JSON before emitting terminal tool-call events. Preserve complete calls before an unfinished trailing call, and count only actual tool calls. Add split-chunk regression tests and docs. Refs #11635. The non-streaming report remains unconfirmed. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .../http/endpoints/openresponses/responses.go | 68 ++++++++----------- .../openresponses/stream_tool_calls.go | 35 ++++++++++ .../openresponses/stream_tool_calls_test.go | 44 ++++++++++++ docs/content/features/text-generation.md | 5 ++ 4 files changed, 112 insertions(+), 40 deletions(-) create mode 100644 core/http/endpoints/openresponses/stream_tool_calls.go create mode 100644 core/http/endpoints/openresponses/stream_tool_calls_test.go diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index 553c01558..6da7f2adc 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 return true } - // Try JSON parsing as fallback - jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true) - if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount { + // Only completed JSON calls can be emitted as completed SSE items. + jsonResults := parseStreamingJSONToolCalls(cleanedResult) + if len(jsonResults) > lastEmittedToolCallCount { for i := lastEmittedToolCallCount; i < len(jsonResults); i++ { - jsonObj := jsonResults[i] - if name, ok := jsonObj["name"].(string); ok && name != "" { - args := "{}" - if argsVal, ok := jsonObj["arguments"]; ok { - if argsStr, ok := argsVal.(string); ok { - args = argsStr - } else { - argsBytes, _ := json.Marshal(argsVal) - args = string(argsBytes) - } - } + tc := jsonResults[i] + toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) + outputIndex++ - toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) - outputIndex++ - - functionCallItem := &schema.ORItemField{ - Type: "function_call", - ID: toolCallID, - Status: "completed", - CallID: toolCallID, - Name: name, - Arguments: args, - } - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.added", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ - - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.done", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ + functionCallItem := &schema.ORItemField{ + Type: "function_call", + ID: toolCallID, + Status: "completed", + CallID: toolCallID, + Name: tc.Name, + Arguments: tc.Arguments, } + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.added", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ + + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.done", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ } lastEmittedToolCallCount = len(jsonResults) c.Response().Flush() diff --git a/core/http/endpoints/openresponses/stream_tool_calls.go b/core/http/endpoints/openresponses/stream_tool_calls.go new file mode 100644 index 000000000..b8f0f185d --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls.go @@ -0,0 +1,35 @@ +package openresponses + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/pkg/functions" +) + +func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults { + // Partial parsing heals unfinished arguments. The caller emits terminal + // events and never revisits emitted calls, so only accept complete JSON. + // Keep completed objects returned before an unfinished trailing object. + objects, _ := functions.ParseJSONIterative(text, false) + var calls []functions.FuncCallResults + for _, object := range objects { + name, ok := object["name"].(string) + if !ok || name == "" { + continue + } + arguments := "{}" + if value, ok := object["arguments"]; ok { + if s, ok := value.(string); ok { + arguments = s + } else { + data, err := json.Marshal(value) + if err != nil { + continue + } + arguments = string(data) + } + } + calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments}) + } + return calls +} diff --git a/core/http/endpoints/openresponses/stream_tool_calls_test.go b/core/http/endpoints/openresponses/stream_tool_calls_test.go new file mode 100644 index 000000000..1fa6bca2a --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls_test.go @@ -0,0 +1,44 @@ +package openresponses + +import ( + "github.com/mudler/LocalAI/pkg/functions" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Streaming JSON tool calls", func() { + It("waits for the arguments before completing a split call", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("does not complete a call at any intermediate token boundary", func() { + text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}` + for end := 1; end < len(text); end++ { + Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end]) + } + Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1)) + }) + + It("keeps completed calls while the next call is incomplete", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("preserves string arguments and calls that take no arguments", func() { + Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + {Name: "status", Arguments: `{}`}, + })) + }) + + It("does not count unrelated JSON objects as emitted calls", func() { + Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "status", Arguments: `{}`}, + })) + }) +}) diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 490877e21..286359339 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -434,6 +434,11 @@ curl http://localhost:8080/v1/responses \ }' ``` +For streaming requests with JSON tool output, LocalAI waits for the complete JSON +object before emitting a completed `function_call` item. Arguments can span +multiple tokens. Read the arguments from the `response.output_item.done` event +before executing the tool. + #### Reasoning Configuration Configure reasoning effort and summary style: From 5794495a37b0cdc186ac6a14648991c730042499 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:29 +0200 Subject: [PATCH 35/49] fix(responses): preserve streamed output items (#12048) Keep each message and reasoning item at its announced output index. Include the answer in completed responses with reasoning or fallback function calls, and retain reasoning supplied through backend deltas. Add regression coverage for stream indices, final output, plain text, and automatic tool parsing. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .../http/endpoints/openresponses/responses.go | 60 +++---- .../openresponses/responses_stream_test.go | 148 ++++++++++++++++++ docs/content/features/text-generation.md | 9 ++ 3 files changed, 176 insertions(+), 41 deletions(-) create mode 100644 core/http/endpoints/openresponses/responses_stream_test.go diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index 6da7f2adc..d62fa7534 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -2412,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Non-tool-call streaming path + messageOutputIndex := outputIndex + var reasoningOutputIndex int // Emit output_item.added for message currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String()) messageItem := &schema.ORItemField{ @@ -2424,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2436,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &emptyTextPart, }) @@ -2459,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Handle reasoning item - if extractor.Reasoning() != "" { + if extractor.Reasoning() != "" || reasoningDelta != "" { // Check if we need to create reasoning item if currentReasoningID == "" { outputIndex++ + reasoningOutputIndex = outputIndex currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String()) reasoningItem := &schema.ORItemField{ Type: "reasoning", @@ -2472,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2484,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &emptyPart, }) @@ -2497,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Delta: strPtr(reasoningDelta), Logprobs: emptyLogprobs(), @@ -2514,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Delta: strPtr(contentDelta), Logprobs: emptyLogprobs(), @@ -2583,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Text: strPtr(finalReasoning), Logprobs: emptyLogprobs(), @@ -2596,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &reasoningPart, }) @@ -2612,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2646,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Text: strPtr(result), Logprobs: logprobsPtr(mcpStreamLogprobs), @@ -2659,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &resultPart, }) @@ -2671,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2711,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 // Emit response.completed now := time.Now().Unix() - // Collect final output items (reasoning first, then messages, then tool calls) - var finalOutputItems []schema.ORItemField - // Add reasoning item if it exists - if currentReasoningID != "" && finalReasoning != "" { - finalOutputItems = append(finalOutputItems, schema.ORItemField{ - Type: "reasoning", - ID: currentReasoningID, - Status: "completed", - Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)}, - }) - } - // Add message item - if len(collectedOutputItems) > 0 { - // Use collected items (may include reasoning already) - for _, item := range collectedOutputItems { - if item.Type == "message" { - finalOutputItems = append(finalOutputItems, item) - } - } - } else { - finalOutputItems = append(finalOutputItems, *messageItem) - } - // Add function_call items from fallback - for _, item := range collectedOutputItems { - if item.Type == "function_call" { - finalOutputItems = append(finalOutputItems, item) - } - } + // The final output array must use the indices announced in the stream. + // The message is opened first, followed by reasoning and fallback calls. + finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...) responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{ InputTokens: noToolTokenUsage.Prompt, OutputTokens: noToolTokenUsage.Completion, diff --git a/core/http/endpoints/openresponses/responses_stream_test.go b/core/http/endpoints/openresponses/responses_stream_test.go new file mode 100644 index 000000000..13ddb10e1 --- /dev/null +++ b/core/http/endpoints/openresponses/responses_stream_test.go @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: MIT +package openresponses + +import ( + "context" + "encoding/json" + "net/http/httptest" + "strings" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/schema" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/LocalAI/pkg/model" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Responses stream item consistency", func() { + DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) { + originalInference := backend.ModelInferenceFunc + DeferCleanup(func() { backend.ModelInferenceFunc = originalInference }) + backend.ModelInferenceFunc = func( + ctx context.Context, prompt string, messages schema.Messages, + images, videos, audios []string, loader *model.ModelLoader, + cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig, + tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string, + logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string, + ) (func() (backend.LLMResponse, error), error) { + return func() (backend.LLMResponse, error) { + for i, token := range tokens { + usage := backend.TokenUsage{} + if len(chatDeltas) > 0 { + usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]} + } + if !tokenCallback(token, usage) { + break + } + } + return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil + }, nil + } + cfg := &config.ModelConfig{} + cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback + cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)(.*?)`} + recorder := httptest.NewRecorder() + request := httptest.NewRequest("POST", "/v1/responses", nil) + c := echo.New().NewContext(request, recorder) + input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true} + err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil) + Expect(err).NotTo(HaveOccurred()) + Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n")) + + var events []schema.ORStreamEvent + var completed *schema.ORResponseResource + for _, line := range strings.Split(recorder.Body.String(), "\n") { + if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" { + continue + } + var event schema.ORStreamEvent + Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed()) + Expect(event.Type).NotTo(Equal("error")) + events = append(events, event) + if event.Type == "response.completed" { + completed = event.Response + } + } + Expect(completed).NotTo(BeNil()) + wantCount := 1 + if wantReasoning != "" { + wantCount++ + } + if fallback { + wantCount++ + } + Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls") + + indices := map[string]int{} + done := map[string]int{} + deltas := map[string]string{} + for i, event := range events { + Expect(event.SequenceNumber).To(Equal(i)) + if event.Type == "response.output_item.added" { + Expect(event.Item).NotTo(BeNil()) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(indices).NotTo(HaveKey(event.Item.ID)) + Expect(*event.OutputIndex).To(Equal(len(indices))) + indices[event.Item.ID] = *event.OutputIndex + } + id := event.ItemID + if event.Item != nil { + id = event.Item.ID + } + if id == "" { + continue + } + Expect(indices).To(HaveKey(id)) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id) + Expect(completed.Output[indices[id]].ID).To(Equal(id)) + if event.Type == "response.output_item.done" { + done[id]++ + Expect(event.Item.Status).To(Equal("completed")) + Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type)) + if event.Item.Type == "function_call" { + Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name)) + Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments)) + } else { + Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content)) + } + } + if event.Type == "response.output_text.delta" { + deltas[id] += *event.Delta + } + } + Expect(indices).To(HaveLen(wantCount)) + for _, item := range completed.Output { + Expect(done[item.ID]).To(Equal(1)) + switch item.Type { + case "message", "reasoning": + want := wantAnswer + if item.Type == "reasoning" { + want = wantReasoning + } + parts, ok := item.Content.([]any) + Expect(ok).To(BeTrue()) + Expect(parts).To(HaveLen(1)) + Expect(parts[0].(map[string]any)["text"]).To(Equal(want)) + if !fallback { + Expect(deltas[item.ID]).To(Equal(want)) + } + case "function_call": + Expect(item.Name).To(Equal("get_weather")) + Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`)) + Expect(item.CallID).NotTo(BeEmpty()) + default: + Fail("unexpected output item type: " + item.Type) + } + } + }, + Entry("tagged reasoning and answer", []string{"", "Let me think.", "", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false), + Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false), + Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false), + Entry("automatic fallback tool call", []string{`{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "", "", true), + Entry("reasoning and automatic fallback tool call", []string{"", "Let me think.", "", `{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "Let me think.", "", true), + ) +}) diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 286359339..dadb5e9af 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -340,6 +340,15 @@ curl http://localhost:8080/v1/responses \ }' ``` +#### Streaming responses + +Set `"stream": true` to receive Server-Sent Events. Each `response.output_item.added` event assigns an `output_index` to an item. +Use that index and the item ID to associate later deltas and completion events with the same item. + +If a request without explicit tools produces reasoning, the stream uses separate items for reasoning and answer text. +Each item keeps its original index throughout the stream. +The `response.completed` event includes both items in the same index order, followed by any automatically parsed tool calls. + #### Background Processing Run requests in the background for long-running tasks: From f154bd990a720130959e02c8eeea4e571e27117b Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:34 +0200 Subject: [PATCH 36/49] feat(system): report per-model DRM VRAM (#12026) * feat(system): report per-model DRM VRAM Expose optional resident device memory for local backend process trees. Deduplicate DRM clients and omit unsupported or incomplete readings. Document accounting limits and preserve a measured zero in JSON. Closes #11970. Assisted-by: Codex:gpt-6 * fix(system): document trusted procfs reads Scope G304 annotations to paths built from the fixed procfs root, integer process IDs, and kernel directory entries. These reads accept no user-controlled path components. Assisted-by: Codex:GPT-6 gosec --------- Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/http/endpoints/localai/system.go | 6 + .../endpoints/localai/system_info_test.go | 49 ++++++ core/schema/localai.go | 3 + core/schema/system_info_test.go | 25 +++ docs/content/reference/system-info.md | 24 +++ pkg/xsysinfo/process_vram_linux.go | 165 ++++++++++++++++++ pkg/xsysinfo/process_vram_linux_test.go | 105 +++++++++++ pkg/xsysinfo/process_vram_other.go | 9 + swagger/docs.go | 4 + swagger/swagger.json | 4 + swagger/swagger.yaml | 5 + 11 files changed, 399 insertions(+) create mode 100644 core/http/endpoints/localai/system_info_test.go create mode 100644 core/schema/system_info_test.go create mode 100644 pkg/xsysinfo/process_vram_linux.go create mode 100644 pkg/xsysinfo/process_vram_linux_test.go create mode 100644 pkg/xsysinfo/process_vram_other.go diff --git a/core/http/endpoints/localai/system.go b/core/http/endpoints/localai/system.go index 996c9a781..9c4ba7503 100644 --- a/core/http/endpoints/localai/system.go +++ b/core/http/endpoints/localai/system.go @@ -8,6 +8,7 @@ import ( "github.com/mudler/LocalAI/core/schema" "github.com/mudler/LocalAI/core/services/monitoring" "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/xsysinfo" ) // SystemInformations returns the system informations @@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app entry.Process = proc } } + if pid, ok := localPID(m); ok { + if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok { + entry.SizeVRAM = &used + } + } sysmodels = append(sysmodels, entry) } if sampler != nil { diff --git a/core/http/endpoints/localai/system_info_test.go b/core/http/endpoints/localai/system_info_test.go new file mode 100644 index 000000000..83f7daa07 --- /dev/null +++ b/core/http/endpoints/localai/system_info_test.go @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: MIT +package localai_test + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/http/endpoints/localai" + "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/system" + process "github.com/mudler/go-processmanager" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SystemInformations memory", func() { + It("keeps model metadata and omits VRAM for remote or stopped backends", func() { + path, err := os.MkdirTemp("", "system-info-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, path) + configFile := filepath.Join(path, "remote.yaml") + Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed()) + cl := config.NewModelConfigLoader(path) + Expect(cl.ReadModelConfig(configFile)).To(Succeed()) + ml := model.NewModelLoader(&system.SystemState{}) + store := model.NewInMemoryModelStore() + store.Set("remote", model.NewModel("remote", "worker:50051", nil)) + store.Set("stopped", model.NewModel("stopped", "", &process.Process{})) + ml.SetModelStore(store) + app := echo.New() + app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil)) + rec := httptest.NewRecorder() + app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil)) + Expect(rec.Code).To(Equal(http.StatusOK)) + var response struct { + Models []map[string]any `json:"loaded_models"` + } + Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed()) + Expect(response.Models).To(ConsistOf( + map[string]any{"id": "remote", "backend": "llama-cpp"}, + map[string]any{"id": "stopped"}, + )) + }) +}) diff --git a/core/schema/localai.go b/core/schema/localai.go index 7e5d5e314..dc99a1dbe 100644 --- a/core/schema/localai.go +++ b/core/schema/localai.go @@ -208,6 +208,9 @@ type SysInfoModel struct { // when the model has no local process (a distributed worker holds it) or // the process could not be read. Process *SysInfoProcess `json:"process,omitempty"` + // SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + // the backend process tree has no complete supported reading. + SizeVRAM *uint64 `json:"size_vram,omitempty"` } // SysInfoProcess is a point-in-time reading of one backend process. diff --git a/core/schema/system_info_test.go b/core/schema/system_info_test.go new file mode 100644 index 000000000..79a1bdba1 --- /dev/null +++ b/core/schema/system_info_test.go @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: MIT +package schema_test + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/core/schema" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SysInfoModel memory", func() { + It("omits unavailable VRAM while preserving a measured zero", func() { + entry := schema.SysInfoModel{ID: "model"} + encoded, err := json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`)) + + zero := uint64(0) + entry.SizeVRAM = &zero + encoded, err = json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`)) + }) +}) diff --git a/docs/content/reference/system-info.md b/docs/content/reference/system-info.md index b825e4e06..ed4de00ff 100644 --- a/docs/content/reference/system-info.md +++ b/docs/content/reference/system-info.md @@ -28,6 +28,29 @@ Returns available backends and currently loaded models. | `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM | | `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against | | `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) | +| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes | + +### Per-model VRAM + +On Linux, `size_vram` reports resident device memory for the local backend +process and its child processes. LocalAI reads `drm-resident-local*` and +`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU. +Host-memory regions are excluded. The reading includes buffers attributed +to the backend, without separating weights, KV cache, and other allocations. +See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html) +for these counters. + +The field is omitted when accounting is unavailable or incomplete. This +includes external and distributed backends, macOS, proprietary NVIDIA +drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters, +and unreadable process information. +A present value of `0` means the supported counters report zero bytes. +Treat an absent field as unknown. + +This is a snapshot of driver accounting, not a memory reservation. Shared +buffers can appear in different clients' counters, and allocations can change +during collection. Do not treat the sum across models as exclusive physical +GPU usage. These readings do not replace capacity checks when scheduling work. ### Usage @@ -49,6 +72,7 @@ curl http://localhost:8080/system { "id": "my-llama-model", "backend": "llama-cpp", + "size_vram": 5368709120, "process": { "pid": 48213, "rss_bytes": 5368709120, diff --git a/pkg/xsysinfo/process_vram_linux.go b/pkg/xsysinfo/process_vram_linux.go new file mode 100644 index 000000000..3dd0a59a9 --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux.go @@ -0,0 +1,165 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "bufio" + "bytes" + "math" + "os" + "path/filepath" + "strconv" + "strings" +) + +// ProcessVRAM reports device-local resident bytes accounted to a process tree +// by DRM. Unsupported or incomplete accounting returns false, not a measured zero. +func ProcessVRAM(pid int) (uint64, bool) { + return processVRAM("/proc", pid) +} + +func processVRAM(procRoot string, pid int) (uint64, bool) { + if pid <= 0 { + return 0, false + } + clients := map[string]uint64{} + seen := map[int]bool{} + pending := []int{pid} + for len(pending) > 0 { + current := pending[len(pending)-1] + pending = pending[:len(pending)-1] + if seen[current] { + continue + } + seen[current] = true + base := filepath.Join(procRoot, strconv.Itoa(current)) + fds, err := os.ReadDir(filepath.Join(base, "fd")) + if err != nil { + return 0, false + } + for _, fd := range fds { + target, err := os.Readlink(filepath.Join(base, "fd", fd.Name())) + if err != nil { + return 0, false + } + // A mixed DRM/NVIDIA tree cannot provide a complete DRM reading. + if strings.HasPrefix(target, "/dev/nvidia") { + return 0, false + } + if !strings.HasPrefix(target, "/dev/dri/render") { + // Primary nodes can also own allocations. Until their device + // identity is resolved, omitting them would undercount the tree. + if strings.HasPrefix(target, "/dev/dri/") { + return 0, false + } + continue + } + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and fd.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name())) + if err != nil { + return 0, false + } + client, used, ok := drmResidentClient(data) + if !ok { + return 0, false + } + key := target + ":" + client + // dup() and fork() can expose the same client more than once. The + // snapshot is not atomic; retain its largest observed reading. + clients[key] = max(clients[key], used) + } + + // A worker may be spawned by any thread, not just the thread leader. + tasks, err := os.ReadDir(filepath.Join(base, "task")) + if err != nil || len(tasks) == 0 { + return 0, false + } + for _, task := range tasks { + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and task.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children")) + if err != nil { + return 0, false + } + for _, raw := range strings.Fields(string(data)) { + child, err := strconv.Atoi(raw) + if err != nil || child <= 0 { + return 0, false + } + pending = append(pending, child) + } + } + } + var total uint64 + for _, used := range clients { + if used > math.MaxUint64-total { + return 0, false + } + total += used + } + return total, len(clients) > 0 +} + +func drmResidentClient(data []byte) (string, uint64, bool) { + var client string + var total uint64 + found := false + scanner := bufio.NewScanner(bytes.NewReader(data)) + for scanner.Scan() { + key, value, ok := strings.Cut(scanner.Text(), ":") + if !ok { + continue + } + if key == "drm-client-id" { + id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + if err != nil { + return "", 0, false + } + client = strconv.FormatUint(id, 10) + } + region, resident := strings.CutPrefix(key, "drm-resident-") + if !resident || !isVRAMRegion(region) { + continue + } + used, ok := drmResidentBytes(value) + if !ok || used > math.MaxUint64-total { + return "", 0, false + } + total += used + found = true + } + return client, total, scanner.Err() == nil && client != "" && found +} + +func drmResidentBytes(value string) (uint64, bool) { + fields := strings.Fields(value) + if len(fields) == 0 || len(fields) > 2 { + return 0, false + } + n, err := strconv.ParseUint(fields[0], 10, 64) + if err != nil { + return 0, false + } + unit := uint64(1) + if len(fields) == 2 { + switch strings.ToLower(fields[1]) { + case "b": + case "kib": + unit = 1 << 10 + case "mib": + unit = 1 << 20 + case "gib": + unit = 1 << 30 + default: + return 0, false + } + } + if n > math.MaxUint64/unit { + return 0, false + } + return n * unit, true +} diff --git a/pkg/xsysinfo/process_vram_linux_test.go b/pkg/xsysinfo/process_vram_linux_test.go new file mode 100644 index 000000000..4de1f6cdd --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux_test.go @@ -0,0 +1,105 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "os" + "path/filepath" + "strconv" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("ProcessVRAM", func() { + var root string + write := func(path, contents string) { + Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed()) + Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed()) + } + addProcess := func(pid int, children string) { + base := filepath.Join(root, strconv.Itoa(pid)) + Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed()) + write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children) + } + addFD := func(pid, fd int, render, info string) { + base := filepath.Join(root, strconv.Itoa(pid)) + name := strconv.Itoa(fd) + Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed()) + write(filepath.Join(base, "fdinfo", name), info) + } + BeforeEach(func() { + var err error + root, err = os.MkdirTemp("", "process-vram-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, root) + addProcess(100, "") + }) + + It("sums resident device memory across GPUs and child processes without duplicate clients", func() { + write(filepath.Join(root, "100/task/101/children"), "200") + addProcess(200, "") + info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n" + addFD(100, 3, "renderD128", info) + addFD(100, 4, "renderD128", info) + addFD(200, 3, "renderD128", info) + addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(Equal(uint64(384 * 1024 * 1024))) + }) + + It("distinguishes a measured zero from unavailable accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(BeZero()) + }) + + DescribeTable("does not invent readings from unsupported or invalid accounting", + func(info string) { + addFD(100, 3, "renderD128", info) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"), + Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"), + Entry("no client identity", "drm-resident-vram0: 128 MiB\n"), + Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"), + Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"), + Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"), + ) + + It("omits a partial reading if a child cannot be inspected", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + write(filepath.Join(root, "100/task/100/children"), "200") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + It("omits a partial reading if another DRM client lacks accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + addFD(100, 4, "renderD129", "drm-client-id: 8\n") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + DescribeTable("omits mixed readings with unsupported GPU descriptors", + func(target string) { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed()) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("primary DRM node", "/dev/dri/card0"), + Entry("NVIDIA device", "/dev/nvidia0"), + ) + + It("returns unavailable for missing processes or no DRM descriptors", func() { + for _, pid := range []int{-1, 0, 100, 999} { + _, ok := processVRAM(root, pid) + Expect(ok).To(BeFalse()) + } + }) +}) diff --git a/pkg/xsysinfo/process_vram_other.go b/pkg/xsysinfo/process_vram_other.go new file mode 100644 index 000000000..06cc49186 --- /dev/null +++ b/pkg/xsysinfo/process_vram_other.go @@ -0,0 +1,9 @@ +//go:build !linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting. +func ProcessVRAM(pid int) (uint64, bool) { + return 0, false +} diff --git a/swagger/docs.go b/swagger/docs.go index dc2fcf063..c447043ae 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -7810,6 +7810,10 @@ const docTemplate = `{ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.json b/swagger/swagger.json index a1e71cbdb..b4e49b347 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -7807,6 +7807,10 @@ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 12de303a0..34fcfc8d7 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2654,6 +2654,11 @@ definitions: type: string id: type: string + size_vram: + description: |- + SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + the backend process tree has no complete supported reading. + type: integer type: object schema.SystemInformationResponse: properties: From 490b952d062dd13f1ba5bcc7eef297979cdc781d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:38 +0200 Subject: [PATCH 37/49] feat(gallery): publish signed OCI fallbacks (#12182) * feat(gallery): publish signed OCI fallbacks Publish both official gallery indexes with their local base configs so an outage of the HTTP and GitHub sources can fall back to Quay. Keep artifact signing policies separate from backend image policies, and expose each moving gallery tag only after its digest is signed. Assisted-by: Codex:gpt-6 * fix(gallery): confine packaged files to selected roots Use directory-scoped file access to reject symlink escapes during gallery packaging. Create private bundle files for the publishing runner. Assisted-by: Codex:GPT-6 --------- Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/gallery_publish.yml | 78 ++++++++++++++++ core/config/gallery.go | 12 ++- core/config/gallery_test.go | 21 +++++ core/config/runtime_settings_startup.go | 4 +- core/config/runtime_settings_startup_test.go | 12 ++- core/gallery/entry_url.go | 2 +- core/gallery/gallery.go | 2 +- core/gallery/gallery_mirrors.go | 4 +- core/gallery/gallery_oci.go | 18 ++-- core/gallery/gallery_oci_test.go | 15 ++++ docs/content/features/backends.md | 2 + docs/content/features/model-gallery.md | 20 +++-- scripts/build/gallery/main.go | 94 ++++++++++++++++++++ scripts/build/gallery/main_test.go | 83 +++++++++++++++++ 14 files changed, 345 insertions(+), 22 deletions(-) create mode 100644 .github/workflows/gallery_publish.yml create mode 100644 scripts/build/gallery/main.go create mode 100644 scripts/build/gallery/main_test.go diff --git a/.github/workflows/gallery_publish.yml b/.github/workflows/gallery_publish.yml new file mode 100644 index 000000000..c27b3d282 --- /dev/null +++ b/.github/workflows/gallery_publish.yml @@ -0,0 +1,78 @@ +name: Publish official OCI galleries + +on: + push: + branches: [master] + paths: + - 'gallery/**' + - 'backend/index.yaml' + - 'scripts/build/gallery/**' + - '.github/workflows/gallery_publish.yml' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: publish-official-galleries + cancel-in-progress: false + +jobs: + publish: + if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master' + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write + env: + COSIGN_EXPERIMENTAL: '1' + GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends + strategy: + matrix: + include: + - source: gallery + tag: gallery-models + - source: backend + tag: gallery-backends + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-go@v6 + with: + go-version-file: go.mod + - name: Test and package gallery + env: + GALLERY_SOURCE: ${{ matrix.source }} + run: | + go test ./scripts/build/gallery -count=1 + go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery" + - uses: oras-project/setup-oras@v1 + with: + version: '1.3.0' + - uses: sigstore/cosign-installer@v3 + with: + cosign-release: 'v2.6.5' + - name: Login to Quay.io + uses: docker/login-action@v4 + with: + registry: quay.io + username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }} + password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }} + - name: Publish and sign gallery + shell: bash + env: + GALLERY_TAG: ${{ matrix.tag }} + run: | + set -euo pipefail + cd "$RUNNER_TEMP/gallery" + files=() + while IFS= read -r -d '' file; do + files+=("${file#./}:application/yaml") + done < <(find . -type f -print0 | sort -z) + # Publish an immutable revision, then expose latest only after signing. + ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA" + oras push --artifact-type application/vnd.localai.gallery.v1 \ + --format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json" + digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json") + cosign sign --yes --new-bundle-format \ + --registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest" + oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG" diff --git a/core/config/gallery.go b/core/config/gallery.go index e22cbc94f..3f6b31ab9 100644 --- a/core/config/gallery.go +++ b/core/config/gallery.go @@ -47,10 +47,13 @@ type Gallery struct { // fallback for availability, not a load-balancing pool: the primary is // always preferred, and a mirror is only consulted after the one before // it fails. Any URI the gallery loader understands works here - // (https://, github:, file://). + // (https://, github:, file://, oci://). Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"` Name string `json:"name" yaml:"name"` Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"` + // ArtifactVerification overrides Verification only for the gallery OCI artifact. + // Backend images keep their separate Verification policy. + ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"` } // Equal reports whether two gallery entries describe the same gallery. @@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool { if !slices.Equal(g.Mirrors, other.Mirrors) { return false } + if g.ArtifactVerification == nil || other.ArtifactVerification == nil { + if g.ArtifactVerification != other.ArtifactVerification { + return false + } + } else if *g.ArtifactVerification != *other.ArtifactVerification { + return false + } if g.Verification == nil || other.Verification == nil { return g.Verification == other.Verification } diff --git a/core/config/gallery_test.go b/core/config/gallery_test.go index 71f4f3a36..7e1848a36 100644 --- a/core/config/gallery_test.go +++ b/core/config/gallery_test.go @@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() { Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery")) }) }) + +var _ = Describe("Gallery artifact verification", func() { + It("compares artifact policies by value and preserves them in JSON and YAML", func() { + a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + Expect(a.Equal(b)).To(BeTrue()) + b.ArtifactVerification.Identity = "another-workflow" + Expect(a.Equal(b)).To(BeFalse()) + b.ArtifactVerification = nil + Expect(a.Equal(b)).To(BeFalse()) + raw, err := json.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + Expect(json.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + raw, err = yaml.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + b = config.Gallery{} + Expect(yaml.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + }) +}) diff --git a/core/config/runtime_settings_startup.go b/core/config/runtime_settings_startup.go index 9808c877b..e5d7e2a46 100644 --- a/core/config/runtime_settings_startup.go +++ b/core/config/runtime_settings_startup.go @@ -17,8 +17,8 @@ import ( // a caching mirror of the files below. The GitHub URI stays as a mirror so an // install still resolves its gallery unchanged whenever the primary is // unreachable - see the fallback chain in core/gallery/gallery_mirrors.go. -const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]` -const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]` +const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` +const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` func mustGalleries(jsonList string) []Gallery { var g []Gallery diff --git a/core/config/runtime_settings_startup_test.go b/core/config/runtime_settings_startup_test.go index 410e457d6..d49c53ac4 100644 --- a/core/config/runtime_settings_startup_test.go +++ b/core/config/runtime_settings_startup_test.go @@ -10,22 +10,22 @@ import ( ) var _ = Describe("default galleries", func() { - It("serves the model gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/models")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"})) }) - It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"})) }) // The mirror is the whole reason this default is safe to ship: if @@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() { Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed()) for _, g := range galleries { Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name) + Expect(g.ArtifactVerification).ToNot(BeNil()) + Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master")) + Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com")) + Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust") } } }) diff --git a/core/gallery/entry_url.go b/core/gallery/entry_url.go index 184ce2dd9..49f4f720d 100644 --- a/core/gallery/entry_url.go +++ b/core/gallery/entry_url.go @@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string { if !looksLikeOCIGallery(candidate) { continue } - dir := ociGalleryCacheDir(basePath, candidate, g.Verification) + dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g)) if dir == "" { continue } diff --git a/core/gallery/gallery.go b/core/gallery/gallery.go index 68d6de9e8..d0c0f43e0 100644 --- a/core/gallery/gallery.go +++ b/core/gallery/gallery.go @@ -646,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]() // would also point relative entry urls at an unpacked tree the new policy has // not produced yet, so they could not be installed. func galleryIndexCacheKey(g config.Gallery) string { - return g.Name + "-" + galleryCacheName(g.URL, g.Verification) + return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g)) } func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) { diff --git a/core/gallery/gallery_mirrors.go b/core/gallery/gallery_mirrors.go index 4058c0a45..083512417 100644 --- a/core/gallery/gallery_mirrors.go +++ b/core/gallery/gallery_mirrors.go @@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { if !looksLikeOCIGallery(g.URL) { return nil } - return g.Verification + return galleryArtifactPolicy(g) } // verifiableCandidates drops the candidates that cannot answer for a signed @@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { // at, and after a refusal it would turn "this artifact is not trusted" into // "use this other, unchecked copy instead". func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string { - if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) { + if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) { return candidates } out := make([]string, 0, len(candidates)) diff --git a/core/gallery/gallery_oci.go b/core/gallery/gallery_oci.go index 19d6d16c0..cbd649229 100644 --- a/core/gallery/gallery_oci.go +++ b/core/gallery/gallery_oci.go @@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) { // later fetch served would hand the user a truncated gallery with no sign that // anything went wrong. func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) { + policy := galleryArtifactPolicy(g) // Checked before the cache: a copy unpacked while strict integrity was // off was never verified, and turning strict integrity on must not keep // serving it for the rest of its TTL. - if g.Verification == nil && requireIntegrity { + if policy == nil && requireIntegrity { return nil, &galleryVerificationError{ strict: true, - err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate), + err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate), } } - cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification) + cacheDir := ociGalleryCacheDir(basePath, candidate, policy) if cacheDir == "" { return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate) } @@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base pullRef := downloader.URI(candidate).OCIReference() - if g.Verification != nil { + if policy != nil { // Resolve first, verify the digest, then pull that same digest. // Nothing has been fetched at this point beyond the manifest, so a // policy failure leaves no content anywhere. @@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base if err != nil { return nil, err } - if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil { + if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil { // Only a decision about the artifact is a refusal. The // verifier also reaches the Sigstore TUF mirror and the // registry, and a timeout or a 5xx there says nothing about @@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base return body, nil } + +func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification { + if g.ArtifactVerification != nil { + return g.ArtifactVerification + } + return g.Verification +} diff --git a/core/gallery/gallery_oci_test.go b/core/gallery/gallery_oci_test.go index 5672da123..4c038b5a2 100644 --- a/core/gallery/gallery_oci_test.go +++ b/core/gallery/gallery_oci_test.go @@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() { }) }) + It("uses the artifact policy without replacing backend image verification", func() { + srv, _, _ := ociRegistry() + url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}}) + backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"} + artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"} + var seen *config.GalleryVerification + stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil }) + g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy} + _, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true) + Expect(source).To(Equal(url)) + Expect(err).ToNot(HaveOccurred()) + Expect(seen).To(Equal(artifactPolicy)) + Expect(g.Verification).To(Equal(backendPolicy)) + }) + It("refuses an unsigned gallery in strict integrity mode", func() { srv, _, blobs := ociRegistry() url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{ diff --git a/docs/content/features/backends.md b/docs/content/features/backends.md index 6085cf536..c0818df85 100644 --- a/docs/content/features/backends.md +++ b/docs/content/features/backends.md @@ -82,6 +82,8 @@ tags: ### Verifying OCI Backends +The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details. + Backend galleries can require keyless Sigstore signatures for every OCI image they provide. Add a `verification` policy to the gallery configuration, then enable strict integrity mode: diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..58aa920f9 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -97,7 +97,7 @@ To use a gallery that needs authentication, such as a private GitHub repository A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted. -Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. +Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. ```json GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}] @@ -151,10 +151,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out ### Signature verification -An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: +An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: ```json -GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] +GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] ``` The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends. @@ -173,9 +173,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}). {{% notice warning %}} -With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. +`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities. + +With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. {{% /notice %}} +### Official gallery publishing + +The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA. + +To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title. + ### Private registries A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use: @@ -207,10 +215,10 @@ GALLERIES=[{"name":"", "url":" Date: Sun, 27 Sep 2026 21:38:15 +0200 Subject: [PATCH 38/49] fix(kokoros): add missing animate3_d stub to Backend trait impl (#12301) * fix(kokoros): add missing animate3_d stub to Backend trait impl #12095 added the Animate3D RPC to backend.proto, but the kokoros service never got a matching method. The tonic-generated Backend trait now requires it, so kokoros fails to build with E0046 whenever the full backend matrix runs. Return Unimplemented, as the other unsupported RPCs do. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] * fix(kokoros): fill new Result fields with defaults backend.proto added a metadata field to Result, so the struct literals in the kokoros service no longer name every field and fail to compile. Spread Default::default() into them, so later additive proto fields do not break the build again. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- backend/rust/kokoros/src/service.rs | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/backend/rust/kokoros/src/service.rs b/backend/rust/kokoros/src/service.rs index aeddbf107..a97cc4bf5 100644 --- a/backend/rust/kokoros/src/service.rs +++ b/backend/rust/kokoros/src/service.rs @@ -132,6 +132,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Kokoros TTS model loaded".into(), + ..Default::default() })) } @@ -180,11 +181,13 @@ impl Backend for KokorosService { return Ok(Response::new(backend::Result { success: false, message: format!("Failed to write WAV: {}", e), + ..Default::default() })); } Ok(Response::new(backend::Result { success: true, message: String::new(), + ..Default::default() })) } Err(e) => { @@ -192,6 +195,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: false, message: format!("TTS error: {}", e), + ..Default::default() })) } } @@ -292,6 +296,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Model freed".into(), + ..Default::default() })) } @@ -348,6 +353,13 @@ impl Backend for KokorosService { Err(Status::unimplemented("Not supported")) } + async fn animate3_d( + &self, + _: Request, + ) -> Result, Status> { + Err(Status::unimplemented("Not supported")) + } + async fn audio_transcription( &self, _: Request, From 78b756627d9d568e077b654f4d675bc4e424885b Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 20:01:43 +0000 Subject: [PATCH 39/49] chore(deps): bump LocalAGI to skip memory writes without a RAG DB An agent with long_term_memory or summary_long_term_memory enabled but no knowledge base crashed with a nil pointer dereference when it saved the conversation (#11975). LocalAGI now logs a warning and skips the write in that case (mudler/LocalAGI#501). Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- go.mod | 24 +---------------------- go.sum | 60 +++------------------------------------------------------- 2 files changed, 4 insertions(+), 80 deletions(-) diff --git a/go.mod b/go.mod index c81fbd82e..5ab9e209b 100644 --- a/go.mod +++ b/go.mod @@ -83,11 +83,8 @@ require ( ) require ( - cyphar.com/go-pathrs v0.2.1 // indirect filippo.io/bigmod v0.1.1-0.20260103110540-f8a47775ebe5 // indirect filippo.io/keygen v0.0.0-20260114151900-8e2790ea4c5b // indirect - github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 // indirect - github.com/AdamKorcz/go-118-fuzz-build v0.0.0-20230306123547-8075edf89bb0 // indirect github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 // indirect github.com/atotto/clipboard v0.1.4 // indirect github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.9 // indirect @@ -110,19 +107,12 @@ require ( github.com/cenkalti/backoff/v5 v5.0.3 // indirect github.com/charmbracelet/bubbles v0.21.0 // indirect github.com/charmbracelet/bubbletea v1.3.10 // indirect - github.com/chasefleming/elem-go v0.30.0 // indirect github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc // indirect github.com/chromedp/chromedp v0.15.1 // indirect github.com/chromedp/sysutil v1.1.0 // indirect - github.com/containerd/containerd/api v1.8.0 // indirect - github.com/containerd/fifo v1.1.0 // indirect - github.com/containerd/ttrpc v1.2.7 // indirect - github.com/containerd/typeurl/v2 v2.2.0 // indirect github.com/cyberphone/json-canonicalization v0.0.0-20241213102144-19d51d7fe467 // indirect - github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 // indirect github.com/digitorus/pkcs7 v0.0.0-20230818184609-3a137a874352 // indirect github.com/digitorus/timestamp v0.0.0-20231217203849-220c5c2851b7 // indirect - github.com/docker/go-events v0.0.0-20190806004212-e31b211e4f1c // indirect github.com/dunglas/httpsfv v1.1.0 // indirect github.com/erikgeiser/coninput v0.0.0-20211004153227-1c3628e74d0f // indirect github.com/filecoin-project/go-clock v0.1.0 // indirect @@ -149,14 +139,10 @@ require ( github.com/gobwas/httphead v0.1.0 // indirect github.com/gobwas/pool v0.2.1 // indirect github.com/gobwas/ws v1.4.0 // indirect - github.com/gofiber/template v1.8.3 // indirect - github.com/gofiber/template/html/v2 v2.1.3 // indirect - github.com/gofiber/utils v1.1.0 // indirect github.com/google/certificate-transparency-go v1.3.2 // indirect github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0 // indirect github.com/in-toto/attestation v1.1.2 // indirect github.com/in-toto/in-toto-golang v0.9.0 // indirect - github.com/inconshreveable/mousetrap v1.1.0 // indirect github.com/invopop/jsonschema v0.13.0 // indirect github.com/jinzhu/inflection v1.0.0 // indirect github.com/jinzhu/now v1.1.5 // indirect @@ -164,17 +150,12 @@ require ( github.com/klippa-app/go-pdfium v1.19.2 // indirect github.com/mattn/go-localereader v0.0.1 // indirect github.com/mattn/go-sqlite3 v1.14.28 // indirect - github.com/moby/locker v1.0.1 // indirect github.com/moby/moby/api v1.54.2 // indirect github.com/moby/moby/client v0.4.1 // indirect - github.com/moby/sys/mountinfo v0.7.2 // indirect - github.com/moby/sys/signal v0.7.0 // indirect github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect github.com/muesli/cancelreader v0.2.2 // indirect github.com/nats-io/nuid v1.0.1 // indirect github.com/oklog/ulid v1.3.1 // indirect - github.com/opencontainers/runtime-spec v1.2.0 // indirect - github.com/opencontainers/selinux v1.13.1 // indirect github.com/secure-systems-lab/go-securesystemslib v0.9.1 // indirect github.com/shibumi/go-pathspec v1.3.0 // indirect github.com/sigstore/protobuf-specs v0.5.1 // indirect @@ -182,8 +163,6 @@ require ( github.com/sigstore/rekor-tiles/v2 v2.0.1 // indirect github.com/sigstore/sigstore v1.10.0 // indirect github.com/sigstore/timestamp-authority/v2 v2.0.3 // indirect - github.com/spf13/cobra v1.10.2 // indirect - github.com/spf13/pflag v1.0.10 // indirect github.com/standard-webhooks/standard-webhooks/libraries v0.0.0-20260508151727-1282bb917829 // indirect github.com/stretchr/testify v1.11.1 // indirect github.com/sv-tools/openapi v0.2.1 // indirect @@ -195,7 +174,6 @@ require ( github.com/transparency-dev/merkle v0.0.2 // indirect github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect go.mongodb.org/mongo-driver v1.17.6 // indirect - google.golang.org/genproto v0.0.0-20250922171735-9219d122eba9 // indirect google.golang.org/genproto/googleapis/api v0.0.0-20260209200024-4cfbd4190f57 // indirect sigs.k8s.io/yaml v1.6.0 // indirect ) @@ -259,7 +237,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e + github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect diff --git a/go.sum b/go.sum index 863118db1..de770503e 100644 --- a/go.sum +++ b/go.sum @@ -50,8 +50,6 @@ cloud.google.com/go/storage v1.5.0/go.mod h1:tpKbwo567HUNpVclU5sGELwQWBDZ8gh0Zeo cloud.google.com/go/storage v1.6.0/go.mod h1:N7U0C8pVQ/+NIKOBQyamJIeKQKkZ+mxpohlUTyfDhBk= cloud.google.com/go/storage v1.8.0/go.mod h1:Wv1Oy7z6Yz3DshWRJFhqM/UCfaWIRTdp0RXyy7KQOVs= cloud.google.com/go/storage v1.10.0/go.mod h1:FLPqc6j+Ki4BU591ie1oL6qBQGu2Bl/tZ9ullr3+Kg0= -cyphar.com/go-pathrs v0.2.1 h1:9nx1vOgwVvX1mNBWDu93+vaceedpbsDqo+XuBGL40b8= -cyphar.com/go-pathrs v0.2.1/go.mod h1:y8f1EMG7r+hCuFf/rXsKqMJrJAUoADZGNh5/vZPKcGc= dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8= dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA= dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU= @@ -67,8 +65,6 @@ fyne.io/systray v1.12.0 h1:CA1Kk0e2zwFlxtc02L3QFSiIbxJ/P0n582YrZHT7aTM= fyne.io/systray v1.12.0/go.mod h1:RVwqP9nYMo7h5zViCBHri2FgjXF7H2cub7MAq4NSoLs= github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk= github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6/go.mod h1:8o94RPi1/7XTJvwPpRSzSUedZrtlirdB3r9Z20bi2f8= -github.com/AdamKorcz/go-118-fuzz-build v0.0.0-20230306123547-8075edf89bb0 h1:59MxjQVfjXsBpLy+dbd2/ELV5ofnUkUZBvWSC85sheA= -github.com/AdamKorcz/go-118-fuzz-build v0.0.0-20230306123547-8075edf89bb0/go.mod h1:OahwfttHWG6eJ0clwcfBAHoDI6X/LV/15hx/wlMZSrU= github.com/AdamKorcz/go-fuzz-headers-1 v0.0.0-20230919221257-8b5d3ce2d11d h1:zjqpY4C7H15HjRPEenkS4SAn3Jy2eRRjkjZbGR30TOg= github.com/AdamKorcz/go-fuzz-headers-1 v0.0.0-20230919221257-8b5d3ce2d11d/go.mod h1:XNqJ7hv2kY++g8XEHREpi+JqZo3+0l+CH2egBVN4yqM= github.com/Azure/azure-sdk-for-go/sdk/azcore v1.20.0 h1:JXg2dwJUmPB9JmtVmdEB16APJ7jurfbY5jnfXpJoRMc= @@ -281,8 +277,6 @@ github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf h1:rLG0Y github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf/go.mod h1:B3UgsnsBZS/eX42BlaNiJkD1pPOUa+oF1IYC6Yd2CEU= github.com/charmbracelet/x/term v0.2.1 h1:AQeHeLZ1OqSXhrAWpYUtZyX1T3zVxfpZuEQMIQaGIAQ= github.com/charmbracelet/x/term v0.2.1/go.mod h1:oQ4enTYFV7QN4m0i9mzHrViD7TQKvNEEkHUMCmsxdUg= -github.com/chasefleming/elem-go v0.30.0 h1:BlhV1ekv1RbFiM8XZUQeln1Ikb4D+bu2eDO4agREvok= -github.com/chasefleming/elem-go v0.30.0/go.mod h1:hz73qILBIKnTgOujnSMtEj20/epI+f6vg71RUilJAA4= github.com/chengxilo/virtualterm v1.0.4 h1:Z6IpERbRVlfB8WkOmtbHiDbBANU7cimRIof7mk9/PwM= github.com/chengxilo/virtualterm v1.0.4/go.mod h1:DyxxBZz/x1iqJjFxTFcr6/x+jSpqN0iwWCOK1q10rlY= github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc h1:wkN/LMi5vc60pBRWx6qpbk/aEvq3/ZVNpnMvsw8PVVU= @@ -304,31 +298,18 @@ github.com/codahale/rfc6979 v0.0.0-20141003034818-6a90f24967eb h1:EDmT6Q9Zs+SbUo github.com/codahale/rfc6979 v0.0.0-20141003034818-6a90f24967eb/go.mod h1:ZjrT6AXHbDs86ZSdt/osfBi5qfexBrKUdONk989Wnk4= github.com/containerd/cgroups v1.1.0 h1:v8rEWFl6EoqHB+swVNjVoCJE8o3jX7e8nqBGPLaDFBM= github.com/containerd/cgroups v1.1.0/go.mod h1:6ppBcbh/NOOUU+dMKrykgaBnK9lCIBxHqJDGwsa1mIw= -github.com/containerd/containerd v1.7.31 h1:jn3IMuTV4Bb1Uwb0MFPW2ASJAD3W1lh6QqqZHIZwDh4= -github.com/containerd/containerd v1.7.31/go.mod h1:jdwD6s/BhV4XVJGrvtziNPVA+83n66TwptVaPKprq4E= -github.com/containerd/containerd v1.7.32 h1:S54xuVcPxeLaYgaRABtpJ2VyVUVsy0IGf7qHBs+sbY8= -github.com/containerd/containerd v1.7.32/go.mod h1:jdwD6s/BhV4XVJGrvtziNPVA+83n66TwptVaPKprq4E= github.com/containerd/containerd v1.7.33 h1:iAkYGC/ifR/V+0eR4iXWHNGYUF0DF2PmGV5iz4Irj5M= github.com/containerd/containerd v1.7.33/go.mod h1:gSbSCVjPCdkfJCjyrzz7aRC+xFlqVbatNpfHfVCYGUM= -github.com/containerd/containerd/api v1.8.0 h1:hVTNJKR8fMc/2Tiw60ZRijntNMd1U+JVMyTRdsD2bS0= -github.com/containerd/containerd/api v1.8.0/go.mod h1:dFv4lt6S20wTu/hMcP4350RL87qPWLVa/OHOwmmdnYc= github.com/containerd/continuity v0.4.4 h1:/fNVfTJ7wIl/YPMHjf+5H32uFhl63JucB34PlCpMKII= github.com/containerd/continuity v0.4.4/go.mod h1:/lNJvtJKUQStBzpVQ1+rasXO1LAWtUQssk28EZvJ3nE= github.com/containerd/errdefs v1.0.0 h1:tg5yIfIlQIrxYtu9ajqY42W3lpS19XqdxRQeEwYG8PI= github.com/containerd/errdefs v1.0.0/go.mod h1:+YBYIdtsnF4Iw6nWZhJcqGSg/dwvV7tyJ/kCkyJ2k+M= github.com/containerd/errdefs/pkg v0.3.0 h1:9IKJ06FvyNlexW690DXuQNx2KA2cUJXx151Xdx3ZPPE= github.com/containerd/errdefs/pkg v0.3.0/go.mod h1:NJw6s9HwNuRhnjJhM7pylWwMyAkmCQvQ4GpJHEqRLVk= -github.com/containerd/fifo v1.1.0 h1:4I2mbh5stb1u6ycIABlBw9zgtlK8viPI9QkQNRQEEmY= -github.com/containerd/fifo v1.1.0/go.mod h1:bmC4NWMbXlt2EZ0Hc7Fx7QzTFxgPID13eH0Qu+MAb2o= github.com/containerd/log v0.1.0 h1:TCJt7ioM2cr/tfR8GPbGf9/VRAX8D2B4PjzCpfX540I= github.com/containerd/log v0.1.0/go.mod h1:VRRf09a7mHDIRezVKTRCrOq78v577GXq3bSa3EhrzVo= github.com/containerd/platforms v0.2.1 h1:zvwtM3rz2YHPQsF2CHYM8+KtB5dvhISiXh5ZpSBQv6A= github.com/containerd/platforms v0.2.1/go.mod h1:XHCb+2/hzowdiut9rkudds9bE5yJ7npe7dG/wG+uFPw= -github.com/containerd/ttrpc v1.2.7 h1:qIrroQvuOL9HQ1X6KHe2ohc7p+HP/0VE6XPU7elJRqQ= -github.com/containerd/ttrpc v1.2.7/go.mod h1:YCXHsb32f+Sq5/72xHubdiJRQY9inL4a4ZQrAbN1q9o= -github.com/containerd/typeurl v1.0.2 h1:Chlt8zIieDbzQFzXzAeBEF92KhExuE4p9p92/QmY7aY= -github.com/containerd/typeurl/v2 v2.2.0 h1:6NBDbQzr7I5LHgp34xAXYF5DOTQDn05X58lsPEmzLso= -github.com/containerd/typeurl/v2 v2.2.0/go.mod h1:8XOOxnyatxSWuG8OfsZXVnAF4iZfedjS/8UHSPJnX4g= github.com/coreos/go-oidc/v3 v3.18.0 h1:V9orjXynvu5wiC9SemFTWnG4F45v403aIcjWo0d41+A= github.com/coreos/go-oidc/v3 v3.18.0/go.mod h1:DYCf24+ncYi+XkIH97GY1+dqoRlbaSI26KVTCI9SrY4= github.com/coreos/go-semver v0.3.0/go.mod h1:nnelYz7RCh+5ahJtPPxZlU+153eP4D4r3EedlOD2RNk= @@ -338,7 +319,6 @@ github.com/cpuguy83/dockercfg v0.3.2 h1:DlJTyZGBDlXqUZ2Dk2Q3xHs/FtnooJJVaad2S9GK github.com/cpuguy83/dockercfg v0.3.2/go.mod h1:sugsbF4//dDlL/i+S+rtpIWp+5h0BHJHfjj5/jFyUJc= github.com/cpuguy83/go-md2man/v2 v2.0.0-20190314233015-f79a8a8ca69d/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU= github.com/cpuguy83/go-md2man/v2 v2.0.0/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU= -github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g= github.com/creachadair/mds v0.21.3 h1:RRgEAPIb52cU0q7UxGyN+13QlCVTZIL4slRr0cYYQfA= github.com/creachadair/mds v0.21.3/go.mod h1:1ltMWZd9yXhaHEoZwBialMaviWVUpRPvMwVP7saFAzM= github.com/creachadair/otp v0.5.0 h1:q3Th7CXm2zlmCdBjw5tEPFOj4oWJMnVL5HXlq0sNKS0= @@ -351,8 +331,6 @@ github.com/cyphar/filepath-securejoin v0.6.1 h1:5CeZ1jPXEiYt3+Z6zqprSAgSWiggmpVy github.com/cyphar/filepath-securejoin v0.6.1/go.mod h1:A8hd4EnAeyujCJRrICiOWqjS1AX0a9kM5XL+NwKoYSc= github.com/danieljoos/wincred v1.2.2 h1:774zMFJrqaeYCK2W57BgAem/MLi6mtSE47MB6BOJ0i0= github.com/danieljoos/wincred v1.2.2/go.mod h1:w7w4Utbrz8lqeMbDAK0lkNJUv5sAOkFi7nd/ogr0Uh8= -github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 h1:flLYmnQFZNo04x2NPehMbf30m7Pli57xwZ0NFqR/hb0= -github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2/go.mod h1:NtWqRzAp/1tw+twkW8uuBenEVVYndEAZACWU3F3xdoQ= github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM= @@ -384,8 +362,6 @@ github.com/docker/docker-credential-helpers v0.9.3 h1:gAm/VtF9wgqJMoxzT3Gj5p4AqI github.com/docker/docker-credential-helpers v0.9.3/go.mod h1:x+4Gbw9aGmChi3qTLZj8Dfn0TD20M/fuWy0E5+WDeCo= github.com/docker/go-connections v0.7.0 h1:6SsRfJddP22WMrCkj19x9WKjEDTB+ahsdiGYf0mN39c= github.com/docker/go-connections v0.7.0/go.mod h1:no1qkHdjq7kLMGUXYAduOhYPSJxxvgWBh7ogVvptn3Q= -github.com/docker/go-events v0.0.0-20190806004212-e31b211e4f1c h1:+pKlWGMw7gf6bQ+oDZB4KHQFypsfjYlq/C4rfL7D3g8= -github.com/docker/go-events v0.0.0-20190806004212-e31b211e4f1c/go.mod h1:Uw6UezgYA44ePAFQYUehOuCzmy5zmg/+nl2ZfMWGkpA= github.com/docker/go-units v0.5.0 h1:69rxXcBk27SvSaaxTtLh/8llcHD8vYHT7WSdRZ/jvr4= github.com/docker/go-units v0.5.0/go.mod h1:fgPhTUdO+D/Jk86RDLlptpiXQzgHJF7gydDDbaIK4Dk= github.com/dsnet/compress v0.0.2-0.20210315054119-f66993602bf5 h1:iFaUwBSo5Svw6L7HYpRu/0lE3e0BaElwnNO1qkNQxBY= @@ -575,12 +551,6 @@ github.com/godbus/dbus/v5 v5.1.0 h1:4KLkAxT3aOY8Li4FRJe/KvhoNFFxo0m6fNuFUO8QJUk= github.com/godbus/dbus/v5 v5.1.0/go.mod h1:xhWf0FNVPg57R7Z0UbKHbJfkEywrmjJnf7w5xrFpKfA= github.com/gofiber/fiber/v2 v2.52.13 h1:TOKP64iqC9b5P49VrBW5tHhUOvDyrtJ0xePEfzJbCbk= github.com/gofiber/fiber/v2 v2.52.13/go.mod h1:YEcBbO/FB+5M1IZNBP9FO3J9281zgPAreiI1oqg8nDw= -github.com/gofiber/template v1.8.3 h1:hzHdvMwMo/T2kouz2pPCA0zGiLCeMnoGsQZBTSYgZxc= -github.com/gofiber/template v1.8.3/go.mod h1:bs/2n0pSNPOkRa5VJ8zTIvedcI/lEYxzV3+YPXdBvq8= -github.com/gofiber/template/html/v2 v2.1.3 h1:n1LYBtmr9C0V/k/3qBblXyMxV5B0o/gpb6dFLp8ea+o= -github.com/gofiber/template/html/v2 v2.1.3/go.mod h1:U5Fxgc5KpyujU9OqKzy6Kn6Qup6Tm7zdsISR+VpnHRE= -github.com/gofiber/utils v1.1.0 h1:vdEBpn7AzIUJRhe+CiTOJdUcTg4Q9RK+pEa0KPbLdrM= -github.com/gofiber/utils v1.1.0/go.mod h1:poZpsnhBykfnY1Mc0KeEa6mSHrS3dV0+oBWyeQmb2e0= github.com/gofrs/flock v0.13.0 h1:95JolYOvGMqeH31+FC7D2+uULf6mG61mEZ/A8dRYMzw= github.com/gofrs/flock v0.13.0/go.mod h1:jxeyy9R1auM5S6JYDBhDt+E2TCo7DkratH4Pgi8P+Z0= github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q= @@ -993,8 +963,6 @@ github.com/moby/docker-image-spec v1.3.1 h1:jMKff3w6PgbfSa69GfNg+zN/XLhfXJGnEx3N github.com/moby/docker-image-spec v1.3.1/go.mod h1:eKmb5VW8vQEh/BAr2yvVNvuiJuY6UIocYsFu/DxxRpo= github.com/moby/go-archive v0.2.0 h1:zg5QDUM2mi0JIM9fdQZWC7U8+2ZfixfTYoHL7rWUcP8= github.com/moby/go-archive v0.2.0/go.mod h1:mNeivT14o8xU+5q1YnNrkQVpK+dnNe/K6fHqnTg4qPU= -github.com/moby/locker v1.0.1 h1:fOXqR41zeveg4fFODix+1Ch4mj/gT0NE1XJbp/epuBg= -github.com/moby/locker v1.0.1/go.mod h1:S7SDdo5zpBK84bzzVlKr2V0hz+7x9hWbYC/kq7oQppc= github.com/moby/moby/api v1.54.2 h1:wiat9QAhnDQjA7wk1kh/TqHz2I1uUA7M7t9SAl/JNXg= github.com/moby/moby/api v1.54.2/go.mod h1:+RQ6wluLwtYaTd1WnPLykIDPekkuyD/ROWQClE83pzs= github.com/moby/moby/client v0.4.1 h1:DMQgisVoMkmMs7fp3ROSdiBnoAu8+vo3GggFl06M/wY= @@ -1005,8 +973,6 @@ github.com/moby/sys/mountinfo v0.7.2 h1:1shs6aH5s4o5H2zQLn796ADW1wMrIwHsyJ2v9Kou github.com/moby/sys/mountinfo v0.7.2/go.mod h1:1YOa8w8Ih7uW0wALDUgT1dTTSBrZ+HiBLGws92L2RU4= github.com/moby/sys/sequential v0.6.0 h1:qrx7XFUd/5DxtqcoH1h438hF5TmOvzC/lspjy7zgvCU= github.com/moby/sys/sequential v0.6.0/go.mod h1:uyv8EUTrca5PnDsdMGXhZe6CCe8U/UiTWd+lL+7b/Ko= -github.com/moby/sys/signal v0.7.0 h1:25RW3d5TnQEoKvRbEKUGay6DCQ46IxAVTT9CUMgmsSI= -github.com/moby/sys/signal v0.7.0/go.mod h1:GQ6ObYZfqacOwTtlXvcmh9A26dVRul/hbOZn88Kg8Tg= github.com/moby/sys/user v0.4.0 h1:jhcMKit7SA80hivmFJcbB1vqmw//wU61Zdui2eQXuMs= github.com/moby/sys/user v0.4.0/go.mod h1:bG+tYYYJgaMtRKgEmuueC0hJEAZWwtIbZTB+85uoHjs= github.com/moby/sys/userns v0.1.0 h1:tVLXkFOxVu9A64/yh59slHVv9ahO9UIev4JZusOLG/g= @@ -1028,24 +994,16 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI= github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8= github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM= github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw= -github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxFoYsAObvAuOqXBakRPMD0PWxWG5EE= -github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c= -github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc= -github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o= -github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q= -github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= +github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c h1:pyfcuJuXcGNLIMk8m25gg0T0eYe4mRN1ioI2iuUv8Fs= +github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= github.com/mudler/edgevpn v0.34.0/go.mod h1:yki7uMi5LR9gSMrw8PdPieuxsrk8BLV2Ui7VBEmbbIA= github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc h1:RxwneJl1VgvikiX28EkpdAyL4yQVnJMrbquKospjHyA= github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc/go.mod h1:O7SwdSWMilAWhBZMK9N9Y/oBDyMMzshE3ju8Xkexwig= -github.com/mudler/go-processmanager v0.1.2-0.20260720195933-3d64f5c974fc h1:NEFmd7+JoImN5dZI81/vcBRjtMg+GfEa7nEjQ029hd0= -github.com/mudler/go-processmanager v0.1.2-0.20260720195933-3d64f5c974fc/go.mod h1:h6kmHUZeafr+k5hRYpGLMzJFH4hItHffgpRo2QIkP+o= github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6 h1:/nFm1Ttf8g1BnWtEth986JR34pCh9rzae5A2vKBZosc= github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6/go.mod h1:h6kmHUZeafr+k5hRYpGLMzJFH4hItHffgpRo2QIkP+o= -github.com/mudler/localrecall v0.6.3 h1:uXOrP9JmetzxgVKzSrawviyBHZfAcvPBBIrvVUdZjDA= -github.com/mudler/localrecall v0.6.3/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU= github.com/mudler/localrecall v0.6.5 h1:Q0atTJFFAyumKZG5dbGSrvQ+wsuA88hywIOfHdxpBEU= github.com/mudler/localrecall v0.6.5/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU= github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 h1:Ry8RiWy8fZ6Ff4E7dPmjRsBrnHOnPeOOj2LhCgyjQu0= @@ -1128,10 +1086,6 @@ github.com/opencontainers/go-digest v1.0.0 h1:apOUWs51W5PlhuyGyz9FCeeBIOUDA/6nW8 github.com/opencontainers/go-digest v1.0.0/go.mod h1:0JzlMkj0TRzQZfJkVvzbP0HBR3IKzErnv2BNG4W4MAM= github.com/opencontainers/image-spec v1.1.1 h1:y0fUlFfIZhPF1W537XOLg0/fcx6zcHCJwooC2xJA040= github.com/opencontainers/image-spec v1.1.1/go.mod h1:qpqAh3Dmcf36wStyyWU+kCeDgrGnAve2nCC8+7h8Q0M= -github.com/opencontainers/runtime-spec v1.2.0 h1:z97+pHb3uELt/yiAWD691HNHQIF07bE7dzrbT927iTk= -github.com/opencontainers/runtime-spec v1.2.0/go.mod h1:jwyrGlmzljRJv/Fgzds9SsS/C5hL+LL3ko9hs6T5lQ0= -github.com/opencontainers/selinux v1.13.1 h1:A8nNeceYngH9Ow++M+VVEwJVpdFmrlxsN22F+ISDCJE= -github.com/opencontainers/selinux v1.13.1/go.mod h1:S10WXZ/osk2kWOYKy1x2f/eXF5ZHJoUs8UU/2caNRbg= github.com/opentracing/opentracing-go v1.2.0 h1:uEJPy/1a5RIPAJ0Ov+OIO8OxWu77jEv+1B0VhjKrZUs= github.com/opentracing/opentracing-go v1.2.0/go.mod h1:GxEUsuufX4nBwe+T+Wl9TAgYrxe9dPLANfrWvHYVTgc= github.com/orisano/pixelmatch v0.0.0-20220722002657-fb0b55479cde h1:x0TT0RDC7UhAVbbWWBzr41ElhJx5tXPWkIHA2HWPRuw= @@ -1245,7 +1199,6 @@ github.com/rs/zerolog v1.31.0/go.mod h1:/7mN4D5sKwJLZQ2b/znpjC3/GQWY/xaDXUM0kKWR github.com/russross/blackfriday v1.6.0 h1:KqfZb0pUVN2lYqZUYRddxF4OR8ZMURnJIG5Y3VRLtww= github.com/russross/blackfriday v1.6.0/go.mod h1:ti0ldHuxg49ri4ksnFxlkCfN+hvslNlmVHqNRXXJNAY= github.com/russross/blackfriday/v2 v2.0.1/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= -github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= github.com/ruudk/golang-pdf417 v0.0.0-20181029194003-1af4ab5afa58/go.mod h1:6lfFZQK844Gfx8o5WFuvpxWRwnSoipWe/p622j1v06w= github.com/ryanuber/columnize v0.0.0-20160712163229-9b3edd62028f/go.mod h1:sm1tb6uqfes/u+d4ooFouqFdy9/2g9QGwK3SQygK0Ts= github.com/ryanuber/go-glob v1.0.0 h1:iQh3xXAumdQ+4Ufa5b25cRpC5TYKlno6hsv6Cb3pkBk= @@ -1279,13 +1232,10 @@ github.com/shirou/gopsutil/v3 v3.24.5 h1:i0t8kL+kQTvpAYToeuiVk3TgDeKOFioZO3Ztz/i github.com/shirou/gopsutil/v3 v3.24.5/go.mod h1:bsoOS1aStSs9ErQ1WWfxllSeS1K5D+U30r2NfcubMVk= github.com/shirou/gopsutil/v4 v4.26.3 h1:2ESdQt90yU3oXF/CdOlRCJxrP+Am1aBYubTMTfxJ1qc= github.com/shirou/gopsutil/v4 v4.26.3/go.mod h1:LZ6ewCSkBqUpvSOf+LsTGnRinC6iaNUNMGBtDkJBaLQ= -github.com/shoenig/go-m1cpu v0.1.6 h1:nxdKQNcEB6vzgA2E2bvzKIYRuNj7XNJ4S/aRSwKzFtM= -github.com/shoenig/go-m1cpu v0.1.6/go.mod h1:1JJMcUBvfNwpq05QDQVAnx3gUHr9IYF7GNg9SUEw2VQ= github.com/shoenig/go-m1cpu v0.2.2 h1:4nc55oVv7nygGnfI9bhLCLzUEs4794y0Bkqx4q2zy7Y= github.com/shoenig/go-m1cpu v0.2.2/go.mod h1:KkDOw6m3ZJQAPHbrzkZki4hnx+pDRR1Lo+ldA56wD5w= -github.com/shoenig/test v0.6.4 h1:kVTaSd7WLz5WZ2IaoM0RSzRsUD+m8wRR+5qvntpn4LU= -github.com/shoenig/test v0.6.4/go.mod h1:byHiCGXqrVaflBLAMq/srcZIHynQPQgeyvkvXnjqq0k= github.com/shoenig/test v1.7.0 h1:eWcHtTXa6QLnBvm0jgEabMRN/uJ4DMV3M8xUGgRkZmk= +github.com/shoenig/test v1.7.0/go.mod h1:UxJ6u/x2v/TNs/LoLxBNJRV9DiwBBKYxXSyczsBHFoI= github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME= github.com/shurcooL/go v0.0.0-20200502201357-93f07166e636/go.mod h1:TDJrrUr11Vxrven61rcy3hJMUqaf/CLWYhHNPmT14Lk= @@ -1341,7 +1291,6 @@ github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU= github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4= github.com/spf13/jwalterweatherman v1.1.0/go.mod h1:aNWZUN0dPAAO/Ljvb5BEdw96iTZ0EXowPYD95IqWIGo= github.com/spf13/pflag v1.0.5/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= -github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= github.com/spf13/viper v1.8.1/go.mod h1:o0Pch8wJ9BVSWGQMbra6iw0oQ5oktSIBaujf1rJH9Ns= @@ -1666,8 +1615,6 @@ golang.org/x/net v0.15.0/go.mod h1:idbUs1IY1+zTqbi8yxTbhexhEEk5ur9LInksu6HrEpk= golang.org/x/net v0.21.0/go.mod h1:bIjVDfnllIU7BJ2DNgfnXvpSvtn8VRwhlsaeUTyUS44= golang.org/x/net v0.25.0/go.mod h1:JkAGAh7GEvH74S6FOH42FLoXpXbE/aqXSrIQjXgsiwM= golang.org/x/net v0.33.0/go.mod h1:HXLR5J+9DxmrqMwG9qjGCxZ+zKXxBru04zlTvWlWuN4= -golang.org/x/net v0.54.0 h1:2zJIZAxAHV/OHCDTCOHAYehQzLfSXuf/5SoL/Dv6w/w= -golang.org/x/net v0.54.0/go.mod h1:Sj4oj8jK6XmHpBZU/zWHw3BV3abl4Kvi+Ut7cQcY+cQ= golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8= golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww= golang.org/x/oauth2 v0.0.0-20180821212333-d2e6202438be/go.mod h1:N/0e6XlmueqKjAGxoOufVs8QHGRruUQn6yWY3a++T0U= @@ -1756,7 +1703,6 @@ golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBc golang.org/x/sys v0.0.0-20210616094352-59db8d763f22/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20210630005230-0f9fa26af87c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20210809222454-d867a43fc93e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= -golang.org/x/sys v0.0.0-20211025201205-69cdffdb9359/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20220520151302-bc2c85ada10a/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20220715151400-c0bba94af5f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= From 250d90aa2ed4262849433cb22cddea33334841e8 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 20:13:00 +0000 Subject: [PATCH 40/49] chore(deps): bump LocalAGI to f2a2af4 Pick up the LocalAGI fixes merged since 985e5f5: - MCP: a failed ping no longer closes borrowed sessions (#500) - agent: nil user message guard in knowledgeBaseLookup (#485), 10m fallback on an unparsable periodic_runs (#497), self-correction on unknown tool calls (#481) - core: idempotent JobResult.Finish (#491) - sse: ordered broadcast and race-free message history (#496) - mautrix 0.25.2 (#328), which moves several indirect deps Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- go.mod | 12 ++++++------ go.sum | 24 +++++++++++------------- 2 files changed, 17 insertions(+), 19 deletions(-) diff --git a/go.mod b/go.mod index 5ab9e209b..4c76ab5ee 100644 --- a/go.mod +++ b/go.mod @@ -84,6 +84,7 @@ require ( require ( filippo.io/bigmod v0.1.1-0.20260103110540-f8a47775ebe5 // indirect + filippo.io/edwards25519 v1.1.0 // indirect filippo.io/keygen v0.0.0-20260114151900-8e2790ea4c5b // indirect github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 // indirect github.com/atotto/clipboard v0.1.4 // indirect @@ -149,7 +150,7 @@ require ( github.com/jolestar/go-commons-pool/v2 v2.1.2 // indirect github.com/klippa-app/go-pdfium v1.19.2 // indirect github.com/mattn/go-localereader v0.0.1 // indirect - github.com/mattn/go-sqlite3 v1.14.28 // indirect + github.com/mattn/go-sqlite3 v1.14.32 // indirect github.com/moby/moby/api v1.54.2 // indirect github.com/moby/moby/client v0.4.1 // indirect github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect @@ -237,7 +238,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c + github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect @@ -245,7 +246,7 @@ require ( github.com/philippgille/chromem-go v0.7.0 // indirect github.com/pion/transport/v4 v4.0.1 // indirect github.com/pjbgf/sha1cd v0.6.0 // indirect - github.com/rs/zerolog v1.31.0 // indirect + github.com/rs/zerolog v1.34.0 // indirect github.com/saintfish/chardet v0.0.0-20230101081208-5e3ef4b5456d // indirect github.com/segmentio/asm v1.1.3 // indirect github.com/segmentio/encoding v0.5.4 // indirect @@ -265,14 +266,13 @@ require ( github.com/valyala/fasttemplate v1.2.2 // indirect github.com/xanzy/ssh-agent v0.3.3 // indirect go.etcd.io/bbolt v1.4.3 // indirect - go.mau.fi/util v0.3.0 // indirect + go.mau.fi/util v0.9.2 // indirect go.starlark.net v0.0.0-20250417143717-f57e51f710eb // indirect google.golang.org/appengine v1.6.8 // indirect gopkg.in/warnings.v0 v0.1.2 // indirect gopkg.in/yaml.v2 v2.4.0 // indirect jaytaylor.com/html2text v0.0.0-20230321000545-74c2419ad056 // indirect - maunium.net/go/maulogger/v2 v2.4.1 // indirect - maunium.net/go/mautrix v0.17.0 // indirect + maunium.net/go/mautrix v0.25.2 // indirect mvdan.cc/xurls/v2 v2.6.0 // indirect ) diff --git a/go.sum b/go.sum index de770503e..4b5fbbf02 100644 --- a/go.sum +++ b/go.sum @@ -917,8 +917,8 @@ github.com/mattn/go-runewidth v0.0.9/go.mod h1:H031xJmbD/WCDINGzjvQ9THkh0rPKHF+m github.com/mattn/go-runewidth v0.0.12/go.mod h1:RAqKPSqVFrSLVXbA8x7dzmKdmGzieGRCM46jaSJTDAk= github.com/mattn/go-runewidth v0.0.17 h1:78v8ZlW0bP43XfmAfPsdXcoNCelfMHsDmd/pkENfrjQ= github.com/mattn/go-runewidth v0.0.17/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w= -github.com/mattn/go-sqlite3 v1.14.28 h1:ThEiQrnbtumT+QMknw63Befp/ce/nUPgBPMlRFEum7A= -github.com/mattn/go-sqlite3 v1.14.28/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y= +github.com/mattn/go-sqlite3 v1.14.32 h1:JD12Ag3oLy1zQA+BNn74xRgaBbdhbNIDYvQUEuuErjs= +github.com/mattn/go-sqlite3 v1.14.32/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y= github.com/mdelapenya/tlscert v0.2.0 h1:7H81W6Z/4weDvZBNOfQte5GpIMo0lGYEeWbkGp5LJHI= github.com/mdelapenya/tlscert v0.2.0/go.mod h1:O4njj3ELLnJjGdkN7M/vIVCpZ+Cf0L6muqOG4tLSl8o= github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3vE= @@ -994,8 +994,8 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI= github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8= github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM= github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw= -github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c h1:pyfcuJuXcGNLIMk8m25gg0T0eYe4mRN1ioI2iuUv8Fs= -github.com/mudler/LocalAGI v0.0.0-20260927195918-985e5f5ba84c/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= +github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c h1:6dvy3mqvRge8Ddji4er30fhoAasQTBJSYwGs1xVVSr8= +github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= @@ -1193,9 +1193,9 @@ github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6L github.com/rogpeppe/go-internal v1.3.0/go.mod h1:M8bDsm7K2OlrFYOpmOWEs/qY81heoFRclV5y23lUDJ4= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= -github.com/rs/xid v1.5.0/go.mod h1:trrq9SKmegXys3aeAKXMUTdJsYXVwGY3RLcfgqegfbg= -github.com/rs/zerolog v1.31.0 h1:FcTR3NnLWW+NnTwwhFWiJSZr4ECLpqCm6QsEnyvbV4A= -github.com/rs/zerolog v1.31.0/go.mod h1:/7mN4D5sKwJLZQ2b/znpjC3/GQWY/xaDXUM0kKWRHss= +github.com/rs/xid v1.6.0/go.mod h1:7XoLgs4eV+QndskICGsho+ADou8ySMSjJKDIan90Nz0= +github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY= +github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ= github.com/russross/blackfriday v1.6.0 h1:KqfZb0pUVN2lYqZUYRddxF4OR8ZMURnJIG5Y3VRLtww= github.com/russross/blackfriday v1.6.0/go.mod h1:ti0ldHuxg49ri4ksnFxlkCfN+hvslNlmVHqNRXXJNAY= github.com/russross/blackfriday/v2 v2.0.1/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= @@ -1440,8 +1440,8 @@ go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E= go.etcd.io/etcd/api/v3 v3.5.0/go.mod h1:cbVKeC6lCfl7j/8jBhAK6aIYO9XOjdptoxU/nLQcPvs= go.etcd.io/etcd/client/pkg/v3 v3.5.0/go.mod h1:IJHfcCEKxYu1Os13ZdwCwIUTUVGYTSAM3YSwc9/Ac1g= go.etcd.io/etcd/client/v2 v2.305.0/go.mod h1:h9puh54ZTgAKtEbut2oe9P4L/oqKCVB6xsXlzd7alYQ= -go.mau.fi/util v0.3.0 h1:Lt3lbRXP6ZBqTINK0EieRWor3zEwwwrDT14Z5N8RUCs= -go.mau.fi/util v0.3.0/go.mod h1:9dGsBCCbZJstx16YgnVMVi3O2bOizELoKpugLD4FoGs= +go.mau.fi/util v0.9.2 h1:+S4Z03iCsGqU2WY8X2gySFsFjaLlUHFRDVCYvVwynKM= +go.mau.fi/util v0.9.2/go.mod h1:055elBBCJSdhRsmub7ci9hXZPgGr1U6dYg44cSgRgoU= go.mongodb.org/mongo-driver v1.17.6 h1:87JUG1wZfWsr6rIz3ZmpH90rL5tea7O3IHuSwHUpsss= go.mongodb.org/mongo-driver v1.17.6/go.mod h1:Hy04i7O2kC4RS06ZrhPRqj/u4DTYkFDAAccj+rVKqgQ= go.opencensus.io v0.21.0/go.mod h1:mSImk1erAIZhrmZN+AvHh14ztQfjbGwt4TtuofqLduU= @@ -1996,10 +1996,8 @@ k8s.io/klog/v2 v2.130.1 h1:n9Xl7H1Xvksem4KFG4PYbdQCQxqc/tTUyrgXaOhHSzk= k8s.io/klog/v2 v2.130.1/go.mod h1:3Jpz1GvMt720eyJH1ckRHK1EDfpxISzJ7I9OYgaDtPE= lukechampine.com/blake3 v1.4.1 h1:I3Smz7gso8w4/TunLKec6K2fn+kyKtDxr/xcQEN84Wg= lukechampine.com/blake3 v1.4.1/go.mod h1:QFosUxmjB8mnrWFSNwKmvxHpfY72bmD2tQ0kBMM3kwo= -maunium.net/go/maulogger/v2 v2.4.1 h1:N7zSdd0mZkB2m2JtFUsiGTQQAdP0YeFWT7YMc80yAL8= -maunium.net/go/maulogger/v2 v2.4.1/go.mod h1:omPuYwYBILeVQobz8uO3XC8DIRuEb5rXYlQSuqrbCho= -maunium.net/go/mautrix v0.17.0 h1:scc1qlUbzPn+wc+3eAPquyD+3gZwwy/hBANBm+iGKK8= -maunium.net/go/mautrix v0.17.0/go.mod h1:j+puTEQCEydlVxhJ/dQP5chfa26TdvBO7X6F3Ataav8= +maunium.net/go/mautrix v0.25.2 h1:CUG23zp754yGOTMh9Q4mVSENS9FyweE/G+6ZsPDMCUU= +maunium.net/go/mautrix v0.25.2/go.mod h1:EWgYyp2iFZP7pnSm+rufHlO8YVnA2KnoNBDpwekiAwI= mvdan.cc/xurls/v2 v2.6.0 h1:3NTZpeTxYVWNSokW3MKeyVkz/j7uYXYiMtXRUfmjbgI= mvdan.cc/xurls/v2 v2.6.0/go.mod h1:bCvEZ1XvdA6wDnxY7jPPjEmigDtvtvPXAD/Exa9IMSk= oras.land/oras-go/v2 v2.6.0 h1:X4ELRsiGkrbeox69+9tzTu492FMUu7zJQW6eJU+I2oc= From b7155a9d974fa01d852d5abfb5caa6fc9025741f Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 20:37:03 +0000 Subject: [PATCH 41/49] chore(deps): bump LocalAGI to 7e0947d Pick up the LocalAGI PRs merged after f2a2af4: - per-collection embedding and reranker models, locked per collection so one agent's upload or rerank no longer stalls the others (#499) - required_tool_before_finish: a tool the agent must call successfully before it may answer (#495) - allowed_tools / excluded_tools per agent, applied to MCP tools too (#480) Document the new agent settings. They show up in the single-node agent form, which reads LocalAGI's config metadata; distributed mode keeps its own field list and does not offer them yet. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- docs/content/features/agents.md | 4 ++++ go.mod | 2 +- go.sum | 4 ++-- 3 files changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/content/features/agents.md b/docs/content/features/agents.md index cde92862e..86bc4a31a 100644 --- a/docs/content/features/agents.md +++ b/docs/content/features/agents.md @@ -188,6 +188,10 @@ Each agent has its own configuration that controls its behavior. Key settings in - **Connectors** - external integrations (Slack, Discord, etc.) - **Knowledge Base** - collections of documents for RAG - **MCP Servers** - Model Context Protocol servers for additional tool access +- **Allowed / Excluded Tools** (`allowed_tools`, `excluded_tools`) - limit the tools the agent can see, including MCP tools. The agent always keeps its control actions (`send_message`, `stop`, `update_state`). If a tool is in both lists, it is excluded. +- **Required Tool Before Finish** (`required_tool_before_finish`) - a tool the agent must call successfully before it can give its final answer, for example a validation or policy check. `required_tool_before_finish_prompt` changes the reminder the model gets when it tries to finish early. `required_tool_before_finish_attempts` sets how many reminders it gets before the answer goes through anyway (default 3). + +The tool lists and the required-tool settings are available only in single-node mode. The agent form in distributed mode does not show them yet. The pool-level defaults (API URL, API key, models) can be set via environment variables. Individual agents can further override these in their configuration, allowing them to use different LLM providers (OpenAI, other LocalAI instances, etc.) on a per-agent basis. diff --git a/go.mod b/go.mod index 4c76ab5ee..ed69ee4ea 100644 --- a/go.mod +++ b/go.mod @@ -238,7 +238,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c + github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect diff --git a/go.sum b/go.sum index 4b5fbbf02..6a01b5bd9 100644 --- a/go.sum +++ b/go.sum @@ -994,8 +994,8 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI= github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8= github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM= github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw= -github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c h1:6dvy3mqvRge8Ddji4er30fhoAasQTBJSYwGs1xVVSr8= -github.com/mudler/LocalAGI v0.0.0-20260927200938-f2a2af4e982c/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg= +github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca h1:bHlzSuOc5cKvHGF21ZjFFUV7IlkS3wt9YkBsWtkcaC4= +github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= From 84e2fc5eacb0154169bc80d253f14519c397ec22 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 20:49:12 +0000 Subject: [PATCH 42/49] feat(agents): support tool lists and required tool in distributed mode LocalAGI 7e0947d added allowed_tools/excluded_tools and the required_tool_before_finish gate. Single-node agents get them through LocalAGI's runtime, but the distributed executor drives cogito directly and its static config meta did not list the fields, so the agent form hid them and the worker ignored them. The distributed config now parses the tool lists from a JSON array or a comma/newline separated string, and the meta entries match LocalAGI's. The executor filters the knowledge base, skill and MCP tools (MCP via cogito.WithMCPToolFilter) before the model sees them, and re-prompts the model when it answers before the required tool returned "ok": true, up to the configured number of reminders. LocalAGI keeps its filter and gate helpers unexported, so a minimal copy lives in core/services/agents/toolpolicy.go. A spec compares the meta entries with LocalAGI's to catch drift. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- core/services/agents/config.go | 7 + core/services/agents/configmeta.go | 20 ++ core/services/agents/executor.go | 65 +++- core/services/agents/toolpolicy.go | 220 ++++++++++++++ core/services/agents/toolpolicy_test.go | 387 ++++++++++++++++++++++++ docs/content/features/agents.md | 2 - 6 files changed, 693 insertions(+), 8 deletions(-) create mode 100644 core/services/agents/toolpolicy.go create mode 100644 core/services/agents/toolpolicy_test.go diff --git a/core/services/agents/config.go b/core/services/agents/config.go index 0016cf51e..875eedd36 100644 --- a/core/services/agents/config.go +++ b/core/services/agents/config.go @@ -107,6 +107,13 @@ type AgentConfig struct { LoopDetection int `json:"loop_detection"` EnableAutoCompaction bool `json:"enable_auto_compaction"` AutoCompactionThreshold int `json:"auto_compaction_threshold"` + + // Tool policy (see toolpolicy.go) + RequiredToolBeforeFinish string `json:"required_tool_before_finish"` + RequiredToolBeforeFinishPrompt string `json:"required_tool_before_finish_prompt"` + RequiredToolBeforeFinishAttempts int `json:"required_tool_before_finish_attempts"` + AllowedTools ToolNames `json:"allowed_tools"` + ExcludedTools ToolNames `json:"excluded_tools"` } // ConnectorConfig defines a connector integration (Slack, Discord, etc.). diff --git a/core/services/agents/configmeta.go b/core/services/agents/configmeta.go index d767aaff8..9f36ed1a9 100644 --- a/core/services/agents/configmeta.go +++ b/core/services/agents/configmeta.go @@ -134,6 +134,14 @@ func defaultFields() []ConfigField { {Name: "enable_reasoning_tool", Label: "Enable Reasoning for Tools", Type: FieldCheckbox, DefaultValue: true, Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, {Name: "enable_reasoning_for_instruct", Label: "Enable Reasoning for Instruct Models", Type: FieldCheckbox, DefaultValue: false, HelpText: "Force structured reasoning before tool selection (recommended for instruct-tuned models)", Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, {Name: "enable_guided_tools", Label: "Enable Guided Tools", Type: FieldCheckbox, DefaultValue: false, HelpText: "Filter tools through guidance using descriptions", Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, + {Name: "allowed_tools", Label: "Allowed Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "get_document_content, search", + HelpText: "Comma or newline separated tool names. When set, the agent is offered only these tools (actions, knowledge base tools and MCP tools). send_message, stop and update_state are always kept. Leave empty to offer every tool.", + Tags: ConfigFieldTags{Section: "AdvancedSettings"}, + }, + {Name: "excluded_tools", Label: "Excluded Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "search_memory", + HelpText: "Comma or newline separated tool names that are never offered to the agent, even if they are in Allowed Tools. send_message, stop and update_state cannot be excluded; use their own settings instead.", + Tags: ConfigFieldTags{Section: "AdvancedSettings"}, + }, {Name: "enable_skills", Label: "Enable Skills", Type: FieldCheckbox, DefaultValue: false, HelpText: "Inject skills into the agent", Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, {Name: "skills_mode", Label: "Skills Injection Mode", Type: FieldSelect, DefaultValue: "prompt", Options: []ConfigFieldOption{ @@ -146,6 +154,18 @@ func defaultFields() []ConfigField { }, {Name: "parallel_jobs", Label: "Parallel Jobs", Type: FieldNumber, DefaultValue: 5, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, {Name: "max_attempts", Label: "Max Attempts", Type: FieldNumber, DefaultValue: 2, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, + {Name: "required_tool_before_finish", Label: "Required Tool Before Finish", Type: FieldText, DefaultValue: "", Placeholder: "check_policy", + HelpText: "Name of a tool the agent must call successfully (a JSON result with \"ok\": true) before it may send its final answer. Has no effect if the agent does not have this tool. Leave empty to disable.", + Tags: ConfigFieldTags{Section: "AdvancedSettings"}, + }, + {Name: "required_tool_before_finish_prompt", Label: "Required Tool Prompt", Type: FieldTextarea, DefaultValue: "", + HelpText: "Instruction sent to the model when it tries to finish before the required tool has passed. Leave empty to use a default that names the tool.", + Tags: ConfigFieldTags{Section: "AdvancedSettings"}, + }, + {Name: "required_tool_before_finish_attempts", Label: "Required Tool Attempts", Type: FieldNumber, DefaultValue: 3, Min: 1, Step: 1, + HelpText: "How many times the model is told to run the required tool before its answer is sent anyway", + Tags: ConfigFieldTags{Section: "AdvancedSettings"}, + }, {Name: "max_iterations", Label: "Max Iterations", Type: FieldNumber, DefaultValue: 1, Min: 1, Step: 1, HelpText: "Maximum tool loop iterations per execution", Tags: ConfigFieldTags{Section: "AdvancedSettings"}}, // MCP diff --git a/core/services/agents/executor.go b/core/services/agents/executor.go index 2787aeabc..9fa3c336e 100644 --- a/core/services/agents/executor.go +++ b/core/services/agents/executor.go @@ -4,6 +4,7 @@ import ( "cmp" "context" "encoding/json" + "errors" "fmt" "strings" "time" @@ -181,6 +182,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m // Build cogito options var cogitoOpts []cogito.Option + // Local tools are collected first so the tool filter applies to all of + // them at once; cogito only runs tools it offered, so filtering what is + // offered also filters the lookup of the model's tool calls. + var localTools []cogito.ToolDefinitionInterface + filter := newToolFilter(cfg.AllowedTools, cfg.ExcludedTools) // MCP sessions sessions, cleanup := setupMCPSessions(ctx, cfg) @@ -188,7 +194,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m defer cleanup() } if len(sessions) > 0 { - cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...)) + cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...), cogito.WithMCPToolFilter(filter.mcpToolFilter())) } // KB tools (search_memory / add_memory) — when kb mode is "tools" or "both" @@ -197,7 +203,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m if kbResults <= 0 { kbResults = 5 } - cogitoOpts = append(cogitoOpts, cogito.WithTools( + localTools = append(localTools, cogito.NewToolDefinition( KBSearchMemoryTool{APIURL: effectiveURL, APIKey: effectiveKey, Collection: cfg.Name, MaxResults: kbResults, UserID: userID, CitationCollector: kbCitations}, KBSearchMemoryArgs{}, @@ -210,7 +216,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m "add_memory", "Store content in memory for later retrieval", ), - )) + ) } // Skill tools — when skills_mode is "tools" or "both" @@ -220,18 +226,36 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m allSkills, _ := skillProvider.ListSkills() filtered := FilterSkills(allSkills, cfg.SelectedSkills) if len(filtered) > 0 { - cogitoOpts = append(cogitoOpts, cogito.WithTools( + localTools = append(localTools, cogito.NewToolDefinition( RequestSkillTool{Skills: filtered}, RequestSkillArgs{}, "request_skill", "Request a skill by name. Available skills: "+skillNames(filtered), ), - )) + ) } } } + localTools = filter.filterTools(localTools) + if len(localTools) > 0 { + cogitoOpts = append(cogitoOpts, cogito.WithTools(localTools...)) + } + + // Required-tool gate: the agent must run the configured tool to success + // before its answer is final. It is enforced on the output because a + // model follows "always call X first" unreliably. + requiredTool := cfg.RequiredToolBeforeFinish + requiredPassed := false + requiredAttempts := 0 + maxRequiredAttempts := cfg.RequiredToolBeforeFinishAttempts + if maxRequiredAttempts <= 0 { + maxRequiredAttempts = defaultRequiredFinishAttempts + } + requiredPrompt := requiredFinishPromptFor(requiredTool, cfg.RequiredToolBeforeFinishPrompt) + requiredAvailable := requiredTool != "" && requiredToolAvailable(ctx, requiredTool, localTools, sessions, filter) + // Sink state is always disabled — the agent responds directly when no tools match. cogitoOpts = append(cogitoOpts, cogito.DisableSinkState) @@ -250,8 +274,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m } // Tool call result callback - if cb.OnToolResult != nil || cb.OnToolCall != nil { + if cb.OnToolResult != nil || cb.OnToolCall != nil || requiredAvailable { cogitoOpts = append(cogitoOpts, cogito.WithToolCallResultCallback(func(t cogito.ToolStatus) { + if requiredAvailable && t.Name == requiredTool && requiredToolResultOK(t.Result) { + requiredPassed = true + } if isInternalCogitoTool(t.Name) { return } @@ -327,6 +354,32 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m return "", fmt.Errorf("agent execution failed: %w", err) } + for len(result.Messages) > 0 && textFinalizationNeedsRequiredTool(requiredAvailable, requiredPassed, + requiredAttempts, maxRequiredAttempts, result.LastMessage().Role, result.LastMessage().Content) { + requiredAttempts++ + xlog.Info("required-tool gate: answer without the required tool, nudging", + "agent", cfg.Name, "tool", requiredTool, "attempt", requiredAttempts) + answered := result + next, err := cogito.ExecuteTools(llm, result.AddMessage(cogito.UserMessageRole, requiredPrompt), cogitoOpts...) + if err != nil && ctx.Err() != nil { + if cb.OnStatus != nil { + cb.OnStatus("error: " + err.Error()) + } + return "", fmt.Errorf("agent execution failed: %w", err) + } + // A failed retry must not throw away the answer the model already gave. + if err != nil && !errors.Is(err, cogito.ErrNoToolSelected) { + xlog.Error("required-tool gate: retry failed, keeping the previous answer", "agent", cfg.Name, "error", err) + result = answered + break + } + result = next + } + if requiredAvailable && !requiredPassed && requiredAttempts >= maxRequiredAttempts { + xlog.Warn("required-tool gate: bypass after max attempts, answer finalized ungated", + "agent", cfg.Name, "tool", requiredTool) + } + // Extract response response := "" if len(result.Messages) > 0 { diff --git a/core/services/agents/toolpolicy.go b/core/services/agents/toolpolicy.go new file mode 100644 index 000000000..cb5d97cdd --- /dev/null +++ b/core/services/agents/toolpolicy.go @@ -0,0 +1,220 @@ +package agents + +// Tool policy for the distributed executor: the allowed/excluded tool lists and +// the required-tool-before-finish gate. The semantics mirror LocalAGI's +// core/agent/toolfilter.go and the gate in core/agent/agent.go. Those helpers +// are unexported there, so the small pieces below are kept in step by hand; the +// meta parity spec in toolpolicy_test.go catches drift in the form fields. + +import ( + "context" + "encoding/json" + "fmt" + "strings" + + gomcp "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/mudler/cogito" + "github.com/mudler/xlog" +) + +// ToolNames is a list of tool names. The agent form submits it as a comma or +// newline separated string and the API as a JSON array, so both are accepted. +type ToolNames []string + +// UnmarshalJSON accepts a JSON array of strings, a comma or newline separated +// string, or null. Names are trimmed and empty entries dropped. +func (t *ToolNames) UnmarshalJSON(data []byte) error { + var value any + if err := json.Unmarshal(data, &value); err != nil { + return err + } + var raw []string + switch v := value.(type) { + case nil: + *t = nil + return nil + case string: + raw = strings.FieldsFunc(v, func(r rune) bool { return r == ',' || r == '\n' || r == '\r' }) + case []any: + for _, item := range v { + name, ok := item.(string) + if !ok { + return fmt.Errorf("expected a list of tool names, got %T", item) + } + raw = append(raw, name) + } + default: + return fmt.Errorf("expected a list of tool names or a comma separated string, got %T", value) + } + var names ToolNames + for _, n := range raw { + if n = strings.TrimSpace(n); n != "" { + names = append(names, n) + } + } + *t = names + return nil +} + +// controlActionNames are LocalAGI's loop-driving actions. The distributed +// executor does not offer them today, but an agent config is shared between +// both modes, so the filter must treat them the same way in both. +var controlActionNames = map[string]struct{}{ + "send_message": {}, + "stop": {}, + "update_state": {}, +} + +// toolFilter is an allow/deny list over tool names. A nil *toolFilter allows +// everything. +type toolFilter struct { + allow map[string]struct{} + deny map[string]struct{} +} + +func newToolFilter(allow, deny []string) *toolFilter { + f := &toolFilter{allow: toNameSet(allow), deny: toNameSet(deny)} + if len(f.allow) == 0 && len(f.deny) == 0 { + return nil + } + return f +} + +func toNameSet(names []string) map[string]struct{} { + set := make(map[string]struct{}, len(names)) + for _, n := range names { + if n = strings.TrimSpace(n); n != "" { + set[n] = struct{}{} + } + } + return set +} + +func (f *toolFilter) allows(name string) bool { + if f == nil { + return true + } + if _, ok := controlActionNames[name]; ok { + return true + } + if _, denied := f.deny[name]; denied { + return false + } + if len(f.allow) == 0 { + return true + } + _, allowed := f.allow[name] + return allowed +} + +func (f *toolFilter) filterTools(tools []cogito.ToolDefinitionInterface) []cogito.ToolDefinitionInterface { + if f == nil { + return tools + } + out := make([]cogito.ToolDefinitionInterface, 0, len(tools)) + for _, t := range tools { + if f.allows(t.Tool().Function.Name) { + out = append(out, t) + } + } + return out +} + +// mcpToolFilter is needed on top of filterTools because cogito discovers MCP +// tools straight from the live sessions. +func (f *toolFilter) mcpToolFilter() cogito.MCPToolFilter { + if f == nil { + return nil + } + return func(_ *gomcp.ClientSession, toolName string) bool { + return f.allows(toolName) + } +} + +// defaultRequiredFinishAttempts bounds the reminders: a gate that can loop +// forever is worse than one that gives up loudly. +const defaultRequiredFinishAttempts = 3 + +func requiredFinishPromptFor(tool, override string) string { + if override != "" { + return override + } + return "Before you send your final answer you MUST first call the tool " + tool + + " and it must succeed (ok:true). Call " + tool + " now; only send the final " + + "message after it passes." +} + +// requiredToolResultOK reports whether a tool result is a JSON object with a +// top-level "ok": true. When the result is not JSON as a whole (MCP content +// may wrap it in text), each top-level object embedded in it is checked. +func requiredToolResultOK(result string) bool { + trimmed := strings.TrimSpace(result) + if json.Valid([]byte(trimmed)) { + return jsonObjectOK([]byte(trimmed)) + } + for i := 0; i < len(result); { + j := strings.IndexByte(result[i:], '{') + if j < 0 { + return false + } + start := i + j + dec := json.NewDecoder(strings.NewReader(result[start:])) + var raw json.RawMessage + if err := dec.Decode(&raw); err != nil { + i = start + 1 + continue + } + if jsonObjectOK(raw) { + return true + } + // Skip the whole object so its nested objects are not checked on their own. + i = start + int(dec.InputOffset()) + } + return false +} + +func jsonObjectOK(data []byte) bool { + var obj map[string]json.RawMessage + if err := json.Unmarshal(data, &obj); err != nil { + return false + } + var ok bool + if err := json.Unmarshal(obj["ok"], &ok); err != nil { + return false + } + return ok +} + +// requiredToolAvailable reports whether the model is offered the required +// tool. The gate stays inert otherwise, so a pool-wide setting is harmless for +// agents that lack the tool. MCP sessions are only listed when the tool is not +// a local one. +func requiredToolAvailable(ctx context.Context, name string, local []cogito.ToolDefinitionInterface, sessions []*gomcp.ClientSession, filter *toolFilter) bool { + if name == "" || !filter.allows(name) { + return false + } + if cogito.Tools(local).Find(name) != nil { + return true + } + for _, s := range sessions { + res, err := s.ListTools(ctx, nil) + if err != nil { + xlog.Warn("required-tool gate: failed to list MCP tools", "error", err) + continue + } + for _, t := range res.Tools { + if t.Name == name { + return true + } + } + } + return false +} + +// textFinalizationNeedsRequiredTool reports whether the run ended with a +// non-empty assistant answer although the required tool has not passed and +// reminders are left. +func textFinalizationNeedsRequiredTool(toolAvailable, toolPassed bool, attempts, max int, lastRole, lastContent string) bool { + return toolAvailable && !toolPassed && attempts < max && + lastRole == "assistant" && strings.TrimSpace(lastContent) != "" +} diff --git a/core/services/agents/toolpolicy_test.go b/core/services/agents/toolpolicy_test.go new file mode 100644 index 000000000..4a18d3f80 --- /dev/null +++ b/core/services/agents/toolpolicy_test.go @@ -0,0 +1,387 @@ +package agents + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sort" + "sync" + "sync/atomic" + + "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/mudler/LocalAGI/core/state" + "github.com/mudler/cogito" + openai "github.com/sashabaranov/go-openai" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +// mcpFixture serves an MCP server over SSE whose tools return fixed results, +// so the executor reaches it through the same transport as a real agent. +type mcpFixture struct { + server *httptest.Server + calls map[string]*atomic.Int32 +} + +func newMCPFixture(results map[string]string) *mcpFixture { + srv := mcp.NewServer(&mcp.Implementation{Name: "fixture", Version: "v0.0.1"}, nil) + fx := &mcpFixture{calls: map[string]*atomic.Int32{}} + for name, result := range results { + counter := &atomic.Int32{} + fx.calls[name] = counter + srv.AddTool(&mcp.Tool{ + Name: name, + Description: "fixture tool " + name, + InputSchema: json.RawMessage(`{"type":"object","properties":{}}`), + }, func(context.Context, *mcp.CallToolRequest) (*mcp.CallToolResult, error) { + counter.Add(1) + return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: result}}}, nil + }) + } + fx.server = httptest.NewServer(mcp.NewSSEHandler(func(*http.Request) *mcp.Server { return srv }, nil)) + return fx +} + +func (fx *mcpFixture) close() { fx.server.Close() } +func (fx *mcpFixture) callCount(n string) int32 { return fx.calls[n].Load() } + +// policyLLM answers each chat completion through respond (plain text answer +// when respond is nil) and records every +// request, so specs can see which tools were offered and which messages the +// executor added. +type policyLLM struct { + mu sync.Mutex + requests []openai.ChatCompletionRequest + asked [][]openai.ChatCompletionMessage + respond func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage + answer string +} + +func (m *policyLLM) Ask(_ context.Context, f cogito.Fragment) (cogito.Fragment, error) { + m.mu.Lock() + m.asked = append(m.asked, append([]openai.ChatCompletionMessage(nil), f.Messages...)) + m.mu.Unlock() + return f.AddMessage(cogito.AssistantMessageRole, m.answer), nil +} + +func (m *policyLLM) CreateChatCompletion(_ context.Context, req openai.ChatCompletionRequest) (cogito.LLMReply, cogito.LLMUsage, error) { + m.mu.Lock() + m.requests = append(m.requests, req) + m.mu.Unlock() + msg := openai.ChatCompletionMessage{Role: "assistant", Content: m.answer} + if m.respond != nil { + msg = m.respond(req) + } + return cogito.LLMReply{ + ChatCompletionResponse: openai.ChatCompletionResponse{ + Choices: []openai.ChatCompletionChoice{{Message: msg}}, + }, + }, cogito.LLMUsage{}, nil +} + +// offeredTools returns the sorted tool names of the first request that +// offered tools to the model. +func (m *policyLLM) offeredTools() []string { + m.mu.Lock() + defer m.mu.Unlock() + for _, req := range m.requests { + if len(req.Tools) == 0 { + continue + } + names := []string{} + for _, t := range req.Tools { + if t.Function != nil { + names = append(names, t.Function.Name) + } + } + sort.Strings(names) + return names + } + return nil +} + +// nudges counts the user messages carrying prompt in the longest conversation +// the model saw, which is the number of times the gate sent it. +func (m *policyLLM) nudges(prompt string) int { + m.mu.Lock() + defer m.mu.Unlock() + best := 0 + count := func(msgs []openai.ChatCompletionMessage) { + n := 0 + for _, msg := range msgs { + if msg.Role == "user" && msg.Content == prompt { + n++ + } + } + if n > best { + best = n + } + } + for _, req := range m.requests { + count(req.Messages) + } + for _, msgs := range m.asked { + count(msgs) + } + return best +} + +func toolCallMessage(name string) openai.ChatCompletionMessage { + return openai.ChatCompletionMessage{ + Role: "assistant", + ToolCalls: []openai.ToolCall{{ + ID: "call-" + name, + Type: openai.ToolTypeFunction, + Function: openai.FunctionCall{Name: name, Arguments: `{}`}, + }}, + } +} + +func lastMessage(req openai.ChatCompletionRequest) openai.ChatCompletionMessage { + if len(req.Messages) == 0 { + return openai.ChatCompletionMessage{} + } + return req.Messages[len(req.Messages)-1] +} + +var _ = Describe("tool policy settings", func() { + Describe("config parsing", func() { + It("accepts the tool lists as a comma or newline separated string", func() { + var cfg AgentConfig + Expect(ParseConfigJSON(`{"allowed_tools":"a, b\nc,,","excluded_tools":" d \r\n"}`, &cfg)).To(Succeed()) + Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b", "c"})) + Expect([]string(cfg.ExcludedTools)).To(Equal([]string{"d"})) + }) + + It("accepts the tool lists as a JSON array", func() { + var cfg AgentConfig + Expect(ParseConfigJSON(`{"allowed_tools":["a"," b ",""],"excluded_tools":null}`, &cfg)).To(Succeed()) + Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b"})) + Expect(cfg.ExcludedTools).To(BeEmpty()) + }) + + It("rejects a list with non-string entries", func() { + var cfg AgentConfig + Expect(ParseConfigJSON(`{"allowed_tools":[1]}`, &cfg)).ToNot(Succeed()) + }) + + It("keeps every setting when the config is stored through LocalAGI's config", func() { + // The REST handlers decode into state.AgentConfig and store its JSON; + // the distributed dispatcher decodes that JSON into AgentConfig. + var in state.AgentConfig + Expect(json.Unmarshal([]byte(`{ + "name": "a", + "allowed_tools": "search, check_policy", + "excluded_tools": ["add_memory"], + "required_tool_before_finish": "check_policy", + "required_tool_before_finish_prompt": "run it", + "required_tool_before_finish_attempts": 4 + }`), &in)).To(Succeed()) + stored, err := json.Marshal(in) + Expect(err).ToNot(HaveOccurred()) + + var out AgentConfig + Expect(ParseConfigJSON(string(stored), &out)).To(Succeed()) + Expect([]string(out.AllowedTools)).To(Equal([]string{"search", "check_policy"})) + Expect([]string(out.ExcludedTools)).To(Equal([]string{"add_memory"})) + Expect(out.RequiredToolBeforeFinish).To(Equal("check_policy")) + Expect(out.RequiredToolBeforeFinishPrompt).To(Equal("run it")) + Expect(out.RequiredToolBeforeFinishAttempts).To(Equal(4)) + + again, err := json.Marshal(out) + Expect(err).ToNot(HaveOccurred()) + var back state.AgentConfig + Expect(json.Unmarshal(again, &back)).To(Succeed()) + Expect(back.AllowedTools).To(Equal([]string{"search", "check_policy"})) + Expect(back.RequiredToolBeforeFinishAttempts).To(Equal(4)) + }) + }) + + Describe("config meta", func() { + It("describes the settings exactly like LocalAGI does", func() { + upstream := map[string]ConfigField{} + for _, f := range state.NewAgentConfigMeta(nil, nil, nil, nil).Fields { + upstream[f.Name] = ConfigField{ + Name: f.Name, Type: string(f.Type), Label: f.Label, DefaultValue: f.DefaultValue, + Placeholder: f.Placeholder, HelpText: f.HelpText, Min: f.Min, Max: f.Max, Step: f.Step, + Tags: ConfigFieldTags{Section: f.Tags.Section}, + } + } + local := map[string]ConfigField{} + for _, f := range DefaultConfigMeta().Fields { + local[f.Name] = f + } + for _, name := range []string{ + "allowed_tools", "excluded_tools", + "required_tool_before_finish", "required_tool_before_finish_prompt", "required_tool_before_finish_attempts", + } { + Expect(upstream).To(HaveKey(name)) + Expect(local).To(HaveKeyWithValue(name, upstream[name]), name) + } + }) + }) + + Describe("tool filter", func() { + It("keeps the control actions even when they are excluded or not allowed", func() { + f := newToolFilter([]string{"search"}, []string{"send_message", "stop", "update_state", "search"}) + for _, name := range []string{"send_message", "stop", "update_state"} { + Expect(f.allows(name)).To(BeTrue(), name) + } + Expect(f.allows("search")).To(BeFalse()) + Expect(f.allows("other")).To(BeFalse()) + }) + }) + + Describe("ExecuteChatWithLLM", func() { + var fx *mcpFixture + + BeforeEach(func() { + fx = newMCPFixture(map[string]string{ + "check_policy": `{"ok":true}`, + "mcp_allowed": "allowed result", + "mcp_blocked": "blocked result", + }) + }) + + AfterEach(func() { fx.close() }) + + baseConfig := func() *AgentConfig { + return &AgentConfig{ + Name: "policy-agent", + Model: "test-model", + MCPServers: []MCPServer{{URL: fx.server.URL}}, + EnableKnowledgeBase: true, + KBMode: KBModeTools, + } + } + + Context("with allowed and excluded tools", func() { + It("offers the model only the allowed tools that are not excluded, MCP tools included", func() { + llm := &policyLLM{answer: "final"} + cfg := baseConfig() + cfg.AllowedTools = ToolNames{"mcp_allowed", "search_memory", "add_memory"} + cfg.ExcludedTools = ToolNames{"add_memory"} + + _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(llm.offeredTools()).To(Equal([]string{"mcp_allowed", "search_memory"})) + }) + + It("offers every tool when no list is set", func() { + llm := &policyLLM{answer: "final"} + _, err := ExecuteChatWithLLM(context.Background(), llm, baseConfig(), "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(llm.offeredTools()).To(Equal([]string{"add_memory", "check_policy", "mcp_allowed", "mcp_blocked", "search_memory"})) + }) + + It("does not run a filtered MCP tool the model calls anyway", func() { + var calls atomic.Int32 + llm := &policyLLM{answer: "final", respond: func(openai.ChatCompletionRequest) openai.ChatCompletionMessage { + if calls.Add(1) == 1 { + return toolCallMessage("mcp_blocked") + } + return openai.ChatCompletionMessage{Role: "assistant", Content: "done"} + }} + cfg := baseConfig() + cfg.ExcludedTools = ToolNames{"mcp_blocked"} + + _, _ = ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(fx.callCount("mcp_blocked")).To(BeZero()) + }) + }) + + Context("with a required tool before finish", func() { + const prompt = "RUN check_policy NOW" + + It("nudges the model until the required tool passes, then returns its answer", func() { + llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage { + if last := lastMessage(req); last.Role == "user" && last.Content == prompt { + return toolCallMessage("check_policy") + } + return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"} + }} + cfg := baseConfig() + cfg.RequiredToolBeforeFinish = "check_policy" + cfg.RequiredToolBeforeFinishPrompt = prompt + + result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(result).To(Equal("final answer")) + Expect(fx.callCount("check_policy")).To(Equal(int32(1))) + Expect(llm.nudges(prompt)).To(Equal(1)) + }) + + It("lets the answer through after the configured number of reminders", func() { + llm := &policyLLM{answer: "stubborn answer"} + cfg := baseConfig() + cfg.RequiredToolBeforeFinish = "check_policy" + cfg.RequiredToolBeforeFinishPrompt = prompt + cfg.RequiredToolBeforeFinishAttempts = 2 + + result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(result).To(Equal("stubborn answer")) + Expect(llm.nudges(prompt)).To(Equal(2)) + Expect(fx.callCount("check_policy")).To(BeZero()) + }) + + It("uses three reminders and a prompt naming the tool by default", func() { + llm := &policyLLM{answer: "stubborn answer"} + cfg := baseConfig() + cfg.RequiredToolBeforeFinish = "check_policy" + + _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(llm.nudges(requiredFinishPromptFor("check_policy", ""))).To(Equal(3)) + Expect(requiredFinishPromptFor("check_policy", "")).To(ContainSubstring("check_policy")) + }) + + It("keeps nudging when the required tool fails", func() { + fx.close() + fx = newMCPFixture(map[string]string{"check_policy": `{"ok":false}`}) + llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage { + if last := lastMessage(req); last.Role == "user" && last.Content == prompt { + return toolCallMessage("check_policy") + } + return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"} + }} + cfg := baseConfig() + cfg.RequiredToolBeforeFinish = "check_policy" + cfg.RequiredToolBeforeFinishPrompt = prompt + cfg.RequiredToolBeforeFinishAttempts = 2 + + _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(fx.callCount("check_policy")).To(Equal(int32(2))) + Expect(llm.nudges(prompt)).To(Equal(2)) + }) + + It("does nothing when the agent does not have the required tool", func() { + llm := &policyLLM{answer: "final"} + cfg := baseConfig() + cfg.RequiredToolBeforeFinish = "check_policy" + cfg.RequiredToolBeforeFinishPrompt = prompt + cfg.ExcludedTools = ToolNames{"check_policy"} + + result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{}) + Expect(err).ToNot(HaveOccurred()) + Expect(result).To(Equal("final")) + Expect(llm.nudges(prompt)).To(BeZero()) + }) + }) + }) +}) + +var _ = DescribeTable("requiredToolResultOK", + func(result string, want bool) { + Expect(requiredToolResultOK(result)).To(Equal(want)) + }, + Entry("top-level ok true", `{"ok":true}`, true), + Entry("top-level ok false", `{"ok":false}`, false), + Entry("ok as a string", `{"ok":"true"}`, false), + Entry("ok nested in another object", `{"data":{"ok":true}}`, false), + Entry("object embedded in text", `result: {"ok": true, "n": 1} done`, true), + Entry("plain text", `"ok": true`, false), +) diff --git a/docs/content/features/agents.md b/docs/content/features/agents.md index 86bc4a31a..4e9026942 100644 --- a/docs/content/features/agents.md +++ b/docs/content/features/agents.md @@ -191,8 +191,6 @@ Each agent has its own configuration that controls its behavior. Key settings in - **Allowed / Excluded Tools** (`allowed_tools`, `excluded_tools`) - limit the tools the agent can see, including MCP tools. The agent always keeps its control actions (`send_message`, `stop`, `update_state`). If a tool is in both lists, it is excluded. - **Required Tool Before Finish** (`required_tool_before_finish`) - a tool the agent must call successfully before it can give its final answer, for example a validation or policy check. `required_tool_before_finish_prompt` changes the reminder the model gets when it tries to finish early. `required_tool_before_finish_attempts` sets how many reminders it gets before the answer goes through anyway (default 3). -The tool lists and the required-tool settings are available only in single-node mode. The agent form in distributed mode does not show them yet. - The pool-level defaults (API URL, API key, models) can be set via environment variables. Individual agents can further override these in their configuration, allowing them to use different LLM providers (OpenAI, other LocalAI instances, etc.) on a per-agent basis. ## Skills From eb97d5f5493493de02cdc64118895ba1ab3c8688 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:29:48 +0200 Subject: [PATCH 43/49] chore(deps): bump grpcio from 1.83.1 to 1.84.0 in /backend/python/coqui (#12106) Bumps [grpcio](https://github.com/grpc/grpc) from 1.83.1 to 1.84.0. - [Release notes](https://github.com/grpc/grpc/releases) - [Commits](https://github.com/grpc/grpc/compare/v1.83.1...v1.84.0) --- updated-dependencies: - dependency-name: grpcio dependency-version: 1.84.0 dependency-type: direct:production update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/coqui/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/python/coqui/requirements.txt b/backend/python/coqui/requirements.txt index 305f95f12..35c72a10f 100644 --- a/backend/python/coqui/requirements.txt +++ b/backend/python/coqui/requirements.txt @@ -1,4 +1,4 @@ -grpcio==1.83.1 +grpcio==1.84.0 protobuf certifi packaging==26.3 \ No newline at end of file From afebecc63d9b083a0a2ffe650f058f056c833c0a Mon Sep 17 00:00:00 2001 From: leilei3167 Date: Mon, 28 Sep 2026 06:11:41 +0800 Subject: [PATCH 44/49] fix(ollama): report on-disk size for /api/tags and /api/ps (#11989) Hardcoding size/size_vram as 0 made Ollama clients treat loaded models as free. Prefer ModelFileName+ModelPath Stat when available, and omit size_vram (and size) when the value is unknown instead of emitting literal zeros. Resolve each listed model by its stored ID so tagged variants use their own weights. Fixes #11969 Signed-off-by: lei_lei --- core/http/endpoints/ollama/models.go | 51 +++++-- core/http/endpoints/ollama/models_test.go | 163 +++++++++++++++++++++- core/schema/ollama.go | 15 +- docs/content/getting-started/models.md | 2 + 4 files changed, 216 insertions(+), 15 deletions(-) diff --git a/core/http/endpoints/ollama/models.go b/core/http/endpoints/ollama/models.go index 60e58b9ea..34c8b0d6e 100644 --- a/core/http/endpoints/ollama/models.go +++ b/core/http/endpoints/ollama/models.go @@ -3,6 +3,8 @@ package ollama import ( "crypto/sha256" "fmt" + "os" + "path/filepath" "strings" "time" @@ -37,7 +39,7 @@ func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) ec Name: ollamaName, Model: ollamaName, ModifiedAt: time.Now().UTC(), - Size: 0, + Size: modelOnDiskSize(bcl, ml, name), Digest: digest, Details: details, Capabilities: caps, @@ -101,13 +103,15 @@ func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) e details, caps := modelMetaFromConfig(bcl, name) entry := schema.OllamaPsEntry{ - Name: ollamaName, - Model: ollamaName, - Size: 0, - Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))), - Details: details, - ExpiresAt: time.Now().Add(24 * time.Hour).UTC(), - SizeVRAM: 0, + Name: ollamaName, + Model: ollamaName, + Size: modelOnDiskSize(bcl, ml, name), + Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))), + Details: details, + ExpiresAt: time.Now().Add(24 * time.Hour).UTC(), + // SizeVRAM is left unset: LocalAI has no authoritative per-model + // VRAM figure to report, and a literal 0 is worse than omitting + // the field (clients treat 0 as "costs nothing"). Capabilities: caps, } models = append(models, entry) @@ -143,6 +147,37 @@ func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.Oll return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg) } +// modelOnDiskSize returns the on-disk byte size of a model's primary weight +// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath. +// Returns nil when the size is unknown so callers omit the JSON field instead +// of emitting an authoritative 0 (issue #11969). +func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 { + if ml == nil || ml.ModelPath == "" { + return nil + } + + // List endpoints pass the stored model ID, including any configured tag. + configName := name + rel := configName + if bcl != nil { + if cfg, exists := bcl.GetModelConfig(configName); exists { + if fileName := cfg.ModelFileName(); fileName != "" { + rel = fileName + } + } + } + if rel == "" { + return nil + } + + info, err := os.Stat(filepath.Join(ml.ModelPath, rel)) + if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 { + return nil + } + size := info.Size() + return &size +} + func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails { family := cfg.Backend details := schema.OllamaModelDetails{ diff --git a/core/http/endpoints/ollama/models_test.go b/core/http/endpoints/ollama/models_test.go index c4d0d6b5e..bf3c64b90 100644 --- a/core/http/endpoints/ollama/models_test.go +++ b/core/http/endpoints/ollama/models_test.go @@ -13,6 +13,8 @@ import ( "github.com/mudler/LocalAI/core/config" "github.com/mudler/LocalAI/core/http/endpoints/ollama" "github.com/mudler/LocalAI/core/schema" + "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/system" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" ) @@ -163,8 +165,165 @@ parameters: }) Describe("ListModelsEndpoint", func() { - It("includes capabilities and details for each listed model in /api/tags", func() { - Skip("covered by per-entry tests; integration smoke test") + var ( + tmpDir string + bcl *config.ModelConfigLoader + ml *model.ModelLoader + ) + + BeforeEach(func() { + var err error + tmpDir, err = os.MkdirTemp("", "ollama-tags-test-*") + Expect(err).ToNot(HaveOccurred()) + + systemState, err := system.GetSystemState(system.WithModelPath(tmpDir)) + Expect(err).ToNot(HaveOccurred()) + ml = model.NewModelLoader(systemState) + bcl = config.NewModelConfigLoader(tmpDir) + }) + + AfterEach(func() { + _ = os.RemoveAll(tmpDir) + }) + + writeConfig := func(name, yaml string) { + path := filepath.Join(tmpDir, name+".yaml") + Expect(os.WriteFile(path, []byte(yaml), 0o644)).To(Succeed()) + Expect(bcl.ReadModelConfig(path)).To(Succeed()) + } + + callTags := func() (schema.OllamaListResponse, []byte) { + req := httptest.NewRequest(http.MethodGet, "/api/tags", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + handler := ollama.ListModelsEndpoint(bcl, ml) + Expect(handler(c)).To(Succeed()) + Expect(rec.Code).To(Equal(http.StatusOK)) + + var resp schema.OllamaListResponse + Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed()) + return resp, rec.Body.Bytes() + } + + It("uses the exact configured name when a model has a tag", func() { + Expect(os.WriteFile(filepath.Join(tmpDir, "base.gguf"), []byte("base"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(tmpDir, "tagged.gguf"), []byte("tagged-weights"), 0o644)).To(Succeed()) + writeConfig("chat", "name: chat\nparameters:\n model: base.gguf\n") + writeConfig("tagged", "name: chat:q8\nparameters:\n model: tagged.gguf\n") + + resp, _ := callTags() + var tagged *int64 + for _, entry := range resp.Models { + if entry.Name == "chat:q8" { + tagged = entry.Size + } + } + Expect(tagged).ToNot(BeNil()) + Expect(*tagged).To(Equal(int64(len("tagged-weights")))) + }) + + It("reports on-disk size from ModelFileName+ModelPath and omits size when unknown", func() { + weight := []byte("fake-gguf-weights-0123456789") + Expect(os.WriteFile(filepath.Join(tmpDir, "Llama-3-8B-Q4_K_M.gguf"), weight, 0o644)).To(Succeed()) + writeConfig("chat", ` +name: chat +backend: llama-cpp +template: + chat: "{{ .Input }}" +parameters: + model: Llama-3-8B-Q4_K_M.gguf +`) + writeConfig("missing-weights", ` +name: missing-weights +backend: llama-cpp +template: + chat: "{{ .Input }}" +parameters: + model: does-not-exist.gguf +`) + + resp, raw := callTags() + Expect(resp.Models).To(HaveLen(2)) + + byName := map[string]schema.OllamaModelEntry{} + for _, m := range resp.Models { + byName[m.Name] = m + } + + chat := byName["chat:latest"] + Expect(chat.Size).ToNot(BeNil()) + Expect(*chat.Size).To(Equal(int64(len(weight)))) + Expect(chat.Capabilities).To(ContainElement("completion")) + Expect(chat.Details.QuantizationLevel).To(Equal("Q4_K_M")) + + missing := byName["missing-weights:latest"] + Expect(missing.Size).To(BeNil()) + Expect(string(raw)).ToNot(ContainSubstring(`"size":0`)) + }) + }) + + Describe("ListRunningEndpoint", func() { + var ( + tmpDir string + bcl *config.ModelConfigLoader + ml *model.ModelLoader + ) + + BeforeEach(func() { + var err error + tmpDir, err = os.MkdirTemp("", "ollama-ps-test-*") + Expect(err).ToNot(HaveOccurred()) + + systemState, err := system.GetSystemState(system.WithModelPath(tmpDir)) + Expect(err).ToNot(HaveOccurred()) + ml = model.NewModelLoader(systemState) + bcl = config.NewModelConfigLoader(tmpDir) + }) + + AfterEach(func() { + _ = os.RemoveAll(tmpDir) + }) + + It("reports on-disk size for loaded models and omits size_vram when unknown", func() { + weight := []byte("loaded-model-weights-abcdef") + Expect(os.WriteFile(filepath.Join(tmpDir, "granite-Q4_K_M.gguf"), weight, 0o644)).To(Succeed()) + + cfgPath := filepath.Join(tmpDir, "granite.yaml") + Expect(os.WriteFile(cfgPath, []byte(` +name: granite +backend: llama-cpp +template: + chat: "{{ .Input }}" +parameters: + model: granite-Q4_K_M.gguf +`), 0o644)).To(Succeed()) + Expect(bcl.ReadModelConfig(cfgPath)).To(Succeed()) + + store := model.NewInMemoryModelStore() + store.Set("granite", model.NewModel("granite", "addr", nil)) + ml.SetModelStore(store) + + req := httptest.NewRequest(http.MethodGet, "/api/ps", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + handler := ollama.ListRunningEndpoint(bcl, ml) + Expect(handler(c)).To(Succeed()) + Expect(rec.Code).To(Equal(http.StatusOK)) + + raw := rec.Body.String() + Expect(raw).ToNot(ContainSubstring(`"size":0`)) + Expect(raw).ToNot(ContainSubstring(`"size_vram"`)) + + var resp schema.OllamaPsResponse + Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed()) + Expect(resp.Models).To(HaveLen(1)) + Expect(resp.Models[0].Name).To(Equal("granite:latest")) + Expect(resp.Models[0].Size).ToNot(BeNil()) + Expect(*resp.Models[0].Size).To(Equal(int64(len(weight)))) + Expect(resp.Models[0].SizeVRAM).To(BeNil()) + Expect(resp.Models[0].Details.QuantizationLevel).To(Equal("Q4_K_M")) }) }) }) diff --git a/core/schema/ollama.go b/core/schema/ollama.go index 8ea414dde..ee496508f 100644 --- a/core/schema/ollama.go +++ b/core/schema/ollama.go @@ -293,12 +293,14 @@ type OllamaModelDetails struct { QuantizationLevel string `json:"quantization_level,omitempty"` } -// OllamaModelEntry represents a model in the list response +// OllamaModelEntry represents a model in the list response. +// Size is a pointer so an unknown on-disk size can be omitted instead of +// serializing as the misleading literal 0 (see issue #11969). type OllamaModelEntry struct { Name string `json:"name"` Model string `json:"model"` ModifiedAt time.Time `json:"modified_at"` - Size int64 `json:"size"` + Size *int64 `json:"size,omitempty"` Digest string `json:"digest"` Details OllamaModelDetails `json:"details"` Capabilities []string `json:"capabilities,omitempty"` @@ -309,15 +311,18 @@ type OllamaListResponse struct { Models []OllamaModelEntry `json:"models"` } -// OllamaPsEntry represents a running model in the ps response +// OllamaPsEntry represents a running model in the ps response. +// Size and SizeVRAM are pointers so unknown values are omitted rather than +// reported as authoritative zeros (see issue #11969). SizeVRAM is only set +// when the runtime can provide a real VRAM figure. type OllamaPsEntry struct { Name string `json:"name"` Model string `json:"model"` - Size int64 `json:"size"` + Size *int64 `json:"size,omitempty"` Digest string `json:"digest"` Details OllamaModelDetails `json:"details"` ExpiresAt time.Time `json:"expires_at"` - SizeVRAM int64 `json:"size_vram"` + SizeVRAM *int64 `json:"size_vram,omitempty"` Capabilities []string `json:"capabilities,omitempty"` } diff --git a/docs/content/getting-started/models.md b/docs/content/getting-started/models.md index 9c6ce32f7..2130dfc58 100644 --- a/docs/content/getting-started/models.md +++ b/docs/content/getting-started/models.md @@ -391,6 +391,8 @@ See the [Model Configuration]({{% relref "advanced/model-configuration" %}}) gui ### List Installed Models +Ollama clients can list configured models with `GET /api/tags` and loaded models with `GET /api/ps`. Each entry includes `size` in bytes when LocalAI can resolve a non-empty primary weights file on disk. This is the size of that file, not the total size of a multi-file model or its memory use. Unknown sizes are omitted; `/api/ps` also omits `size_vram` because per-model VRAM use is not available. + ```bash # Via API curl http://localhost:8080/v1/models From 5b10d36b5f64ed609a185a1be0d38cdbf4f94254 Mon Sep 17 00:00:00 2001 From: Tai An Date: Sun, 27 Sep 2026 15:11:46 -0700 Subject: [PATCH 45/49] fix(quantization): pin the producing backend on imported quantized models (#11875) (#11879) * fix(quantization): pin the producing backend on imported quantized models (#11875) ImportModel hands the copied GGUF to importers.ImportLocalPath, which detects the file format and defaults every GGUF to `backend: llama-cpp`. For a model this service just produced with a backend stock llama.cpp cannot read, the generated config names an engine that cannot load the file, and the import silently registers an unloadable model. Correcting `backend:` by hand makes the same file work. The job record already carries the backend that served StartQuantization, so carry it into the config instead of keeping the detected default. The gallery publishes a quantizer as a release channel of the engine that runs its output ("llama-cpp-quantization" is llama.cpp's quantizer, whose GGUF is served by "llama-cpp"), so the channel suffix is stripped to get the serving backend. A backend that both quantizes and serves ("rocmfp4") carries no suffix and passes through unchanged, as do pinned hardware variants ("rocm-rocmfp4"), which are valid values for a config's backend field. An empty job backend leaves the detected default in place. Also replace the importer's generic "Fine-tuned model (GGUF)" description for this path: the model was quantized, not fine-tuned, and the job knows the type. Signed-off-by: Tai An * style: restore trailing newline in service.go for gofmt Signed-off-by: Anai Guo * style(quantization): restore trailing newline in service.go gofmt requires the file to end with a newline; the previous style commit did not actually add it. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Tai An Signed-off-by: Anai Guo Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- core/services/quantization/service.go | 26 ++++++++++++++++++++++ core/services/quantization/service_test.go | 22 ++++++++++++++++++ 2 files changed, 48 insertions(+) diff --git a/core/services/quantization/service.go b/core/services/quantization/service.go index a6a6eefcb..011543205 100644 --- a/core/services/quantization/service.go +++ b/core/services/quantization/service.go @@ -621,6 +621,21 @@ func sanitizeQuantModelName(s string) string { return strings.ToLower(s) } +// inferenceBackendFor returns the backend that can load what a quantization +// backend produced. +// +// The gallery publishes a quantizer as a release channel of the engine that +// runs its output: "llama-cpp-quantization" is llama.cpp's quantizer, and the +// GGUF it writes is served by "llama-cpp". The suffix is a channel marker and +// carries no engine information, so stripping it yields the backend to pin in +// the imported model's config. Names that carry no channel suffix (a backend +// that both quantizes and serves, such as "rocmfp4") are already the engine +// name and pass through unchanged, as do pinned hardware variants +// ("rocm-rocmfp4"), which are valid values for a config's `backend:`. +func inferenceBackendFor(quantBackend string) string { + return strings.TrimSuffix(config.NormalizeBackendName(quantBackend), "-quantization") +} + // ImportModel imports a quantized model into LocalAI asynchronously. func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID string, req schema.QuantizationImportRequest) (string, error) { s.mu.Lock() @@ -719,6 +734,17 @@ func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID str cfg.Name = modelName + // The importer detects the file format and defaults to llama-cpp for any + // GGUF. That is wrong for a model this service just quantized with a + // backend stock llama.cpp cannot read: the job knows which backend + // produced the file, so pin that one instead of the detected default. + if backend := inferenceBackendFor(job.Backend); backend != "" { + cfg.Backend = backend + } + if job.QuantizationType != "" { + cfg.Description = "Quantized model (" + job.QuantizationType + ", GGUF)" + } + // Write YAML config yamlData, err := yaml.Marshal(cfg) if err != nil { diff --git a/core/services/quantization/service_test.go b/core/services/quantization/service_test.go index ae862ffca..8af5e392c 100644 --- a/core/services/quantization/service_test.go +++ b/core/services/quantization/service_test.go @@ -329,6 +329,28 @@ var _ = Describe("QuantizationService", func() { }) }) + Describe("imported model backend", func() { + It("strips the quantization channel suffix so the config pins the serving engine", func() { + Expect(inferenceBackendFor("llama-cpp-quantization")).To(Equal("llama-cpp")) + }) + + It("leaves a backend that both quantizes and serves unchanged", func() { + Expect(inferenceBackendFor("rocmfp4")).To(Equal("rocmfp4")) + }) + + It("keeps a pinned hardware variant, which is a valid backend value", func() { + Expect(inferenceBackendFor("rocm-rocmfp4-quantization")).To(Equal("rocm-rocmfp4")) + }) + + It("normalizes dots the way gallery names are written", func() { + Expect(inferenceBackendFor("llama.cpp-quantization")).To(Equal("llama-cpp")) + }) + + It("returns empty for an unset backend so the detected default is kept", func() { + Expect(inferenceBackendFor("")).To(BeEmpty()) + }) + }) + Describe("compile-time adapter contract", func() { It("satisfies syncstate.Store for *distributed.QuantStore", func() { // Guards against drift between the adapter and the component interface; From 52a6d62bbcb504ced0423bd789911e4036014b80 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 00:11:51 +0200 Subject: [PATCH 46/49] fix: return correct HTTP status codes for saturation and no-nodes-available (#12113) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat: return 429 when backends are saturated When backends are at capacity (per-model max_concurrent or the process-wide --max-concurrent-backend-requests ceiling), the response was 503. The OpenAI SDK, litellm, and most agent harnesses key on 429 for rate-limit backoff and treat 503 as a hard error. Both saturation paths now return 429 with the existing Retry-After header and type: "rate_limit_error" in the JSON body. The per-model admission middleware keeps admission_rejected as the code field so existing alerts that match on it still fire. Non-saturation 503s are unchanged: model cold-loading (with progress body), model-load failure cooldown, PII detector fail-closed, and classifier unavailable. These mean "not ready" rather than "busy". Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto * fix: return 503 when scheduler has no available nodes When the scheduler cannot find any healthy node to serve a model — all nodes are full and eviction cannot free a slot, or a node_selector excludes every candidate — the error fell through to 500. A 500 tells clients something is broken when the condition is transient and retryable. The router now wraps these errors with a new ErrNoAvailableNodes sentinel. The HTTP error handler maps it to 503 via applyNoAvailableNodes, following the same pattern as applyBackendAdmission (429). Unrelated scheduler errors (DB timeouts, registry lookups) still return 500. Three return sites are wrapped: - resolveSelectorCandidates: selector matches zero healthy nodes - scheduleNewModel eviction-busy: all models have in-flight requests - scheduleNewModel eviction-failed: eviction itself errored The existing scheduleAndLoad wrapper ("no available nodes: %w") preserves the sentinel through the chain via errors.Is, as does ModelRouterAdapter. Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto * test(http): use Ginkgo for admission tests Replace forbidden testing.T calls with Ginkgo and Gomega so the lint check accepts the admission handler tests. Assisted-by: Codex:GPT-6 forbidigo * fix(middleware): show the recorded status for admission rejections The admission audit row now records 429, but the Middleware page still printed a hard-coded 503. Read the status from the event, and update the two package comments that still said 503. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/backend/global_admission.go | 5 +- core/http/admission_handler_test.go | 70 ++++++++++++++++++++ core/http/app.go | 24 ++++++- core/http/middleware/admission.go | 9 +-- core/http/middleware/admission_test.go | 4 +- core/http/react-ui/src/pages/Middleware.jsx | 3 +- core/services/nodes/router.go | 13 +++- core/services/nodes/router_test.go | 22 ++++++ core/services/routing/admission/admission.go | 2 +- core/services/routing/pii/types.go | 2 +- docs/content/reference/api-errors.md | 2 + docs/content/reference/cli-reference.md | 2 +- docs/content/reference/runtime-errors.md | 5 +- 13 files changed, 144 insertions(+), 19 deletions(-) create mode 100644 core/http/admission_handler_test.go diff --git a/core/backend/global_admission.go b/core/backend/global_admission.go index 9b216ea83..7ee69aeb1 100644 --- a/core/backend/global_admission.go +++ b/core/backend/global_admission.go @@ -11,8 +11,9 @@ import ( ) // BackendAdmissionError reports that the process-wide backend execution -// ceiling is full. HTTP callers map it to 503; internal callers receive the -// same typed error instead of silently queueing and growing in-flight state. +// ceiling is full. HTTP callers map it to 429 (Too Many Requests) with a +// Retry-After header; internal callers receive the same typed error instead +// of silently queueing and growing in-flight state. type BackendAdmissionError struct { Limit int RetryAfter time.Duration diff --git a/core/http/admission_handler_test.go b/core/http/admission_handler_test.go new file mode 100644 index 000000000..6a0a2201d --- /dev/null +++ b/core/http/admission_handler_test.go @@ -0,0 +1,70 @@ +package http + +import ( + "errors" + "fmt" + "net/http" + "net/http/httptest" + "time" + + "github.com/labstack/echo/v4" + corebackend "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/services/nodes" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Backend admission", func() { + It("maps BackendAdmissionError to 429 with Retry-After", func() { + e := echo.New() + req := httptest.NewRequest(http.MethodPost, "/", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + err := &corebackend.BackendAdmissionError{Limit: 4, RetryAfter: 3 * time.Second} + code := applyBackendAdmission(err, http.StatusInternalServerError, c) + + Expect(code).To(Equal(http.StatusTooManyRequests)) + Expect(rec.Header().Get("Retry-After")).To(Equal("3")) + }) + + It("passes through non-admission errors unchanged", func() { + e := echo.New() + req := httptest.NewRequest(http.MethodPost, "/", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + code := applyBackendAdmission(errors.New("some other error"), http.StatusInternalServerError, c) + Expect(code).To(Equal(http.StatusInternalServerError)) + Expect(rec.Header().Get("Retry-After")).To(BeEmpty()) + }) +}) + +var _ = Describe("No available nodes", func() { + It("maps ErrNoAvailableNodes to 503", func() { + // The scheduler wraps the sentinel in fmt.Errorf chains and via + // errors.Join — errors.Is must still find it. + wrapped := fmt.Errorf("routing model foo: %w", + fmt.Errorf("no available nodes: %w", + fmt.Errorf("no healthy nodes available: %w", + errors.Join(nodes.ErrEvictionBusy, nodes.ErrNoAvailableNodes)))) + + code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusServiceUnavailable)) + }) + + It("maps selector-mismatch chain to 503", func() { + wrapped := fmt.Errorf("routing model bar: %w", + fmt.Errorf("no available nodes: %w", + fmt.Errorf("no healthy nodes match selector for model bar: {\"gpu.vendor\":\"tpu\"}: %w", + nodes.ErrNoAvailableNodes))) + + code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusServiceUnavailable)) + }) + + It("passes through unrelated errors unchanged", func() { + code := applyNoAvailableNodes(errors.New("database timeout"), http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusInternalServerError)) + }) +}) diff --git a/core/http/app.go b/core/http/app.go index f03522a45..c94b323e4 100644 --- a/core/http/app.go +++ b/core/http/app.go @@ -85,7 +85,20 @@ func applyBackendAdmission(err error, code int, c echo.Context) int { return code } c.Response().Header().Set("Retry-After", strconv.Itoa(int(capacityErr.RetryAfter.Seconds()))) - return http.StatusServiceUnavailable + return http.StatusTooManyRequests +} + +// applyNoAvailableNodes maps scheduler "no available nodes" errors to 503. +// When the cluster has no healthy node to serve a model — all are full, a +// node selector excludes every candidate, or eviction could not free a slot — +// the request is retryable, not a server bug. Without this the error fell +// through to 500, which tells clients something is broken when they just +// need to wait for a node. +func applyNoAvailableNodes(err error, code int) int { + if errors.Is(err, nodes.ErrNoAvailableNodes) { + return http.StatusServiceUnavailable + } + return code } // respondModelLoading answers a request whose model is still cold-loading with @@ -208,6 +221,7 @@ func API(application *application.Application) (*echo.Echo, error) { } code = applyModelLoadCooldown(err, code, c) code = applyBackendAdmission(err, code, c) + code = applyNoAvailableNodes(err, code) // Handle 404 errors: serve React SPA for HTML requests, JSON otherwise if code == http.StatusNotFound { @@ -224,8 +238,13 @@ func API(application *application.Application) (*echo.Echo, error) { } // Send custom error page + errType := "" + var capErr *corebackend.BackendAdmissionError + if errors.As(err, &capErr) { + errType = "rate_limit_error" + } c.JSON(code, schema.ErrorResponse{ - Error: &schema.APIError{Message: err.Error(), Code: code}, + Error: &schema.APIError{Message: err.Error(), Code: code, Type: errType}, }) } } else { @@ -240,6 +259,7 @@ func API(application *application.Application) (*echo.Echo, error) { // Opaque errors deliberately withhold the body, so a still-loading // model gets the status and Retry-After but no progress detail. code = applyModelLoading(err, code, c) + code = applyNoAvailableNodes(err, code) c.NoContent(code) } } diff --git a/core/http/middleware/admission.go b/core/http/middleware/admission.go index c79066925..d6134b026 100644 --- a/core/http/middleware/admission.go +++ b/core/http/middleware/admission.go @@ -20,7 +20,7 @@ import ( // SERVED model — a router fanout that lands on a saturated downstream // model gets rejected even though the requested router-model has slack. // -// On reject: HTTP 503, Retry-After header, error JSON. An audit row +// On reject: HTTP 429, Retry-After header, error JSON. An audit row // goes into the shared event store under KindAdmission so admins see // rejection rates alongside PII and proxy events. // @@ -39,9 +39,10 @@ func AdmissionControl(limiter *admission.Limiter, events pii.EventStore) echo.Mi retryAfter := admission.RetryAfter(cfg.Limits.RetryAfterSeconds) recordAdmissionRejection(events, cfg.Name, retryAfter) c.Response().Header().Set("Retry-After", strconv.Itoa(int(retryAfter.Seconds()))) - return c.JSON(http.StatusServiceUnavailable, map[string]any{ + return c.JSON(http.StatusTooManyRequests, map[string]any{ "error": map[string]any{ - "type": "admission_rejected", + "type": "rate_limit_error", + "code": "admission_rejected", "message": fmt.Sprintf("model %q is at capacity (max_concurrent=%d); retry after %s", cfg.Name, max, retryAfter), }, }) @@ -61,7 +62,7 @@ func recordAdmissionRejection(events pii.EventStore, modelName string, retryAfte if events == nil { return } - statusCode := http.StatusServiceUnavailable + statusCode := http.StatusTooManyRequests durMS := retryAfter.Milliseconds() id := fmt.Sprintf("adm_%d_%s", admissionEventSeq.Add(1), randHex(4)) _ = events.Record(context.Background(), pii.PIIEvent{ diff --git a/core/http/middleware/admission_test.go b/core/http/middleware/admission_test.go index 841a2dd47..1e8649c2f 100644 --- a/core/http/middleware/admission_test.go +++ b/core/http/middleware/admission_test.go @@ -60,7 +60,7 @@ var _ = Describe("Admission", func() { It("rejects when full", func() { // Saturate the limiter outside the middleware, then a request - // at the same model gets 503 with a Retry-After header. + // at the same model gets 429 with a Retry-After header. lim := admission.New() release, ok := lim.Acquire("busy", 1) Expect(ok).To(BeTrue(), "setup acquire should succeed") @@ -75,7 +75,7 @@ var _ = Describe("Admission", func() { return c.String(http.StatusOK, "ok") }) Expect(err).NotTo(HaveOccurred()) - Expect(rec.Code).To(Equal(http.StatusServiceUnavailable)) + Expect(rec.Code).To(Equal(http.StatusTooManyRequests)) Expect(rec.Header().Get("Retry-After")).To(Equal("3")) Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects") Expect(rec.Body.String()).To(ContainSubstring("admission_rejected")) diff --git a/core/http/react-ui/src/pages/Middleware.jsx b/core/http/react-ui/src/pages/Middleware.jsx index 34d55d049..363b35f61 100644 --- a/core/http/react-ui/src/pages/Middleware.jsx +++ b/core/http/react-ui/src/pages/Middleware.jsx @@ -931,7 +931,8 @@ function eventDetails(e) { } case 'admission': { const retry = e.duration_ms != null ? `retry-after ${Math.round(e.duration_ms / 1000)}s` : '' - return `HTTP 503 rejected · ${retry}` + // Older audit rows were recorded as 503; newer ones as 429. + return `HTTP ${e.status_code || 429} rejected · ${retry}` } default: { const len = e.length != null ? `len ${e.length}` : '' diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go index d024b5621..36eee4a47 100644 --- a/core/services/nodes/router.go +++ b/core/services/nodes/router.go @@ -955,7 +955,7 @@ func (r *SmartRouter) resolveSelectorCandidates(ctx context.Context, modelID str return nil, fmt.Errorf("looking up nodes for selector %s: %w", sched.NodeSelector, err) } if len(candidates) == 0 { - return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s", modelID, sched.NodeSelector) + return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s: %w", modelID, sched.NodeSelector, ErrNoAvailableNodes) } return extractNodeIDs(candidates), nil } @@ -1167,9 +1167,9 @@ func (r *SmartRouter) scheduleNewModel(ctx context.Context, backendType, modelID evictedNode, evictErr := r.evictLRUAndFreeNodeFrom(ctx, candidateNodeIDs) if evictErr != nil { if errors.Is(evictErr, ErrEvictionBusy) { - return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", evictErr) + return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", errors.Join(evictErr, ErrNoAvailableNodes)) } - return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", evictErr) + return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", errors.Join(evictErr, ErrNoAvailableNodes)) } node = evictedNode } @@ -2059,6 +2059,13 @@ func (r *SmartRouter) EvictLRU(ctx context.Context, nodeID string) (string, erro // and none can be evicted to make room. var ErrEvictionBusy = errors.New("all models busy, cannot evict") +// ErrNoAvailableNodes is returned when the scheduler cannot find any healthy +// node to serve a model — all nodes are full and eviction cannot free a slot, +// or a node selector excludes every candidate. The HTTP layer maps this to +// 503 so clients treat it as a transient condition rather than a server bug +// (which is what 500 would imply). +var ErrNoAvailableNodes = errors.New("no available nodes") + // evictLRUAndFreeNode finds the globally least-recently-used model with zero in-flight, // unloads it, and returns its node for reuse. If all models are busy, retries briefly. // diff --git a/core/services/nodes/router_test.go b/core/services/nodes/router_test.go index 3735ec5e4..7577e2846 100644 --- a/core/services/nodes/router_test.go +++ b/core/services/nodes/router_test.go @@ -843,6 +843,26 @@ var _ = Describe("SmartRouter", func() { Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no available nodes")) }) + + It("wraps ErrNoAvailableNodes when all nodes are full and eviction cannot help", func() { + // gorm.ErrRecordNotFound is the registry's verdict that no node + // matches — the scheduler then falls through to eviction. With + // DB nil, eviction returns ErrEvictionBusy, and the scheduler + // wraps the error with ErrNoAvailableNodes so the HTTP layer can + // map it to 503 instead of 500. + reg.findIdleErr = errors.New("no idle") + reg.findLeastLoadedErr = gorm.ErrRecordNotFound + + router := NewSmartRouter(reg, SmartRouterOptions{ + Unloader: unloader, + ClientFactory: factory, + }) + + _, err := router.Route(context.Background(), "m5", "models/m5.gguf", "llama-cpp", "", nil, false) + Expect(err).To(HaveOccurred()) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) + Expect(errors.Is(err, ErrEvictionBusy)).To(BeTrue()) + }) }) Describe("UnloadModel (mock-based)", func() { @@ -970,6 +990,7 @@ var _ = Describe("SmartRouter", func() { _, err := router.Route(context.Background(), "aliased-model", "models/aliased.gguf", "llama-cpp", "", nil, false) Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector")) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) }) It("returns error when no nodes match selector", func() { @@ -988,6 +1009,7 @@ var _ = Describe("SmartRouter", func() { _, err := router.Route(context.Background(), "no-match-model", "models/nomatch.gguf", "llama-cpp", "", nil, false) Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector")) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) }) It("uses regular methods when model has no scheduling config", func() { diff --git a/core/services/routing/admission/admission.go b/core/services/routing/admission/admission.go index 168248181..f37273373 100644 --- a/core/services/routing/admission/admission.go +++ b/core/services/routing/admission/admission.go @@ -1,6 +1,6 @@ // Package admission is routing-module subsystem 5: per-model // concurrency control + audit. The middleware acquires a slot -// before the handler runs; on full, the request gets 503 with +// before the handler runs; on full, the request gets 429 with // Retry-After so clients back off rather than pile on. The audit // row goes into the shared event store alongside PII and proxy // rows so admins see a single timeline of routing pressure. diff --git a/core/services/routing/pii/types.go b/core/services/routing/pii/types.go index c2e2510df..ad15ea462 100644 --- a/core/services/routing/pii/types.go +++ b/core/services/routing/pii/types.go @@ -109,7 +109,7 @@ const ( // model's MaxConcurrent ceiling is full. The Host field carries // the model name (overloading the existing column rather than // adding a new one — admins read it as "the thing that was - // busy"); StatusCode is 503. + // busy"); StatusCode is 429. KindAdmission EventKind = "admission" ) diff --git a/docs/content/reference/api-errors.md b/docs/content/reference/api-errors.md index 9bd9dea02..20f63786f 100644 --- a/docs/content/reference/api-errors.md +++ b/docs/content/reference/api-errors.md @@ -88,7 +88,9 @@ The `/v1/responses` endpoint returns errors with this structure: | 404 | Not Found | Model or resource does not exist | | 409 | Conflict | Resource already exists (e.g., duplicate token) | | 422 | Unprocessable Entity | Validation failed (e.g., invalid parameter range) | +| 429 | Too Many Requests | All backends are saturated (per-model `max_concurrent` or process-wide `--max-concurrent-backend-requests` ceiling reached). Includes a `Retry-After` header and `type: "rate_limit_error"` so OpenAI-compatible clients and harnesses back off automatically | | 500 | Internal Server Error | Backend inference failure, unexpected server errors | +| 503 | Service Unavailable | No healthy node available to serve the model (cluster is full, eviction cannot free a slot, or a `node_selector` excludes all candidates). Also used during model-load cooldown and while a model is still cold-loading. Retryable | ## Global Error Handling diff --git a/docs/content/reference/cli-reference.md b/docs/content/reference/cli-reference.md index 8cf47e74a..73fa07722 100644 --- a/docs/content/reference/cli-reference.md +++ b/docs/content/reference/cli-reference.md @@ -95,7 +95,7 @@ For more information on VRAM management, see [VRAM and Memory Management]({{%rel | Parameter | Default | Description | Environment Variable | |-----------|---------|-------------|----------------------| | `--address` | `:8080` | Bind address for the API server | `$LOCALAI_ADDRESS`, `$ADDRESS` | -| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 503 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` | +| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 429 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` | | `--cors` | `false` | Enable CORS (Cross-Origin Resource Sharing) | `$LOCALAI_CORS`, `$CORS` | | `--cors-allow-origins` | | Comma-separated list of allowed CORS origins | `$LOCALAI_CORS_ALLOW_ORIGINS`, `$CORS_ALLOW_ORIGINS` | | `--disable-csrf` | `false` | Disable CSRF middleware (enabled by default) | `$LOCALAI_DISABLE_CSRF` | diff --git a/docs/content/reference/runtime-errors.md b/docs/content/reference/runtime-errors.md index 1e14dc0b5..fd5ef2bd8 100644 --- a/docs/content/reference/runtime-errors.md +++ b/docs/content/reference/runtime-errors.md @@ -21,8 +21,9 @@ The left column is the literal string as it appears in the LocalAI server log (o | `grpc service not ready` | The backend process was spawned but its gRPC server did not become healthy in time (slow start, crash on startup, or the process died while loading). When a local backend has already exited, the error includes its exit code and last stderr line. | Use the included stderr diagnostic when present; otherwise check the log lines just above. A crash here often means out of memory, a missing shared library, or an incompatible CPU (see `SIGILL`). Increase available RAM/VRAM or pick a smaller quantization. | | `failed to load model: ...` | Returned by the load endpoints and several feature paths (voice, realtime, audio transform) when the model config could not be resolved or the backend load failed. | Confirm the model name exists (`local-ai models list`) and its YAML is valid. The trailing text carries the specific reason. | | HTTP `503` with a `Retry-After` header, after a load failed | Model-load failure cooldown. After a model fails to load, LocalAI refuses new load attempts for that model for a short window so a client that keeps polling a broken model does not respawn a crashing backend on every request. The window starts at `--model-load-failure-cooldown` (default `10s`) and doubles per consecutive failure up to 5m; it resets on the first success. | Fix the underlying load failure (see the rows above), then wait out the `Retry-After` seconds before retrying, or restart LocalAI to clear the cooldown. Set `--model-load-failure-cooldown 0` (or `LOCALAI_MODEL_LOAD_FAILURE_COOLDOWN=0`) to disable the cooldown entirely. See {{% relref "reference/cli-reference" %}}. | -| HTTP `503` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `503` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. | -| HTTP `503` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. | +| HTTP `503` with `no available nodes` or `no healthy nodes match selector` | The scheduler could not find any healthy node to serve the model. All nodes are full and eviction cannot free a slot, or a `node_selector` in the model's scheduling config excludes every candidate. | Retry after a node becomes available or an in-flight request completes and frees a slot. In a cluster, add nodes or replicas. If a selector is set, confirm at least one healthy node matches it. | +| HTTP `429` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `429` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. | +| HTTP `429` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. | | `invalid pitch` (with CUDA) | The prompt exceeded the model's context size. | Reduce the prompt length, or raise the model's context size (`context_size:` in the model YAML). | | `SIGILL` (illegal instruction) on startup | The prebuilt backend binary uses CPU instructions your CPU does not have (for example AVX512, AVX2, F16C, FMA). | Rebuild the backend for your CPU. In a container, set `REBUILD=true` and disable the unsupported instructions, for example `CMAKE_ARGS="-DGGML_F16C=OFF -DGGML_AVX512=OFF -DGGML_AVX2=OFF -DGGML_FMA=OFF" make build`. | | CUDA / VRAM out of memory (backend log shows `out of memory`, `CUDA error: out of memory`, or the process is killed loading) | The model plus its KV cache does not fit in GPU memory. | Use a smaller quantization, reduce `context_size:`, offload fewer layers to the GPU (lower `gpu_layers:`), or free VRAM held by other processes. On multi-GPU hosts, confirm the model is not trying to load entirely onto one device. | From 950c271710297c0d959dcf936eb1ee407309f38f Mon Sep 17 00:00:00 2001 From: Stefan Walcz Date: Mon, 28 Sep 2026 00:52:47 +0200 Subject: [PATCH 47/49] [router] fix: re-seed the knn corpus index when the vector store comes back empty (#12267) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fix(router): re-seed the knn corpus index when the vector store comes back empty The corpus manager records a store as synced by file fingerprint and embedding fingerprint. The local-store backend behind it is an in-memory gRPC process the model loader may evict (active-backend cap, memory pressure) or the idle watchdog may kill, and relaunch on the next request — empty. The file is unchanged, so EnsureLoaded returned early and the router went blind: every probe fell back with similarity 0 while corpus/stats kept reporting the full count. Measured on a production router (LOCALAI_MAX_ACTIVE_BACKENDS=6, four resident models + two router stores): loading any further backend evicted a store, and the idle watchdog killed both after 15 minutes; /stores/find returned 0 hits against a 100-line corpus file whose stored vectors matched fresh embeddings with cosine 1.000. Two parts, because the knn classifier is built once and cached (GetOrBuildClassifier), so the sync at build time is otherwise the only one for the process lifetime: - corpus.Manager remembers one vector it inserted (probe) and, on the synced path, asks the live index for it. A miss means the index was relaunched — fall through and re-seed from the file (no re-embedding). - The router middleware wraps the knn classifier's store so every lookup runs EnsureLoaded first; the loader gets the raw store, so its probe never re-enters the wrapper. A sync error fails the lookup closed, like the build-time load. Specs: corpus package (relaunched empty store is re-seeded under an unchanged file), middleware (relaunched index behind the cached classifier is re-seeded instead of falling back; the spec is red without the wrapper). The test fake now answers Search for inserted vectors. Folds in the maintainer's follow-up (router-corpus-reseed-after-store-relaunch): reviewed and accepted. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz --- core/http/middleware/route_model.go | 38 ++++++++++++- core/http/middleware/route_model_test.go | 60 ++++++++++++++++++++ core/services/routing/corpus/manager.go | 50 +++++++++++++++- core/services/routing/corpus/manager_test.go | 52 +++++++++++++++-- docs/content/operations/middleware.md | 7 ++- 5 files changed, 198 insertions(+), 9 deletions(-) diff --git a/core/http/middleware/route_model.go b/core/http/middleware/route_model.go index dc01ac931..0cd5f23f4 100644 --- a/core/http/middleware/route_model.go +++ b/core/http/middleware/route_model.go @@ -65,6 +65,31 @@ type CorpusLoader interface { EnsureLoaded(ctx context.Context, storeName, embeddingModel, embeddingFingerprint string, embedder backend.Embedder, store backend.VectorStore) (int, error) } +// reseedingVectorStore runs the corpus sync before every lookup. It +// wraps the RAW store and hands that raw store to the loader, so the +// loader's own probe lookup never re-enters this wrapper. A sync error +// fails the lookup closed, like the build-time load does: a decision +// taken on an index that could not be synced is exactly the blind- +// router bug this guards against. +type reseedingVectorStore struct { + backend.VectorStore + ensure func(ctx context.Context) error +} + +func (s *reseedingVectorStore) SearchK(ctx context.Context, vec []float32, k int) ([]backend.Neighbor, error) { + if err := s.ensure(ctx); err != nil { + return nil, fmt.Errorf("router: knn corpus sync before lookup: %w", err) + } + return s.VectorStore.SearchK(ctx, vec, k) +} + +func (s *reseedingVectorStore) Search(ctx context.Context, vec []float32) (float64, []byte, bool, error) { + if err := s.ensure(ctx); err != nil { + return 0, nil, false, fmt.Errorf("router: knn corpus sync before lookup: %w", err) + } + return s.VectorStore.Search(ctx, vec) +} + // ClassifierDeps bundles the backend factories the router middleware // needs to build a classifier and its optional L2 cache. Bundled into // one struct because RouteModel already takes many positional @@ -478,12 +503,23 @@ func buildClassifier(cfg *config.ModelConfig, deps ClassifierDeps) (router.Class // Loading fails closed: a live index from a different embedding // space may have the same vector width and return plausible but // incorrect routes. - if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, vstore); err != nil { + raw := vstore + if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw); err != nil { return nil, fmt.Errorf("router classifier knn: load corpus %q: %w", storeName, err) } else if n > 0 { xlog.Info("router: knn corpus loaded", "router_model", cfg.Name, "store", storeName, "entries", n) } + // The classifier built below is cached for the process lifetime + // (GetOrBuildClassifier), so this sync would otherwise be the + // only one — while the local-store process behind the index can + // be evicted or idle-killed and relaunched EMPTY at any later + // request. Re-check on every lookup; the corpus loader probes + // the live index and re-seeds it from the file on a miss. + vstore = &reseedingVectorStore{VectorStore: raw, ensure: func(ctx context.Context) error { + _, err := deps.Corpus.EnsureLoaded(ctx, storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw) + return err + }} } knnClassifier := router.NewKNNClassifier(embedder, vstore, router.KNNClassifierOptions{ K: rc.KNN.K, diff --git a/core/http/middleware/route_model_test.go b/core/http/middleware/route_model_test.go index 97aa58580..524994957 100644 --- a/core/http/middleware/route_model_test.go +++ b/core/http/middleware/route_model_test.go @@ -581,6 +581,29 @@ var ( errTestKNNInsert = errors.New("knn classifier must never insert into the corpus") ) +// reseedingCorpusLoader mirrors corpus.Manager's contract: EnsureLoaded +// is a no-op while the index still answers, and re-seeds it when the +// store came back empty. It insists on the RAW scripted store, so a +// wrapper leaking into the loader (and recursing) fails the spec. +type reseedingCorpusLoader struct { + seed []backend.Neighbor + calls, reseeds int +} + +func (r *reseedingCorpusLoader) EnsureLoaded(_ context.Context, _, _, _ string, _ backend.Embedder, store backend.VectorStore) (int, error) { + r.calls++ + s, ok := store.(*scriptedVectorStore) + if !ok { + return 0, errors.New("corpus loader must receive the raw store, not a wrapper") + } + if len(s.neighbors) == 0 { + s.neighbors = r.seed + r.reseeds++ + return len(r.seed), nil + } + return 0, nil +} + type failingCorpusLoader struct{ err error } func (f failingCorpusLoader) EnsureLoaded(context.Context, string, string, string, backend.Embedder, backend.VectorStore) (int, error) { @@ -710,6 +733,43 @@ var _ = Describe("RouteModel middleware (knn classifier)", func() { Expect(err.Error()).To(ContainSubstring("knn")) }) + It("re-seeds a relaunched corpus index behind the cached classifier", func() { + // The classifier is built once and cached; the local-store + // process behind its index may be evicted or idle-killed and + // relaunched empty afterwards. Measured in production: every probe + // then fell back with similarity 0 while corpus/stats still + // reported the full count. The lookup path must re-seed. + routerCfg := newKNNRouterModel(modelDir, "smart-router") + writeCandidate(modelDir, "small-model") + writeCandidate(modelDir, "big-model") + seeded := []backend.Neighbor{ + {Similarity: 0.92, Payload: corpusPayload("code-generation")}, + {Similarity: 0.88, Payload: corpusPayload("code-generation")}, + } + vstore.neighbors = seeded + corpus := &reseedingCorpusLoader{seed: seeded} + deps := knnDeps() + deps.EmbedderFingerprint = func(string) (string, error) { return "fp", nil } + deps.Corpus = corpus + registry := router.NewRegistry() + + first, err := GetOrBuildClassifier(registry, routerCfg, deps) + Expect(err).NotTo(HaveOccurred()) + d, err := first.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"}) + Expect(err).NotTo(HaveOccurred()) + Expect(d.Labels).To(ContainElement("code-generation")) + Expect(corpus.reseeds).To(Equal(0), "a healthy index is not re-seeded") + + vstore.neighbors = nil // the store process was relaunched empty + again, err := GetOrBuildClassifier(registry, routerCfg, deps) + Expect(err).NotTo(HaveOccurred()) + Expect(again).To(BeIdenticalTo(first), "the classifier stays cached — the sync must live on the lookup path") + d, err = again.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"}) + Expect(err).NotTo(HaveOccurred()) + Expect(d.Labels).To(ContainElement("code-generation"), "lookup on a relaunched index re-seeds instead of falling back") + Expect(corpus.reseeds).To(Equal(1)) + }) + It("fails closed when the persisted corpus cannot sync into the live index", func() { routerCfg := newKNNRouterModel(modelDir, "smart-router") writeCandidate(modelDir, "small-model") diff --git a/core/services/routing/corpus/manager.go b/core/services/routing/corpus/manager.go index bac177011..6d3d92766 100644 --- a/core/services/routing/corpus/manager.go +++ b/core/services/routing/corpus/manager.go @@ -30,9 +30,10 @@ import ( "sync" "time" + "github.com/mudler/xlog" + "github.com/mudler/LocalAI/core/backend" "github.com/mudler/LocalAI/core/services/routing/router" - "github.com/mudler/xlog" ) // Entry is one labelled exemplar. Vector, EmbeddingModel, and @@ -104,6 +105,9 @@ type storeState struct { syncedFile fileFingerprint needsSync bool indexedEntries int + // probe is a vector we inserted ourselves; storeHolds uses it to + // tell a live index apart from a relaunched, empty one. + probe []float32 } type cachedStats struct { @@ -172,7 +176,17 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e } delete(m.states, storeName) } else if !state.needsSync && state.syncedFile.equal(fileKey) { - return 0, nil + if state.indexedEntries == 0 || store == nil || storeHolds(ctx, store, state.probe) { + return 0, nil + } + // The file is unchanged but the live index no longer answers + // for a vector we inserted: the store backend was relaunched + // (evicted by the active-backend cap or memory pressure, then + // started fresh and empty on this request). Fall through and + // re-seed it from the file — no re-embedding, the vectors are + // persisted. + xlog.Warn("router: knn corpus index came back empty, re-seeding from file", + "store", storeName, "entries", state.indexedEntries) } } @@ -224,10 +238,40 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e embeddingFingerprint: embeddingFingerprint, syncedFile: fileKey, indexedEntries: len(entries), + probe: entries[0].Vector, } return len(entries), nil } +// storeHolds reports whether the live vector index still contains the +// corpus. The local-store backend is an in-memory gRPC process: when +// the model loader evicts it (active-backend cap, memory pressure) and +// relaunches it on the next request, it comes back EMPTY while the +// manager still records the file as synced — from then on every probe +// routes to the fallback with similarity 0, and corpus/stats keeps +// reporting the full count because it reads the file. One nearest- +// neighbour lookup with a vector we inserted ourselves tells the two +// states apart. An index that cannot answer is treated as empty; the +// re-seed that follows surfaces the real error. +func storeHolds(ctx context.Context, store backend.VectorStore, probe []float32) bool { + if len(probe) == 0 { + return true + } + sim, _, ok, err := store.Search(ctx, probe) + return err == nil && ok && sim > 0.999 +} + +// firstVector returns the vector of the first entry across lists — +// the probe storeHolds checks the live index with. +func firstVector(lists ...[]Entry) []float32 { + for _, l := range lists { + if len(l) > 0 { + return l[0].Vector + } + } + return nil +} + // Add validates, embeds, persists, and indexes new exemplars. Entries // whose text is already in the corpus are skipped (an exemplar's // labels are corrected via Clear + reseed, not silent overwrite). @@ -288,6 +332,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF embeddingFingerprint: embeddingFingerprint, syncedFile: m.fingerprint(storeName), indexedEntries: len(existing), + probe: firstVector(existing), } } seen := make(map[string]struct{}, len(existing)) @@ -377,6 +422,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF embeddingFingerprint: embeddingFingerprint, syncedFile: m.fingerprint(storeName), indexedEntries: entryCount, + probe: firstVector(current, added), } } else { m.states[storeName] = storeState{embeddingFingerprint: embeddingFingerprint, needsSync: true, indexedEntries: entryCount} diff --git a/core/services/routing/corpus/manager_test.go b/core/services/routing/corpus/manager_test.go index 6c50c25a6..cd99322f2 100644 --- a/core/services/routing/corpus/manager_test.go +++ b/core/services/routing/corpus/manager_test.go @@ -31,17 +31,34 @@ func (e *countingEmbedder) Embed(_ context.Context, text string) ([]float32, err return []float32{float32(len(text)), e.model}, nil } -// capturingStore records index mutations. Search/SearchK are -// irrelevant to the manager and return clean misses. +// capturingStore records index mutations. Search answers like a live +// index for vectors that were inserted (the manager probes with one of +// its own vectors to detect a relaunched, empty store); SearchK is +// irrelevant to the manager and returns a clean miss. type capturingStore struct { mu sync.Mutex + vecs [][]float32 payloads [][]byte batches int deleted [][]float32 failBatches int } -func (s *capturingStore) Search(_ context.Context, _ []float32) (float64, []byte, bool, error) { +func (s *capturingStore) Search(_ context.Context, vec []float32) (float64, []byte, bool, error) { + s.mu.Lock() + defer s.mu.Unlock() + for i, v := range s.vecs { + if len(v) == len(vec) && func() bool { + for j := range v { + if v[j] != vec[j] { + return false + } + } + return true + }() { + return 1, s.payloads[i], true, nil + } + } return 0, nil, false, nil } @@ -49,9 +66,10 @@ func (s *capturingStore) SearchK(_ context.Context, _ []float32, _ int) ([]backe return nil, nil } -func (s *capturingStore) Insert(_ context.Context, _ []float32, payload []byte) error { +func (s *capturingStore) Insert(_ context.Context, vec []float32, payload []byte) error { s.mu.Lock() defer s.mu.Unlock() + s.vecs = append(s.vecs, vec) s.payloads = append(s.payloads, payload) return nil } @@ -64,8 +82,8 @@ func (s *capturingStore) InsertBatch(_ context.Context, vecs [][]float32, payloa s.failBatches-- return errors.New("transient batch failure") } + s.vecs = append(s.vecs, vecs...) s.payloads = append(s.payloads, payloads...) - _ = vecs return nil } @@ -117,6 +135,30 @@ var _ = Describe("corpus.Manager", func() { _ = os.RemoveAll(dir) }) + It("re-seeds the index when the store comes back empty under an unchanged file", func() { + // The local-store backend is an in-memory process the model loader + // may evict (active-backend cap) and relaunch empty on the next + // request. The file is untouched, so the file fingerprint alone + // says "synced" — and the router goes blind: every probe falls back + // with similarity 0 while corpus/stats still reports the full count. + _, _, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed) + Expect(err).NotTo(HaveOccurred()) + n, err := mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, store) + Expect(err).NotTo(HaveOccurred()) + Expect(n).To(Equal(0), "live index holds the corpus: nothing to do") + + relaunched := &capturingStore{} + n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched) + Expect(err).NotTo(HaveOccurred()) + Expect(n).To(Equal(len(seed)), "empty index under an unchanged file is re-seeded") + Expect(relaunched.payloads).To(HaveLen(len(seed))) + Expect(embedder.calls).To(Equal(len(seed)), "vectors come from the file, nothing is re-embedded") + + n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched) + Expect(err).NotTo(HaveOccurred()) + Expect(n).To(Equal(0), "and the relaunched index counts as synced again") + }) + It("adds entries: embeds, persists, and indexes them", func() { added, skipped, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed) Expect(err).NotTo(HaveOccurred()) diff --git a/docs/content/operations/middleware.md b/docs/content/operations/middleware.md index 43ab02dc1..ac97a10cd 100644 --- a/docs/content/operations/middleware.md +++ b/docs/content/operations/middleware.md @@ -558,7 +558,12 @@ The corpus is persisted as one JSONL file per router under `/router-corpus/` (text, labels, vector, embedding-model name, and embedding fingerprint) — **the file is the source of truth** and survives restarts; the local-store index is rebuilt from it at classifier -build time without re-embedding. The fingerprint follows the effective +build time without re-embedding. Before each KNN lookup, LocalAI checks a stored +vector against the live index. If the store restarts empty after eviction or +an idle timeout, LocalAI restores the index from the file without re-embedding. +A synchronization error fails the lookup. + +The fingerprint follows the effective embedding-model config and local artifact identity, so changing the model or replacing its local weights re-embeds the corpus on the next process load. For remote embedding services whose weights can change invisibly, bump From eec532704cf5a1a0365c1cb30394172677a05bcb Mon Sep 17 00:00:00 2001 From: Tai An Date: Sun, 27 Sep 2026 15:52:51 -0700 Subject: [PATCH 48/49] fix(functions): honor function_arguments_key when building the tool grammar (#11677) * fix(functions): honor function_arguments_key when building the tool grammar All four call sites of `Functions.ToJSONStructure(name, args string)` pass `FunctionsConfig.FunctionNameKey` as *both* arguments, so `FunctionArgumentsKey` never reaches the grammar generator. `ToJSONStructure` writes the two properties into the same map: property[nameKey] = FunctionName{Const: function.Name} property[argsKey] = Argument{...} When `nameKey == argsKey` the second assignment overwrites the first, so a model configured with `function_name_key` gets a grammar carrying only the arguments object -- the `{"const": ""}` constraint is gone and the grammar can no longer express which function was called. With `function_name_key: function`, the generated property set collapses from {"function": {"const": "get_weather"}, "arguments": {...}} to {"function": {"type": "object", "properties": {...}}} Setting only `function_arguments_key` is equally broken in the other direction: the grammar keeps emitting `arguments` while `ParseFunctionCall` (pkg/functions/parse.go) looks up the configured key, so the parsed call comes back with its arguments empty. The default configuration is unaffected -- with both keys empty `ToJSONStructure` falls back to `name`/`arguments` for both parameters, which is why this went unnoticed. The existing `ToJSONStructure()` unit test already calls the helper with two distinct keys, so only the call sites were wrong. Extend that test with a case that keeps both custom keys distinct and asserts the two properties survive. Signed-off-by: Anai-Guo * test(functions): cover configured grammar keys Route grammar construction through FunctionsConfig so the regression test covers the key wiring used by every endpoint. Assisted-by: Codex:gpt-5 * chore: empty commit to trigger workflow approval Signed-off-by: Tai An --------- Signed-off-by: Anai-Guo Signed-off-by: Tai An Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/http/endpoints/openai/chat.go | 2 +- core/http/endpoints/openai/realtime_model.go | 2 +- .../http/endpoints/openresponses/responses.go | 2 +- .../http/endpoints/openresponses/websocket.go | 2 +- pkg/functions/functions.go | 5 ++++ pkg/functions/functions_test.go | 27 +++++++++++++++++++ 6 files changed, 36 insertions(+), 4 deletions(-) diff --git a/core/http/endpoints/openai/chat.go b/core/http/endpoints/openai/chat.go index fbf5924ea..76efbca28 100644 --- a/core/http/endpoints/openai/chat.go +++ b/core/http/endpoints/openai/chat.go @@ -451,7 +451,7 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator } // Update input grammar or json_schema based on use_llama_grammar option - jsStruct := funcs.ToJSONStructure(config.FunctionsConfig.FunctionNameKey, config.FunctionsConfig.FunctionNameKey) + jsStruct := config.FunctionsConfig.ToJSONStructure(funcs) g, err := jsStruct.Grammar(config.FunctionsConfig.GrammarOptions()...) if err == nil { config.Grammar = g diff --git a/core/http/endpoints/openai/realtime_model.go b/core/http/endpoints/openai/realtime_model.go index b525eee26..a02696a1f 100644 --- a/core/http/endpoints/openai/realtime_model.go +++ b/core/http/endpoints/openai/realtime_model.go @@ -293,7 +293,7 @@ func (m *wrappedModel) Predict(ctx context.Context, messages schema.Messages, im } // Generate grammar from function definitions - jsStruct := functions.Functions(funcs).ToJSONStructure(turnCfg.FunctionsConfig.FunctionNameKey, turnCfg.FunctionsConfig.FunctionNameKey) + jsStruct := turnCfg.FunctionsConfig.ToJSONStructure(functions.Functions(funcs)) g, err := jsStruct.Grammar(turnCfg.FunctionsConfig.GrammarOptions()...) if err == nil { turnCfg.Grammar = g diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index d62fa7534..528737273 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -204,7 +204,7 @@ func ResponsesEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, eval } // Generate grammar to constrain model output to valid function calls - jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey) + jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction) g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...) if err == nil { cfg.Grammar = g diff --git a/core/http/endpoints/openresponses/websocket.go b/core/http/endpoints/openresponses/websocket.go index 3a92f275a..f0cb4b331 100644 --- a/core/http/endpoints/openresponses/websocket.go +++ b/core/http/endpoints/openresponses/websocket.go @@ -381,7 +381,7 @@ func handleWSResponseCreate(connCtx context.Context, conn *lockedConn, connectio funcsWithNoAction = funcsWithNoAction.Select(cfg.FunctionToCall()) } - jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey) + jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction) g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...) if err == nil { cfg.Grammar = g diff --git a/pkg/functions/functions.go b/pkg/functions/functions.go index 0686e4f9d..3e604d9df 100644 --- a/pkg/functions/functions.go +++ b/pkg/functions/functions.go @@ -89,6 +89,11 @@ func (f Functions) ToJSONStructure(name, args string) JSONFunctionStructure { return js } +// ToJSONStructure converts functions using the configured property keys. +func (c FunctionsConfig) ToJSONStructure(functions Functions) JSONFunctionStructure { + return functions.ToJSONStructure(c.FunctionNameKey, c.FunctionArgumentsKey) +} + // Select returns a list of functions containing the function with the given name func (f Functions) Select(name string) Functions { var funcs Functions diff --git a/pkg/functions/functions_test.go b/pkg/functions/functions_test.go index e0952c13f..7c10da473 100644 --- a/pkg/functions/functions_test.go +++ b/pkg/functions/functions_test.go @@ -65,6 +65,33 @@ var _ = Describe("LocalAI grammar functions", func() { Expect(fnName.Const).To(Equal("search")) Expect(fnArgs.Properties["query"].(map[string]any)["type"]).To(Equal("string")) }) + + It("keeps the name and the arguments in separate properties when both keys are customized", func() { + var functions Functions = []Function{ + { + Name: "get_weather", + Parameters: map[string]any{ + "properties": map[string]any{ + "city": map[string]any{ + "type": "string", + }, + }, + }, + }, + } + + config := FunctionsConfig{ + FunctionNameKey: "function", + FunctionArgumentsKey: "parameters", + } + js := config.ToJSONStructure(functions) + Expect(js.OneOf[0].Properties).To(HaveLen(2)) + + fnName := js.OneOf[0].Properties["function"].(FunctionName) + fnArgs := js.OneOf[0].Properties["parameters"].(Argument) + Expect(fnName.Const).To(Equal("get_weather")) + Expect(fnArgs.Properties["city"].(map[string]any)["type"]).To(Equal("string")) + }) }) Context("Select()", func() { It("selects one of the functions and returns a list containing only the selected one", func() { From f82efdb43bea49fa69c8109fe0458c357cf5b1dd Mon Sep 17 00:00:00 2001 From: PINYO PATTANAWASANPORN Date: Mon, 28 Sep 2026 05:52:56 +0700 Subject: [PATCH 49/49] fix(models): fallback to application config default context size in /v1/models/capabilities (#12202) (#12216) * fix(models): fallback to application config default context size (#12202) Honor appConfig.ContextSize in /v1/models/capabilities when model context_size is unset. * docs(models): explain context size fallback Describe the application default used by capability discovery and preserve the distinction between total context and per-request limits. Assisted-by: Codex:GPT-6 * fix(models): apply the default context size only when context_size is unset The request path applies the application default context size only when a model leaves context_size unset. An explicit 0 or -1 falls through to the backend fallback. The capabilities endpoint now does the same, so it reports the value the backend uses. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Co-authored-by: Ettore Di Giacinto --- .../endpoints/openai/list_capabilities.go | 6 +++++ .../openai/list_capabilities_test.go | 27 +++++++++++++++++++ docs/content/features/api-discovery.md | 5 ++++ 3 files changed, 38 insertions(+) diff --git a/core/http/endpoints/openai/list_capabilities.go b/core/http/endpoints/openai/list_capabilities.go index 72a645ef6..27583385a 100644 --- a/core/http/endpoints/openai/list_capabilities.go +++ b/core/http/endpoints/openai/list_capabilities.go @@ -36,6 +36,12 @@ func ListModelCapabilitiesEndpoint(bcl *config.ModelConfigLoader, ml *model.Mode for _, m := range modelNames { entry := schema.ModelCapabilities{ID: m, Object: "model"} if cfg, ok := modelConfigFor(bcl, m); ok { + // Mirror the request path: SetDefaults applies the application + // default only when the model leaves context_size unset. An + // explicit 0 or -1 falls through to the backend fallback there. + if cfg.ContextSize == nil && appConfig != nil && appConfig.ContextSize > 0 { + cfg.ContextSize = &appConfig.ContextSize + } entry.Capabilities = cfg.Capabilities() entry.ThreeDOperations = cfg.ThreeDOperations() entry.InputModalities = cfg.InputModalities() diff --git a/core/http/endpoints/openai/list_capabilities_test.go b/core/http/endpoints/openai/list_capabilities_test.go index 0c8a5d4ff..ecfc85f81 100644 --- a/core/http/endpoints/openai/list_capabilities_test.go +++ b/core/http/endpoints/openai/list_capabilities_test.go @@ -160,6 +160,33 @@ parameters: Expect(entry).NotTo(BeNil()) Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize)) }) + + It("uses application config context size when model context_size is unset", func() { + writeConfig("llm-app-default", ` +name: llm-app-default +backend: llama-cpp +parameters: + model: model.gguf +`) + appConf.ContextSize = 8192 + entry := entryFor(call(), "llm-app-default") + Expect(entry).NotTo(BeNil()) + Expect(entry.ContextSize).To(Equal(8192)) + }) + + It("keeps the backend fallback when the model sets a non-positive context_size", func() { + writeConfig("llm-explicit-zero", ` +name: llm-explicit-zero +backend: llama-cpp +context_size: 0 +parameters: + model: model.gguf +`) + appConf.ContextSize = 8192 + entry := entryFor(call(), "llm-explicit-zero") + Expect(entry).NotTo(BeNil()) + Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize)) + }) It("reports an alias with its target's capabilities and context_size", func() { writeConfig("real-llm", ` name: real-llm diff --git a/docs/content/features/api-discovery.md b/docs/content/features/api-discovery.md index f45524423..932634a39 100644 --- a/docs/content/features/api-discovery.md +++ b/docs/content/features/api-discovery.md @@ -141,6 +141,11 @@ curl http://localhost:8080/api/instructions/config-management?format=json An additive, LocalAI-specific superset of `/v1/models`. It returns the same set of models but enriches each entry with the **capabilities** the model supports and the **input/output modalities** it accepts and produces. Use it to decide, before sending a request, whether a given model can take an image, audio, or video attachment directly - or whether the input needs converting/transcribing first. +The reported `context_size` uses a positive model-level value first. +If the model does not set `context_size`, it uses **Settings → Performance → Default Context Size** when positive. +Otherwise, it uses the backend fallback of 4096 tokens. +For llama.cpp with separate KV caches, the reported value accounts for the number of parallel slots. + Because it is purely additive, clients that only understand `/v1/models` keep working unchanged; they simply never call this route. ```bash