From 4eff44a03e12fb1c4a809e95a545bdba7fca5138 Mon Sep 17 00:00:00 2001 From: yzxcj797 <1784931579@qq.com> Date: Sun, 16 Aug 2026 11:07:11 +0800 Subject: [PATCH 01/42] docs: replace dead chatbot-ui example link with repo root --- docs/content/advanced/advanced-usage.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md index f7d9546cf..7850151ae 100644 --- a/docs/content/advanced/advanced-usage.md +++ b/docs/content/advanced/advanced-usage.md @@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master ``` -See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files. +See also [chatbot-ui](https://github.com/mudler/LocalAI-examples) as an example on how to use config files. ### Prompt templates From 6393efc12b87a00e9273dd76dd7404fb58f0ce0e Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:57:53 +0000 Subject: [PATCH 02/42] chore(model-gallery): propose variant groupings Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index 04fa669b6..ee86c092a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -44011,6 +44011,12 @@ sha256: "" uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors - name: kimodo-soma-rp + variants: + - model: kimodo-soma-rp-bf16 + - model: kimodo-soma-rp-q4_k + - model: kimodo-soma-rp-q4_k_m + - model: kimodo-soma-rp-q5_k + - model: kimodo-soma-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44198,6 +44204,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-soma-seed + variants: + - model: kimodo-soma-seed-bf16 + - model: kimodo-soma-seed-q4_k + - model: kimodo-soma-seed-q4_k_m + - model: kimodo-soma-seed-q5_k + - model: kimodo-soma-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44385,6 +44397,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-rp + variants: + - model: kimodo-g1-rp-bf16 + - model: kimodo-g1-rp-q4_k + - model: kimodo-g1-rp-q4_k_m + - model: kimodo-g1-rp-q5_k + - model: kimodo-g1-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -44572,6 +44590,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-seed + variants: + - model: kimodo-g1-seed-bf16 + - model: kimodo-g1-seed-q4_k + - model: kimodo-g1-seed-q4_k_m + - model: kimodo-g1-seed-q5_k + - model: kimodo-g1-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: From e17039fb30972e00287e7a5ad58c2fb550c7698b Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:49:43 +0000 Subject: [PATCH 03/42] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 72 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 72 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..929c337bb 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,76 @@ --- +- name: "swift-qwen3.8-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF + description: | + Website  •  + Learn more  •  + GGUF  •  + Enterprise licensing + + # Swift-Qwen3.8-27B + + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, + using **58.3% fewer thinking tokens** while maintaining near-identical performance + (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. + + The prompt is a sample from LiveCodeBench v6 + + ## Training approach + + We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s + reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. + + Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. + + For maximum gains, Swift also includes a transfer component derived from + BottleCap AI's ThinkingCap-Qwen3.6-27B. + + ## Evaluation scope + + > All results below compare the Qwen3.8-27B BF16 base with the same base plus the + > Swift adapter. + + ## Benchmarks + + ... + license: "other" + tags: + - llm + - gguf + - reasoning + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + min_p: 0 + model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + presence_penalty: 1.5 + repeat_penalty: 1 + temperature: 0.7 + top_k: 20 + top_p: 0.8 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf + - filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From 3c9047654e3d12aecd9467245902446912e56faf Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:58:03 +0000 Subject: [PATCH 04/42] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 46 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..cceda1b77 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,50 @@ --- +- name: "ternary-bonsai-2-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + description: | + # Qwen3.8-27B + + > [!Note] + > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. + > + > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. + + > [!Tip] + > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. + > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. + + Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. + + ... + license: "apache-2.0" + tags: + - llm + - gguf + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf + - filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From e2b617104d17f0b5a949291fd90ea68e948110af Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sat, 26 Sep 2026 17:19:13 +0200 Subject: [PATCH 05/42] chore: :arrow_up: Update leejet/stable-diffusion.cpp to `2f886889e6e8b78738d6b87f7191f6018557c551` (#12274) * :arrow_up: Update leejet/stable-diffusion.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(stablediffusion-ggml): adapt to upstream tiling struct rename Upstream commit 2f88688 renamed the sd_tiling_params_t fields from tile_size_x/y to tile_size_w/h and rel_size_x/y to rel_size_w/h. Update the gosd.cpp wrappers to match so the C++ backend compiles. --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: Ettore Di Giacinto --- backend/go/stablediffusion-ggml/Makefile | 2 +- backend/go/stablediffusion-ggml/cpp/gosd.cpp | 12 ++++++------ backend/go/stablediffusion-ggml/cpp/gosd.h | 4 ++-- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/backend/go/stablediffusion-ggml/Makefile b/backend/go/stablediffusion-ggml/Makefile index 15ffb569e..6c5e70484 100644 --- a/backend/go/stablediffusion-ggml/Makefile +++ b/backend/go/stablediffusion-ggml/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # stablediffusion.cpp (ggml) STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp -STABLEDIFFUSION_GGML_VERSION?=b167b942f77ecb17e7f78e163a8c32ff7ac95c10 +STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551 CMAKE_ARGS+=-DGGML_MAX_NAME=128 diff --git a/backend/go/stablediffusion-ggml/cpp/gosd.cpp b/backend/go/stablediffusion-ggml/cpp/gosd.cpp index 4a1911015..74b2a0387 100644 --- a/backend/go/stablediffusion-ggml/cpp/gosd.cpp +++ b/backend/go/stablediffusion-ggml/cpp/gosd.cpp @@ -710,14 +710,14 @@ void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled) { params->enabled = enabled; } -void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y) { - params->tile_size_x = tile_size_x; - params->tile_size_y = tile_size_y; +void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h) { + params->tile_size_w = tile_size_w; + params->tile_size_h = tile_size_h; } -void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y) { - params->rel_size_x = rel_size_x; - params->rel_size_y = rel_size_y; +void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h) { + params->rel_size_w = rel_size_w; + params->rel_size_h = rel_size_h; } void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap) { diff --git a/backend/go/stablediffusion-ggml/cpp/gosd.h b/backend/go/stablediffusion-ggml/cpp/gosd.h index 31ce72ab7..c6613d4ed 100644 --- a/backend/go/stablediffusion-ggml/cpp/gosd.h +++ b/backend/go/stablediffusion-ggml/cpp/gosd.h @@ -6,8 +6,8 @@ extern "C" { #endif void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled); -void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y); -void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y); +void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h); +void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h); void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap); sd_tiling_params_t* sd_img_gen_params_get_vae_tiling_params(sd_img_gen_params_t *params); From 64c5670382a8ea18bfab3ee64de8df2555e2314c Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sat, 26 Sep 2026 17:19:38 +0200 Subject: [PATCH 06/42] ci: bump Hugo from 0.146.3 to 0.166.0 (#12281) The hugo-theme-relearn submodule was bumped to 9.1.x in #12096, which requires Hugo >= 0.165.0. The pinned 0.146.3 broke the docs site build with a template error in alias.html that could not evaluate the Locale field on langs.Language. Bump HUGO_VERSION to 0.166.0 (latest stable) to satisfy the theme minimum and resolve the alias.html template error. Assisted-by: nib:claude-sonnet-4.5 [bash] [read] [edit] Co-authored-by: Ettore Di Giacinto --- .github/workflows/gh-pages.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/gh-pages.yml b/.github/workflows/gh-pages.yml index 746ed923d..7de870401 100644 --- a/.github/workflows/gh-pages.yml +++ b/.github/workflows/gh-pages.yml @@ -40,7 +40,7 @@ jobs: # fetch their own toolchains, and no step uses sudo, apt, make or unzip. runs-on: ${{ github.repository == 'mudler/LocalAI' && 'arc-runner-set' || 'ubuntu-latest' }} env: - HUGO_VERSION: "0.146.3" + HUGO_VERSION: "0.166.0" steps: - name: Checkout uses: actions/checkout@v7 From a95da0a46c7defbd9c1644e33479583dbd35ea3c Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sat, 26 Sep 2026 18:53:52 +0200 Subject: [PATCH 07/42] fix(gallery): read the models dir once per gallery listing (#12283) The cached gallery listing refreshed each entry's installed flag with one os.Stat per entry, under the cache's global write lock. A gallery holds about 1,900 entries. On a models directory on SMB, one refresh took about 14s. The listing and every row's VRAM estimate run this refresh, and the lock serialized them, so the models page took minutes to load. The installed check now lists the models directory once and looks up each entry in that listing. On the same SMB share the listing takes about 75ms. The answers match os.Stat: a symlink counts only when its target exists, and names with a path separator still use os.Stat. The listing runs before the lock is taken, so the lock covers only the flag updates. Concurrent callers on a cold cache now share one upstream load. Before this, each caller fetched the gallery index and the configs itself. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- core/gallery/gallery.go | 48 ++++--- core/gallery/gallery_installed_scan_test.go | 138 ++++++++++++++++++++ core/gallery/installed_configs.go | 61 +++++++++ 3 files changed, 229 insertions(+), 18 deletions(-) create mode 100644 core/gallery/gallery_installed_scan_test.go create mode 100644 core/gallery/installed_configs.go diff --git a/core/gallery/gallery.go b/core/gallery/gallery.go index bd054b0ff..68d6de9e8 100644 --- a/core/gallery/gallery.go +++ b/core/gallery/gallery.go @@ -4,7 +4,6 @@ import ( "context" "fmt" "os" - "path/filepath" "slices" "strings" "sync" @@ -19,6 +18,7 @@ import ( "github.com/mudler/LocalAI/pkg/vram" "github.com/mudler/LocalAI/pkg/xsync" "github.com/mudler/xlog" + "golang.org/x/sync/singleflight" "gopkg.in/yaml.v3" ) @@ -276,13 +276,12 @@ func FindGalleryElement[T GalleryElement](models []T, name string) T { func AvailableGalleryModels(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryModel], error) { var models []*GalleryModel + isInstalled := installedConfigs(systemState.Model.ModelsPath) + // Get models from galleries for _, gallery := range galleries { galleryModels, err := getGalleryElements(gallery, systemState.Model.ModelsPath, systemState.RequireBackendIntegrity, func(model *GalleryModel) bool { - if _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", model.GetName()))); err == nil { - return true - } - return false + return isInstalled(model.GetName()) }) if err != nil { return nil, err @@ -351,6 +350,7 @@ var ( // same cache-defeating loop the refresh interval exists to stop. availableModelsLoaded bool refreshing atomic.Bool + coldLoad singleflight.Group galleryGeneration atomic.Uint64 lastRefreshUnixNano atomic.Int64 ) @@ -429,12 +429,15 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste availableModelsMu.RUnlock() if loaded { + // The directory is read before taking the lock. Held across the + // filesystem work, the lock serialized every caller behind it, and a + // page view is dozens of concurrent callers. + isInstalled := installedConfigs(systemState.Model.ModelsPath) // Refresh installed status under write lock to avoid races with // concurrent readers and the background refresh goroutine. availableModelsMu.Lock() for _, m := range cached { - _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", m.GetName()))) - m.SetInstalled(err == nil) + m.SetInstalled(isInstalled(m.GetName())) } availableModelsMu.Unlock() // Trigger a background refresh if one is not already running. @@ -442,20 +445,29 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste return cached, nil } - // No cache yet — must do a blocking load. - models, err := AvailableGalleryModels(galleries, systemState) + // No cache yet, so the load blocks. Callers arriving while it runs wait + // for it instead of each starting their own: a page view on a fresh + // server is the listing plus one estimate per row at once, and each load + // fetches the gallery index and every config it references. + v, err, _ := coldLoad.Do("gallery", func() (any, error) { + models, err := AvailableGalleryModels(galleries, systemState) + if err != nil { + return nil, err + } + + availableModelsMu.Lock() + availableModelsCache = models + availableModelsLoaded = true + galleryGeneration.Add(1) + availableModelsMu.Unlock() + lastRefreshUnixNano.Store(time.Now().UnixNano()) + + return models, nil + }) if err != nil { return nil, err } - - availableModelsMu.Lock() - availableModelsCache = models - availableModelsLoaded = true - galleryGeneration.Add(1) - availableModelsMu.Unlock() - lastRefreshUnixNano.Store(time.Now().UnixNano()) - - return models, nil + return v.(GalleryElements[*GalleryModel]), nil } // triggerGalleryRefresh starts a background goroutine that refreshes the diff --git a/core/gallery/gallery_installed_scan_test.go b/core/gallery/gallery_installed_scan_test.go new file mode 100644 index 000000000..7bcf22a1f --- /dev/null +++ b/core/gallery/gallery_installed_scan_test.go @@ -0,0 +1,138 @@ +package gallery_test + +import ( + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "sync" + "sync/atomic" + "time" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/system" +) + +// The models directory is often network storage (SMB, NFS), where every +// filesystem call is a round trip. The cached listing is read by the gallery +// page and by one VRAM estimate per row, so whatever it costs is paid dozens +// of times per page view. +var _ = Describe("Gallery cache installed status", func() { + const index = ` +- name: plain + backend: llama-cpp +- name: linked + backend: llama-cpp +- name: dangling + backend: llama-cpp +- name: later + backend: llama-cpp +- name: absent + backend: llama-cpp +` + + var ( + modelsDir string + state *system.SystemState + galleries []config.Gallery + hits atomic.Int32 + delay time.Duration + ) + + BeforeEach(func() { + var err error + modelsDir, err = os.MkdirTemp("", "gallery-installed") + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(func() { _ = os.RemoveAll(modelsDir) }) + state, err = system.GetSystemState(system.WithModelPath(modelsDir)) + Expect(err).ToNot(HaveOccurred()) + + hits.Store(0) + delay = 0 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + hits.Add(1) + time.Sleep(delay) + _, _ = w.Write([]byte(index)) + })) + DeferCleanup(server.Close) + galleries = []config.Gallery{{Name: "test", URL: server.URL + "/index.yaml"}} + + gallery.ResetGalleryModelCache() + DeferCleanup(gallery.ResetGalleryModelCache) + }) + + installed := func(models gallery.GalleryElements[*gallery.GalleryModel]) map[string]bool { + out := map[string]bool{} + for _, m := range models { + out[m.Name] = m.Installed + } + return out + } + + It("reports what os.Stat would, for files, symlinks and dangling symlinks", func() { + Expect(os.WriteFile(filepath.Join(modelsDir, "plain.yaml"), []byte("name: plain\n"), 0o644)).To(Succeed()) + target := filepath.Join(modelsDir, "target.txt") + Expect(os.WriteFile(target, []byte("name: linked\n"), 0o644)).To(Succeed()) + Expect(os.Symlink(target, filepath.Join(modelsDir, "linked.yaml"))).To(Succeed()) + Expect(os.Symlink(filepath.Join(modelsDir, "missing"), filepath.Join(modelsDir, "dangling.yaml"))).To(Succeed()) + + // Both the blocking first load and the cached path set the flag, and + // they must agree. + for range 2 { + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(Equal(map[string]bool{ + "plain": true, + "linked": true, + "dangling": false, + "later": false, + "absent": false, + })) + } + }) + + It("picks up a config written after the gallery was cached", func() { + _, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + + Expect(os.WriteFile(filepath.Join(modelsDir, "later.yaml"), []byte("name: later\n"), 0o644)).To(Succeed()) + + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(HaveKeyWithValue("later", true)) + }) + + It("reports nothing installed when the models directory is gone", func() { + _, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(os.RemoveAll(modelsDir)).To(Succeed()) + + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(HaveEach(BeFalse())) + }) + + It("shares one upstream load between concurrent callers on a cold cache", func() { + // Slow enough that every caller arrives while the first load is still + // in flight, which is what a page view does to a freshly started + // server: the listing and every row's estimate at once. + delay = 300 * time.Millisecond + + var wg sync.WaitGroup + for range 8 { + wg.Go(func() { + defer GinkgoRecover() + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(models).To(HaveLen(5)) + }) + } + wg.Wait() + + Expect(hits.Load()).To(Equal(int32(1))) + }) +}) diff --git a/core/gallery/installed_configs.go b/core/gallery/installed_configs.go new file mode 100644 index 000000000..fb36e5d29 --- /dev/null +++ b/core/gallery/installed_configs.go @@ -0,0 +1,61 @@ +package gallery + +import ( + "errors" + "io/fs" + "os" + "path/filepath" + "strings" +) + +const modelConfigExt = ".yaml" + +// installedConfigs answers "does /.yaml exist?" for every +// entry of a gallery from a single read of the models directory. +// +// The question used to be asked with one os.Stat per gallery entry. The gallery +// holds thousands of entries and the models directory is often network storage +// (SMB, NFS), where each Stat is a round trip, so one listing cost seconds. The +// listing is read by the gallery page and by one VRAM estimate per row, which +// turned a page view into minutes. +// +// Answers match os.Stat on the same path: a symlink counts only when its target +// exists, and anything else carrying the name counts, directories included. +// Names that are not a plain file name are checked with os.Stat directly, since +// they point outside the listed directory. +func installedConfigs(modelsPath string) func(name string) bool { + statInstalled := func(name string) bool { + _, err := os.Stat(filepath.Join(modelsPath, name+modelConfigExt)) + return err == nil + } + + entries, err := os.ReadDir(modelsPath) + if err != nil { + if errors.Is(err, fs.ErrNotExist) { + return func(string) bool { return false } + } + // A directory that exists but cannot be listed may still answer a + // Stat, so fall back rather than report everything as not installed. + return statInstalled + } + + present := make(map[string]struct{}, len(entries)) + for _, e := range entries { + base, ok := strings.CutSuffix(e.Name(), modelConfigExt) + if !ok { + continue + } + if e.Type()&fs.ModeSymlink != 0 && !statInstalled(base) { + continue + } + present[base] = struct{}{} + } + + return func(name string) bool { + if strings.ContainsRune(name, '/') || strings.ContainsRune(name, filepath.Separator) { + return statInstalled(name) + } + _, ok := present[name] + return ok + } +} From 92b8f1d8ede207e8a7c2862005eef89cbc443d56 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sat, 26 Sep 2026 18:54:08 +0200 Subject: [PATCH 08/42] chore(gallery): add MiMo distill Qwen 9B variants (#12282) Add Q4_K_M and Q8_0 builds with the F16 vision projector and pinned artifact URLs. Document installation and explicit variant selection. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- docs/content/features/model-gallery.md | 9 +++ gallery/index.yaml | 86 ++++++++++++++++++++++++++ 2 files changed, 95 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index 7baa18087..dd48e73c7 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,15 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## MiMo-V2.6-Distill-Qwen-9B + +Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. +This MIT-licensed 9B Qwen3.5 fine-tune targets coding, agent tasks, and visual coding. +The gallery groups Q4_K_M and Q8_0 builds as variants; both include the F16 vision projector. +To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9b --variant mimo-v2.6-distill-qwen-9b-q8`. +The configurations default to 32,768 context tokens and use the model's embedded chat template. +See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..84733ea18 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -302,6 +302,92 @@ - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-MTP-Q4_K_M/mmproj-F32.gguf uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-GGUF/resolve/e146d61e88782677805b3b68ad3adf8674dde80d/mmproj-F32.gguf sha256: 52e6818e4d18eea010c50e5245eaa10a8cc3dcc30efea4ff60cbad8abf5669e1 +- name: mimo-v2.6-distill-qwen-9b + variants: + - model: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q4_K_M GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + sha256: 4bca6f18c73f72270c7a20c2ea2bea581de8246e318714277120369d34048c81 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q8_0 GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + sha256: 2fad0aa11bb9e7aa491ff12f768954f9dd0a6e7d4ce4a897ca73ec420f3b90ae + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf - name: hemmingway-1 variants: - model: hemmingway-1-q8 From 6f1b3d3fa9f812103bab8ddbc37d24cfcc62f360 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:03:15 +0000 Subject: [PATCH 09/42] docs(proxy): clarify optional upstream API keys Document the existing no-auth upstream configuration and distinguish upstream credentials from LocalAI client authentication. Closes #12264 Assisted-by: Codex:gpt-6 --- docs/content/operations/cloud-proxy.md | 39 +++++++++++++++++++++++--- 1 file changed, 35 insertions(+), 4 deletions(-) diff --git a/docs/content/operations/cloud-proxy.md b/docs/content/operations/cloud-proxy.md index 02af25bd0..a258ece14 100644 --- a/docs/content/operations/cloud-proxy.md +++ b/docs/content/operations/cloud-proxy.md @@ -60,9 +60,9 @@ against - and two modes: `proxy.provider` selects the auth scheme and (in translate mode) the wire format. Supported values: `openai`, `anthropic`. -API keys are loaded from either an environment variable (`api_key_env`) or a -file (`api_key_file`). The key never appears in the config file or the admin -UI; pick whichever fits your secret-management setup. +If the upstream requires an API key, configure either an environment variable +(`api_key_env`) or a file (`api_key_file`). The key never appears in the config +file or the admin UI. If the upstream requires no API key, omit both fields. ### OpenAI passthrough @@ -126,7 +126,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak the OpenAI chat-completions wire format. Use `provider: openai` with the -provider's URL and API key: +provider's URL and, if required, its API key: ```yaml name: llama-3-70b-via-together @@ -140,6 +140,37 @@ proxy: upstream_model: meta-llama/Llama-3-70b-chat-hf ``` +### Upstreams without an API key + +For an OpenAI-compatible upstream that accepts requests without authentication, +omit both `api_key_env` and `api_key_file`: + +```yaml +name: internal-chat-proxy +backend: cloud-proxy + +proxy: + mode: passthrough + provider: openai + upstream_url: http://inference.internal:8000/v1/chat/completions + upstream_model: my-model +``` + +Replace the example URL and model name with your upstream's values. LocalAI +loads this configuration without resolving a key and adds no upstream +`Authorization` header. This also applies to OpenAI-compatible upstreams in +translate mode. + +Omitting both fields differs from setting `api_key_env` to an empty or unset +environment variable: the latter causes a backend load error. + +LocalAI's client authentication is separate. Clients must still authenticate +to LocalAI when its authentication is enabled. LocalAI does not forward their +`Authorization` header to the upstream. + +An upstream without API keys can still require another authentication or +payment protocol. Omitting these fields does not implement that protocol. + ### Translate mode In translate mode the cloud-proxy backend converts LocalAI's internal proto From e35a9679703c437405b5a3fbe3a0515d21e21932 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sat, 26 Sep 2026 20:05:14 +0000 Subject: [PATCH 10/42] chore(gallery): add Sharp-Spark 4B variants Add Q4, Q5, and Q6 builds with the embedded chat template. Pin artifact revisions and document installation. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 14 +++ gallery/index.yaml | 114 +++++++++++++++++++++++++ 2 files changed, 128 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..810c82d56 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,20 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Sharp-Spark-X2.5-4B + +Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp. +The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants. +To select the publisher's recommended Q6 build, run: + +```bash +local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6 +``` + +All builds use a 32,768-token default context and the embedded Sharp-Spark chat template. +That template adds a terseness instruction to the system prompt. +See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..c5ffc5b9a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5725,6 +5725,120 @@ - filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442 +- name: sharp-spark-x2.5-4b + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: sharp-spark-x2.5-4b-q5 + - model: sharp-spark-x2.5-4b-q6 + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q4_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837 + +- name: sharp-spark-x2.5-4b-q5 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q5_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26 + +- name: sharp-spark-x2.5-4b-q6 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q6_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc + - &spark-x2-5-4b name: "spark-x2.5-4b-q4" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" From dc4db4b119a2991b10283e49d6c55016bb20d2c3 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:07:50 +0000 Subject: [PATCH 11/42] chore(gallery): add Swift 1.5 GSQ-RCO variants Add four text-only llama.cpp builds with pinned download URLs and verified checksums. Document variant selection and the model license. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 15 +++ gallery/index.yaml | 172 +++++++++++++++++++++++++ 2 files changed, 187 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..6eb034ba2 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,21 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Swift 1.5 Qwen3.8-27B GSQ-RCO + +Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp. +The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model. +To select IQ3_S explicitly, run: + +```bash +local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s +``` + +The configurations use the embedded chat template and default to 32,768 context tokens. +These builds support text chat only: the publisher has no verified vision projector for this release. +They use standard GGUF files without MTP decoding. +See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..47fb9b4f7 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5533,6 +5533,178 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2 +- name: "swift-1.5-qwen3.8-27b-gsq-rco" + variants: + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf - !!merge <<: *qwen3-8-27b name: "qwen3.8-27b-gsq-rco-iq2-xs" variants: [] From 7460312d23dce57f026025343e0557be1594236b Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 04:04:34 +0000 Subject: [PATCH 12/42] chore(gallery): add ThinkingCap Qwen3.8 variants Add Q4_K_M and Q8_0 builds with the F16 vision projector and install docs. Pin artifact revisions and verify SHA256 against HF LFS metadata and HTTP headers. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 15 ++++ gallery/index.yaml | 96 ++++++++++++++++++++++++++ 2 files changed, 111 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..fca3a2c96 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -48,6 +48,21 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9 The configurations default to 32,768 context tokens and use the model's embedded chat template. See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. +## ThinkingCap Qwen3.8-27B + +Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input. +The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector. +LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly: + +```bash +local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8 +``` + +Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings. +MTP speculative decoding is not enabled by these entries. +The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE). +Review that license for permitted use. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..dc1f0a7d6 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -388,6 +388,102 @@ - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: thinkingcap-qwen3.8-27b + variants: + - model: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf +- name: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf - name: hemmingway-1 variants: - model: hemmingway-1-q8 From dcddb641f0564ce3d328748581655b6b1b4a32d0 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 08:05:46 +0000 Subject: [PATCH 13/42] chore(gallery): add Agention Qwen3.8 variants Add IQ4_XS and Q4_K_M GGUF builds with a BF16 vision projector. Pin verified artifacts and document installation and variant selection. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 18 +++++ gallery/index.yaml | 98 ++++++++++++++++++++++++++ 2 files changed, 116 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..bf83b6300 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,24 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Qwen3.8-27B Agention Precision + +The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of +Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image +input and use a 32,768-token context by default. + +Install with automatic variant selection: + +```bash +local-ai models install qwen3.8-27b-agention-iq4-xs +``` + +To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs` +or `--variant qwen3.8-27b-agention-q4-k-m` to the same command. +The files use standard llama.cpp quantization types and the Apache-2.0 license. +See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF) +for quantization details. These entries do not enable MTP speculative decoding. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..0cab8c751 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5247,6 +5247,104 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545 +- name: qwen3.8-27b-agention-iq4-xs + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: qwen3.8-27b-agention-q4-k-m + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf + sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67 + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 +- name: qwen3.8-27b-agention-q4-k-m + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf + sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 - &qwen3-8-27b name: "qwen3.8-27b-q4" variants: From 9ea9277ee69be910faee05bce7cbe9bf9cfdc67a Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:03 +0200 Subject: [PATCH 14/42] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `cdf232cc17e410e60c1bc3b85516c4a41199b662` (#12288) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index 0f2a84e7d..d6bfd7490 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc +IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662 LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From 01017dcdd631baca0886cd8ad7e6ca48d06594b8 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:25 +0200 Subject: [PATCH 15/42] chore: :arrow_up: Update CrispStrobe/CrispASR to `013ae1624dc40ecf059065d577180722439f804e` (#12292) :arrow_up: Update CrispStrobe/CrispASR Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/crispasr/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index 65a4a784d..f2155ffb9 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee +CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From fc6df9efc33626b6a4354fba32b00c9d6b270cca Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:24:52 +0200 Subject: [PATCH 16/42] chore: :arrow_up: Update mudler/parakeet.cpp to `2bf88954dc628b32835734e2e9159550a75a1dc6` (#12291) :arrow_up: Update mudler/parakeet.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/parakeet-cpp/Makefile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index 8fc14bcb8..e288f6fcc 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go From 4524765b9f6a83234415ba1f8b2177f4e9b3ac84 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:25:09 +0200 Subject: [PATCH 17/42] chore: :arrow_up: Update 0xShug0/audio.cpp to `94bd4656399180befc141b17bd6696bf84df0a9f` (#12289) :arrow_up: Update 0xShug0/audio.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/audio-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index 85da233bd..e8d27eb52 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22 +AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) From c9e822215a2cda5362fdd8f679571dc28ccfc32c Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 13:25:27 +0200 Subject: [PATCH 18/42] chore(model-gallery): :arrow_up: update checksum (#12290) :arrow_up: Checksum updates in gallery/index.yaml Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- gallery/index.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..5b1df136a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -205,7 +205,7 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7 + sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf - name: "qwopus3.8-27b-flash" variants: - model: qwopus3.8-27b-flash-q8 From 6043e5e0cb61db03e06be7cb59f299348cc17fd9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:04:40 +0000 Subject: [PATCH 19/42] chore(gallery): add Qwopus Flash V2 variants Add Q4_K_M and Q8_0 builds with vision and MTP decoding. Pin the weights and projector to a verified Hugging Face revision. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 14 ++++ gallery/index.yaml | 94 ++++++++++++++++++++++++++ 2 files changed, 108 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..4839b88ab 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -48,6 +48,20 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9 The configurations default to 32,768 context tokens and use the model's embedded chat template. See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. +## Qwopus3.8 Flash V2 + +Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection: + +```bash +local-ai models install qwopus3.8-27b-flash-v2 +local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8 +``` + +Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector. +Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256. +This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks. +See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index 5b1df136a..d5e56d590 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -206,6 +206,100 @@ - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf +- name: "qwopus3.8-27b-flash-v2" + variants: + - model: qwopus3.8-27b-flash-v2-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q4_K_M GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 +- name: "qwopus3.8-27b-flash-v2-q8" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q8_0 GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78 + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 - name: "qwopus3.8-27b-flash" variants: - model: qwopus3.8-27b-flash-q8 From 065f9691fa27fdfbd4362f582c426412234d0b1d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:05:55 +0000 Subject: [PATCH 20/42] chore(gallery): add Cyber-Tiel-Coder variants Add Q4 and Q8 MTP builds with a shared vision projector and installation docs. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 8 ++ gallery/index.yaml | 104 +++++++++++++++++++++++++ 2 files changed, 112 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..e0a5c60bc 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,14 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Cyber-Tiel-Coder + +Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp. +The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector. +To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`. +Both configurations use the embedded chat template and default to 32,768 context tokens. +The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 5b1df136a..18c2dd662 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -4767,6 +4767,110 @@ - filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41 +- name: cyber-tiel-coder-35b-a3b-q4-mtp + variants: + - model: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 +- name: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961 + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 - &tiel-coder-35b-a3b name: "tiel-coder-35b-a3b-q4" variants: From 7690b06789c3b4a39f48a53f2ac084bb29268138 Mon Sep 17 00:00:00 2001 From: Stefan Walcz Date: Sun, 27 Sep 2026 20:58:01 +0200 Subject: [PATCH 21/42] chore(deps): bump LocalAGI to 8253de9 (re-dial dropped MCP sessions) (#12299) Picks up mudler/LocalAGI 3ce0a08 "fix(mcp): re-dial an MCP session the server has dropped". Agents open their MCP sessions once, when they are created; when the MCP server restarts it forgets them and the go-sdk client does not reconnect by itself, so the agent kept a dead session - or, behind a server that revives unknown session IDs, a stale tool list - until LocalAI restarted. Only go.mod/go.sum change; core/services/agentpool builds and vets against the new version. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz --- go.mod | 4 ++-- go.sum | 2 ++ 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/go.mod b/go.mod index 1d4025b89..c81fbd82e 100644 --- a/go.mod +++ b/go.mod @@ -259,7 +259,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 + github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect @@ -535,7 +535,7 @@ require ( golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect golang.org/x/mod v0.36.0 // indirect golang.org/x/sync v0.20.0 - golang.org/x/sys v0.45.0 // indirect + golang.org/x/sys v0.45.0 golang.org/x/term v0.43.0 golang.org/x/text v0.37.0 golang.org/x/tools v0.45.0 // indirect diff --git a/go.sum b/go.sum index 891a60d05..863118db1 100644 --- a/go.sum +++ b/go.sum @@ -1032,6 +1032,8 @@ github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxF github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= From a592e23778666b95c8ebfd5cefcd45decd4f6f0e Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:40 +0200 Subject: [PATCH 22/42] chore: :arrow_up: Update ggml-org/llama.cpp to `95887577ab5fead779581a7030a83c7752ff3234` (#12272) :arrow_up: Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index b672f2d31..8bf0ff5c0 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4 +LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= From 08827cfd5e60a97f107681a638bde9f0cd6e75d9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:45 +0200 Subject: [PATCH 23/42] chore: :arrow_up: Update mudler/vllm.cpp to `c3bebc357385990f721af66a3a6c69328dd4fc6c` (#12252) :arrow_up: Update mudler/vllm.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/vllm-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index d19d103b1..56a32dbaf 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb +VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun From 4bc5f292fe12f6bff847defe5f6f63b86d6b2c66 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:05:49 +0200 Subject: [PATCH 24/42] chore: :arrow_up: Update TheTom/llama-cpp-turboquant to `a3d5603d110bda29222d2011596cdc84d7fa532d` (#12232) * :arrow_up: Update TheTom/llama-cpp-turboquant Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(turboquant): drop the upstreamed D512 patch Upstream c0e227c guards D512 declarations, dispatch, and instances with GGML_USE_HIP. This prevents the CUDA shared-memory overflow that our patch addressed. The old patch now rejects the guarded source. Remove the obsolete patch for the pinned a3d5603d revision. The remaining patch series applies successfully, and the build-target test passes. Assisted-by: Codex:gpt-6 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/cpp/turboquant/Makefile | 2 +- ...emove-d512-turbo-shared-mem-overflow.patch | 52 ------------------- 2 files changed, 1 insertion(+), 53 deletions(-) delete mode 100644 backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch diff --git a/backend/cpp/turboquant/Makefile b/backend/cpp/turboquant/Makefile index e3482d8db..7a8024ebf 100644 --- a/backend/cpp/turboquant/Makefile +++ b/backend/cpp/turboquant/Makefile @@ -1,7 +1,7 @@ # Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant. # Auto-bumped nightly by .github/workflows/bump_deps.yaml. -TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64 +TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant CMAKE_ARGS?= diff --git a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch b/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch deleted file mode 100644 index 7dfb385c3..000000000 --- a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch +++ /dev/null @@ -1,52 +0,0 @@ -diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh -index 680fd12..ffd6604 100644 ---- a/ggml/src/ggml-cuda/fattn-vec.cuh -+++ b/ggml/src/ggml-cuda/fattn-vec.cuh -@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); -diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu -index 5c614a9..d765cfc 100644 ---- a/ggml/src/ggml-cuda/fattn.cu -+++ b/ggml/src/ggml-cuda/fattn.cu -@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0) - - #ifdef GGML_CUDA_FA_ALL_QUANTS - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16) -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -index a93be56..3630d87 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -index 3c806c2..c8a4d9f 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -index 180902f..1646ef0 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); From 13a01e657a1a389bad564c57c6fb3e82dee55835 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:05:55 +0200 Subject: [PATCH 25/42] chore(deps): bump sentence-transformers from 5.7.0 to 6.1.0 in /backend/python/transformers (#12245) chore(deps): bump sentence-transformers in /backend/python/transformers Bumps [sentence-transformers](https://github.com/huggingface/sentence-transformers) from 5.7.0 to 6.1.0. - [Release notes](https://github.com/huggingface/sentence-transformers/releases) - [Commits](https://github.com/huggingface/sentence-transformers/compare/v5.7.0...v6.1.0) --- updated-dependencies: - dependency-name: sentence-transformers dependency-version: 6.1.0 dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements-cpu.txt | 2 +- backend/python/transformers/requirements-cublas12.txt | 2 +- backend/python/transformers/requirements-cublas13.txt | 2 +- backend/python/transformers/requirements-hipblas.txt | 2 +- backend/python/transformers/requirements-intel.txt | 2 +- backend/python/transformers/requirements-mps.txt | 2 +- 6 files changed, 6 insertions(+), 6 deletions(-) diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt index 3e3206912..6cd324d8d 100644 --- a/backend/python/transformers/requirements-cpu.txt +++ b/backend/python/transformers/requirements-cpu.txt @@ -4,7 +4,7 @@ numba==0.67.0 accelerate transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt index 40bf331d4..632330714 100644 --- a/backend/python/transformers/requirements-cublas12.txt +++ b/backend/python/transformers/requirements-cublas12.txt @@ -4,7 +4,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt index f394f98b1..e91307ce8 100644 --- a/backend/python/transformers/requirements-cublas13.txt +++ b/backend/python/transformers/requirements-cublas13.txt @@ -4,7 +4,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt index e4b1bba11..c97e4d27b 100644 --- a/backend/python/transformers/requirements-hipblas.txt +++ b/backend/python/transformers/requirements-hipblas.txt @@ -5,7 +5,7 @@ transformers>=5.15.1 llvmlite==0.49.0 numba==0.67.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt index 54ee6ce67..0ae7eb9fe 100644 --- a/backend/python/transformers/requirements-intel.txt +++ b/backend/python/transformers/requirements-intel.txt @@ -5,7 +5,7 @@ llvmlite==0.49.0 numba==0.67.0 transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt index ea8ba5ab0..ee8431b28 100644 --- a/backend/python/transformers/requirements-mps.txt +++ b/backend/python/transformers/requirements-mps.txt @@ -4,7 +4,7 @@ numba==0.67.0 accelerate transformers>=5.15.1 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 From 1cacecc460b6e85e3c5f172addf4d7158a9e1332 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:05:59 +0200 Subject: [PATCH 26/42] chore(deps): update numpy requirement from >=2.5.2 to >=2.5.3 in /backend/python/transformers (#12244) chore(deps): update numpy requirement in /backend/python/transformers Updates the requirements on [numpy](https://github.com/numpy/numpy) to permit the latest version. - [Release notes](https://github.com/numpy/numpy/releases) - [Changelog](https://github.com/numpy/numpy/blob/main/doc/RELEASE_WALKTHROUGH.rst) - [Commits](https://github.com/numpy/numpy/compare/v2.5.2...v2.5.3) --- updated-dependencies: - dependency-name: numpy dependency-version: 2.5.3 dependency-type: direct:production ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt index d85aca02a..63043dc5e 100644 --- a/backend/python/transformers/requirements.txt +++ b/backend/python/transformers/requirements.txt @@ -3,4 +3,4 @@ protobuf==7.36.1 certifi setuptools scipy==1.18.0 -numpy>=2.5.2 \ No newline at end of file +numpy>=2.5.3 \ No newline at end of file From de203c2fb5ef00594a869d935d8b583b34d46888 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:06:03 +0200 Subject: [PATCH 27/42] chore(deps): update transformers requirement from >=5.15.1 to >=5.17.0 in /backend/python/transformers (#12105) chore(deps): update transformers requirement Updates the requirements on [transformers](https://github.com/huggingface/transformers) to permit the latest version. - [Release notes](https://github.com/huggingface/transformers/releases) - [Commits](https://github.com/huggingface/transformers/compare/v5.15.1...v5.17.0) --- updated-dependencies: - dependency-name: transformers dependency-version: 5.17.0 dependency-type: direct:production ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements-cpu.txt | 2 +- backend/python/transformers/requirements-cublas12.txt | 2 +- backend/python/transformers/requirements-cublas13.txt | 2 +- backend/python/transformers/requirements-hipblas.txt | 2 +- backend/python/transformers/requirements-intel.txt | 2 +- backend/python/transformers/requirements-mps.txt | 2 +- 6 files changed, 6 insertions(+), 6 deletions(-) diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt index 6cd324d8d..8c5a025e2 100644 --- a/backend/python/transformers/requirements-cpu.txt +++ b/backend/python/transformers/requirements-cpu.txt @@ -2,7 +2,7 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt index 632330714..388a2334c 100644 --- a/backend/python/transformers/requirements-cublas12.txt +++ b/backend/python/transformers/requirements-cublas12.txt @@ -2,7 +2,7 @@ torch==2.7.1 accelerate llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt index e91307ce8..aa49676b1 100644 --- a/backend/python/transformers/requirements-cublas13.txt +++ b/backend/python/transformers/requirements-cublas13.txt @@ -2,7 +2,7 @@ torch==2.9.0 llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt index c97e4d27b..84b62042b 100644 --- a/backend/python/transformers/requirements-hipblas.txt +++ b/backend/python/transformers/requirements-hipblas.txt @@ -1,7 +1,7 @@ --extra-index-url https://download.pytorch.org/whl/rocm7.0 torch==2.10.0+rocm7.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 llvmlite==0.49.0 numba==0.67.0 bitsandbytes diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt index 0ae7eb9fe..b0b565bf1 100644 --- a/backend/python/transformers/requirements-intel.txt +++ b/backend/python/transformers/requirements-intel.txt @@ -3,7 +3,7 @@ torch optimum[openvino] llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt index ee8431b28..9659c943f 100644 --- a/backend/python/transformers/requirements-mps.txt +++ b/backend/python/transformers/requirements-mps.txt @@ -2,7 +2,7 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes sentence-transformers==6.1.0 diffusers From e6b2309b3fe5546833d094483e84437d75400acb Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:06:07 +0200 Subject: [PATCH 28/42] chore(deps): bump grpcio from 1.83.0 to 1.84.0 in /backend/python/transformers (#12102) chore(deps): bump grpcio in /backend/python/transformers Bumps [grpcio](https://github.com/grpc/grpc) from 1.83.0 to 1.84.0. - [Release notes](https://github.com/grpc/grpc/releases) - [Commits](https://github.com/grpc/grpc/compare/v1.83.0...v1.84.0) --- updated-dependencies: - dependency-name: grpcio dependency-version: 1.84.0 dependency-type: direct:production update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- backend/python/transformers/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt index 63043dc5e..af5027f10 100644 --- a/backend/python/transformers/requirements.txt +++ b/backend/python/transformers/requirements.txt @@ -1,4 +1,4 @@ -grpcio==1.83.0 +grpcio==1.84.0 protobuf==7.36.1 certifi setuptools From 7340970ae798a35e1e877e18f06c8a4791160d0e Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:21 +0000 Subject: [PATCH 29/42] chore(gallery): tag swift-qwen3.8-27b as mtp and vision, fix license The entry enables spec_type:draft-mtp, so variant ranking needs the mtp tag. Replace the scraped model-card description, set the Swift Open License v1.0 and link the base model repo. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 42 ++++++++++-------------------------------- 1 file changed, 10 insertions(+), 32 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 929c337bb..eb9e5e15c 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -2,44 +2,21 @@ - name: "swift-qwen3.8-27b" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27b - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF description: | - Website  •  - Learn more  •  - GGUF  •  - Enterprise licensing - - # Swift-Qwen3.8-27B - - Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, - using **58.3% fewer thinking tokens** while maintaining near-identical performance - (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. - - The prompt is a sample from LiveCodeBench v6 - - ## Training approach - - We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s - reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. - - Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. - - For maximum gains, Swift also includes a transfer component derived from - BottleCap AI's ThinkingCap-Qwen3.6-27B. - - ## Evaluation scope - - > All results below compare the Qwen3.8-27B BF16 base with the same base plus the - > Swift adapter. - - ## Benchmarks - - ... - license: "other" + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B. + The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss. + This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding. + The weights use the Swift Open License v1.0. + license: "swift-open-license-1.0" tags: - llm - gguf - reasoning + - vision + - multimodal + - mtp overrides: backend: llama-cpp function: @@ -48,6 +25,7 @@ disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf options: - use_jinja:true From 40d37330bcbdeaac3c124abe7307b427a603646f Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:33 +0000 Subject: [PATCH 30/42] docs: point the config example link at the configurations directory The link text still said chatbot-ui, but it now pointed at the examples repository root. Link the configurations directory, which holds the example model config files, and describe it as such. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- docs/content/advanced/advanced-usage.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md index 7850151ae..8580aa499 100644 --- a/docs/content/advanced/advanced-usage.md +++ b/docs/content/advanced/advanced-usage.md @@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master ``` -See also [chatbot-ui](https://github.com/mudler/LocalAI-examples) as an example on how to use config files. +See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files. ### Prompt templates From a7a6bc2963bc660e07446b00687f2dec77b8a88e Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:50 +0200 Subject: [PATCH 31/42] fix(ci): use Go 1.27 for Darwin backends (#12284) Older Go linkers stamp pure-Go hosts with SDK metadata that disables modern Metal APIs. Select Go 1.27 for Darwin builds and document the backend rebuild requirement. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/backend.yml | 2 +- .github/workflows/backend_build_darwin.yml | 3 ++- .github/workflows/backend_pr.yml | 2 +- docs/content/getting-started/build.md | 4 ++++ 4 files changed, 8 insertions(+), 3 deletions(-) diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 13e67c6fe..d4bb0bf59 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -355,7 +355,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/.github/workflows/backend_build_darwin.yml b/.github/workflows/backend_build_darwin.yml index 6b8b2a89b..19952a5ef 100644 --- a/.github/workflows/backend_build_darwin.yml +++ b/.github/workflows/backend_build_darwin.yml @@ -22,7 +22,8 @@ on: type: string go-version: description: 'Go version to use' - default: '1.24.x' + # Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs. + default: '1.27.x' type: string tag-suffix: description: 'Tag suffix for the built image' diff --git a/.github/workflows/backend_pr.yml b/.github/workflows/backend_pr.yml index c13c444c4..2626f87e8 100644 --- a/.github/workflows/backend_pr.yml +++ b/.github/workflows/backend_pr.yml @@ -281,7 +281,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/docs/content/getting-started/build.md b/docs/content/getting-started/build.md index 77884a412..42e64928d 100644 --- a/docs/content/getting-started/build.md +++ b/docs/content/getting-started/build.md @@ -30,6 +30,10 @@ To install the dependencies follow the instructions below: {{< tabs >}} {{% tab title="Apple" %}} +To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later. +Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts. +Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables. + Install `xcode` from the App Store ```bash From 4a691099abeda53aeb16fdac1596926006c98f31 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:53 +0000 Subject: [PATCH 32/42] chore(gallery): serve ternary-bonsai-2-27b with the bonsai backend PTQ1_0 is a Prism-private GGUF type (GGML_TYPE_PTQ1_0 = 143 in the PrismML llama.cpp fork), so stock llama-cpp cannot load it. Switch to the bonsai backend like the existing ternary-bonsai-27b entries, and replace the scraped Qwen3.8 description and icon. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 28 ++++++++++++---------------- 1 file changed, 12 insertions(+), 16 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index cceda1b77..a348c60dc 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -3,34 +3,30 @@ url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + - https://github.com/PrismML-Eng/llama.cpp description: | - # Qwen3.8-27B - - > [!Note] - > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. - > - > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. - - > [!Tip] - > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. - > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. - - Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. - - ... + Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary + transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits + per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a + Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's + llama.cpp fork) instead of stock llama.cpp. license: "apache-2.0" tags: - llm - gguf - icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + - reasoning + - vision + - multimodal + icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg overrides: - backend: llama-cpp + backend: bonsai function: automatic_tool_parsing_fallback: true grammar: disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf options: - use_jinja:true From 4e94c914c945fdf2f4a19adc4db2f1dc4c8d7c07 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:54 +0200 Subject: [PATCH 33/42] fix(swagger): describe backend metadata as an object (#12178) Swag cannot resolve json.RawMessage in OpenAIResponse and aborts the daily schema generation. Set its Swagger type without changing JSON encoding, and regenerate the checked-in specifications. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/schema/openai.go | 2 +- swagger/docs.go | 3 +++ swagger/swagger.json | 3 +++ swagger/swagger.yaml | 2 ++ 4 files changed, 9 insertions(+), 1 deletion(-) diff --git a/core/schema/openai.go b/core/schema/openai.go index 2aa69969b..6f3717256 100644 --- a/core/schema/openai.go +++ b/core/schema/openai.go @@ -99,7 +99,7 @@ type OpenAIResponse struct { // OpenAI-SDK consumers that filter on a truthy `result.usage` // (continuedev/continue, Kilo Code, Roo Code, etc.). Usage *OpenAIUsage `json:"usage,omitempty"` - Metadata json.RawMessage `json:"metadata,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"` } // StreamOptions mirrors OpenAI's `stream_options` request field. The only diff --git a/swagger/docs.go b/swagger/docs.go index 6ae74b94a..dc2fcf063 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -7313,6 +7313,9 @@ const docTemplate = `{ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, diff --git a/swagger/swagger.json b/swagger/swagger.json index c19487f6c..a1e71cbdb 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -7310,6 +7310,9 @@ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 6f9f6b1b7..12de303a0 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2266,6 +2266,8 @@ definitions: type: array id: type: string + metadata: + type: object model: type: string object: From 657cf9ca838e3acea2c242d10918631eca05cbd9 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:06:59 +0200 Subject: [PATCH 34/42] fix(ci): retain backend digests for release retries (#12160) The v4.10.0 ace-step and VibeVoice merge jobs started just after their digest artifacts expired. Keep the small digest artifacts for seven days so a multi-day release matrix can finish publishing its images. Assisted-by: Codex:gpt-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/backend_build.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.github/workflows/backend_build.yml b/.github/workflows/backend_build.yml index 05d50cf82..3e3af89f0 100644 --- a/.github/workflows/backend_build.yml +++ b/.github/workflows/backend_build.yml @@ -252,7 +252,8 @@ jobs: name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }} path: /tmp/digests/* if-no-files-found: error - retention-days: 1 + # Release matrices and their retries can outlive a one-day artifact. + retention-days: 7 - name: Build (PR) uses: docker/build-push-action@v7 From c6f1e96a7d9f515c95844afb6e1217ba380e9f54 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:07:04 +0200 Subject: [PATCH 35/42] chore(website): refresh the counters (#12039) Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- website/data/stats.yaml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/website/data/stats.yaml b/website/data/stats.yaml index a54e240e8..253b5444d 100644 --- a/website/data/stats.yaml +++ b/website/data/stats.yaml @@ -3,10 +3,10 @@ # The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh, # which runs weekly from .github/workflows/refresh-site-counters.yml. Editing # them by hand works but will be overwritten on the next run. -stars: 48949 -forks: 4430 -contributors: 237 -releases: 135 +stars: 49204 +forks: 4459 +contributors: 245 +releases: 136 # The GitHub API cannot answer for this one, so it is maintained by hand and # the refresh script carries it through untouched. From 1b1bd0f0694411910fb84ec0079fad6bf2c3cd34 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:07:08 +0200 Subject: [PATCH 36/42] fix(compose): request NVIDIA compute capability (#11990) The legacy NVIDIA device reservation requests utility without compute. Docker derives driver capabilities from that list, leaving CUDA libraries unavailable even when monitoring works. Include compute in the legacy example and clarify the matching docs. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- docker-compose.yaml | 3 ++- docs/content/features/distributed-mode.md | 8 ++++++-- docs/content/reference/nvidia-l4t.md | 6 ++++-- 3 files changed, 12 insertions(+), 5 deletions(-) diff --git a/docker-compose.yaml b/docker-compose.yaml index ee137e83c..82b3c18b6 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -59,6 +59,7 @@ services: # capabilities: [gpu, utility] # # For legacy NVIDIA driver (for older NVIDIA Container Toolkit): + # Request compute for CUDA libraries (libcuda.so.1) and utility for NVML. # environment: # NVIDIA_DRIVER_CAPABILITIES: "compute,utility" # init: true @@ -68,7 +69,7 @@ services: # devices: # - driver: nvidia # count: 1 - # capabilities: [gpu, utility] + # capabilities: [gpu, compute, utility] ## Uncomment for PostgreSQL-backed knowledge base (see Agents docs) # postgres: diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md index f8a06539a..86ad8b14e 100644 --- a/docs/content/features/distributed-mode.md +++ b/docs/content/features/distributed-mode.md @@ -417,8 +417,12 @@ usage is reported back to the frontend: NVML library (and therefore `nvidia-smi`) is not available inside the container. CUDA compute still works, but the worker cannot query free VRAM and the Nodes page will show the node as fully used. Set - `NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI - runtime, list `capabilities: [gpu, utility]` on the device reservation). + `NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime. + For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Docker derives driver capabilities from this reservation, so include `compute` + for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone + enables monitoring but does not provide CUDA libraries. - **Run the container with `init: true` (or `docker run --init`).** The worker process becomes PID 1 in the container and cannot reap zombies on diff --git a/docs/content/reference/nvidia-l4t.md b/docs/content/reference/nvidia-l4t.md index 2adac3a84..e3b54020a 100644 --- a/docs/content/reference/nvidia-l4t.md +++ b/docs/content/reference/nvidia-l4t.md @@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things: NVML work inside the container. With `--gpus all` alone (or `--runtime nvidia` without extra flags) only `compute` is wired in on some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility` - to your `docker run`, or `capabilities: [gpu, utility]` in compose / - Kubernetes device reservations. + to your `docker run`. For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility` + alone only provides monitoring libraries and tools. 2. Pass `--init` to `docker run` (or `init: true` in compose) so the container has a proper PID 1 reaper - otherwise short-lived child processes like `nvidia-smi` can intermittently fail with From b9634e0339451318828868c79a788c6e26b2e711 Mon Sep 17 00:00:00 2001 From: Leoy Date: Mon, 28 Sep 2026 03:10:35 +0800 Subject: [PATCH 37/42] fix(modelartifacts): reuse committed sibling files for narrowed allow_patterns (#11484) A request with narrower allow_patterns hashes to a different CacheKey than an already-committed broader sibling, so committedResult misses and materializeLocked re-fetches files the sibling already holds. After the own-tree reuseMaterializedFile miss, consult committed sibling trees for the same Source (type+endpoint+repo+revision), re-verify the file through verifyDownloadedFile (full SHA-256, never size-only), and hard-link it into the writer's staging snapshot (copy fallback only on EXDEV). Each file is matched individually against the sibling's manifest, so a broader request can never inherit a narrower sibling's gaps as if complete. The sibling manifest set is loaded and source-matched once per materialization (files indexed by path) instead of once per staged file, so a models volume with 20 committed artifacts and a 300-file snapshot does one manifest pass rather than ~6000 reads and JSON parses. The sibling-reuse behavior cases live in the package's registered Ginkgo suite so repository test conventions apply. Refs #11047 Signed-off-by: supermario_leo --- pkg/modelartifacts/materializer.go | 141 +++++++++++ .../materializer_sibling_reuse_test.go | 230 ++++++++++++++++++ 2 files changed, 371 insertions(+) create mode 100644 pkg/modelartifacts/materializer_sibling_reuse_test.go diff --git a/pkg/modelartifacts/materializer.go b/pkg/modelartifacts/materializer.go index a3e2a9e8e..87036f756 100644 --- a/pkg/modelartifacts/materializer.go +++ b/pkg/modelartifacts/materializer.go @@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedFiles := 0 skippedBytes := int64(0) tasks := make([]downloader.FileTask, 0, len(snapshot.Files)) + // Sibling manifests are read once, before the staging loop, so the + // per-file reuse lookups below never re-read or re-parse them. + siblings := loadSiblingCandidates(modelsPath, spec, layout) for index, file := range snapshot.Files { if err := ctx.Err(); err != nil { return Result{}, err @@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedBytes += file.Size continue } + // Before reaching for the network, consult committed sibling trees for the + // same Source (type+endpoint+repo+revision). A narrower allow_patterns + // request gets a different CacheKey, so committedResult misses even though a + // broader sibling already holds this exact file; reusing it avoids a + // redundant re-download of tens of gigabytes. The match is re-verified + // through verifyDownloadedFile (full SHA-256), never size-only, and a broader + // request can never inherit a narrower sibling's gaps because each file is + // matched individually against the sibling's manifest. + if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok { + manifest.Files[taskIndex] = entry + completedBytes.Add(file.Size) + skippedFiles++ + skippedBytes += file.Size + continue + } nameSum := sha256.Sum256([]byte(file.Path)) blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:])) blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel)) @@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest return entry, true } +// siblingCandidate is one committed sibling artifact tree that shares this +// request's Source (type+endpoint+repo+revision), with its manifest files +// indexed by path. +type siblingCandidate struct { + final string + filesByPath map[string][]ManifestFile +} + +// loadSiblingCandidates reads the committed sibling manifest set once, before +// the staging loop. Doing it per file instead would re-read and re-parse every +// sibling manifest for every file — 20 committed siblings and a 300-file +// snapshot means 6000 manifest reads before the first byte is fetched. +// +// The current artifact's own committed tree is excluded: it is either absent +// (the reason materializeLocked is running) or already handled by +// committedResult's exact-key fast path. +func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate { + if spec.Resolved == nil || layout.Final == "" { + return nil + } + siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface") + entries, err := os.ReadDir(siblingsRoot) + if err != nil { + return nil + } + var candidates []siblingCandidate + for _, entry := range entries { + if !entry.IsDir() { + continue + } + siblingFinal := filepath.Join(siblingsRoot, entry.Name()) + if siblingFinal == layout.Final { + continue + } + siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json")) + if err != nil { + continue + } + siblingArtifact := siblingManifest.Artifact + if siblingArtifact.Resolved == nil || + siblingArtifact.Source.Type != spec.Source.Type || + siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint || + siblingArtifact.Source.Repo != spec.Source.Repo || + siblingArtifact.Resolved.Revision != spec.Resolved.Revision { + continue + } + byPath := make(map[string][]ManifestFile, len(siblingManifest.Files)) + for _, f := range siblingManifest.Files { + byPath[f.Path] = append(byPath[f.Path], f) + } + candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath}) + } + return candidates +} + +// reuseFromCommittedSibling looks for a file already committed under a sibling +// artifact tree — same Source (type+endpoint+repo+revision), different +// allow/ignore patterns — and stages it for this writer instead of fetching. +// A narrower allow_patterns request gets a different CacheKey (path.go:62), so +// committedResult misses and materializeLocked would otherwise re-download +// files an already-committed broader sibling already holds. +// +// The match is never size-only: the sibling file is re-hashed through the +// shared verifyDownloadedFile against the current request's SnapshotFile (its +// LFS or git blob OID), so the staged entry is byte-for-byte identical to a +// fresh download. A broader request can never stand in for files a narrower +// sibling lacks, because each requested file is matched individually against +// the sibling's manifest file set. Hard-link keeps the shared models volume +// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the +// kernel cannot hard-link. +func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) { + snapshotRel := path.Join("snapshot", file.Path) + snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel)) + for _, sibling := range candidates { + for _, siblingFile := range sibling.filesByPath[file.Path] { + if siblingFile.Size != file.Size { + continue + } + siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path)) + verified, err := verifyDownloadedFile(siblingPath, file) + if err != nil { + continue + } + if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil { + return ManifestFile{}, false + } + _ = root.Remove(snapshotRel) + if err := linkOrCopy(siblingPath, snapshotAbs); err != nil { + return ManifestFile{}, false + } + return verified, true + } + } + return ManifestFile{}, false +} + +// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only +// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking +// keeps the shared models volume neutral — a narrowed request does not double +// the storage of a broad sibling's files. +func linkOrCopy(src, dst string) error { + if err := os.Link(src, dst); err == nil { + return nil + } else if !errors.Is(err, syscall.EXDEV) { + return err + } + in, err := os.Open(src) + if err != nil { + return err + } + defer func() { _ = in.Close() }() + out, err := os.Create(dst) + if err != nil { + return err + } + if _, err := io.Copy(out, in); err != nil { + _ = out.Close() + _ = os.Remove(dst) + return err + } + return out.Close() +} + func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) { file, err := os.Open(fileName) if err != nil { diff --git a/pkg/modelartifacts/materializer_sibling_reuse_test.go b/pkg/modelartifacts/materializer_sibling_reuse_test.go new file mode 100644 index 000000000..572b255d8 --- /dev/null +++ b/pkg/modelartifacts/materializer_sibling_reuse_test.go @@ -0,0 +1,230 @@ +package modelartifacts_test + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strconv" + "strings" + "sync" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + hfapi "github.com/mudler/LocalAI/pkg/huggingface-api" + "github.com/mudler/LocalAI/pkg/modelartifacts" +) + +const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567" + +// recordingResolver serves a fixed full file set filtered by each request's +// allow/ignore patterns, so a narrower request genuinely resolves to a strict +// subset of a broader sibling's files. The HTTP server behind it records every +// fetch, which is the signal the sibling-reuse fix is verified through. The +// function under fix is never mocked: a real Manager drives the real staging + +// commit path against this stub collaborator. +type recordingResolver struct { + endpoint string + repo string + files []hfapi.SnapshotFile + server *httptest.Server + + mu sync.Mutex + fetched map[string]int +} + +func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver { + r := &recordingResolver{ + endpoint: "https://huggingface.co", + repo: "owner/repo", + files: files, + fetched: map[string]int{}, + } + r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + name := strings.TrimPrefix(req.URL.Path, "/file/") + body, ok := contents[name] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + r.mu.Lock() + r.fetched[name]++ + r.mu.Unlock() + w.Header().Set("Content-Length", strconv.Itoa(len(body))) + _, _ = w.Write(body) + })) + return r +} + +func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) { + files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns) + if err != nil { + return hfapi.Snapshot{}, err + } + out := make([]hfapi.SnapshotFile, len(files)) + for i, f := range files { + f.URL = r.server.URL + "/file/" + f.Path + out[i] = f + } + return hfapi.Snapshot{ + Endpoint: r.endpoint, Repo: r.repo, + RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out, + }, nil +} + +func (r *recordingResolver) fetchCount(path string) int { + r.mu.Lock() + defer r.mu.Unlock() + return r.fetched[path] +} + +func (r *recordingResolver) resetFetches() { + r.mu.Lock() + defer r.mu.Unlock() + r.fetched = map[string]int{} +} + +func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile { + paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"} + files := make([]hfapi.SnapshotFile, 0, len(paths)) + for _, p := range paths { + sum := sha256.Sum256(contents[p]) + files = append(files, hfapi.SnapshotFile{ + Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]), + }) + } + return files +} + +// The narrow-request case proves the fix for #11047: +// a request with narrower allow_patterns (a strict subset) reuses files an +// already-committed broader sibling holds, hard-linking instead of re-fetching. +// +// On master this is RED: a narrower allow_patterns set hashes to a different +// CacheKey (path.go:62), so committedResult misses and materializeLocked +// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On +// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling, +// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0, +// os.SameFile true). +var _ = Describe("committed sibling reuse", func() { + It("reuses files from a broader committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit the broad sibling: all three files, fetched from the resolver. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(broad.CacheHit).To(BeFalse()) + Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0), + "the broad sibling must have fetched a/first.bin to commit it") + + resolver.resetFetches() + + // Narrowed request: a strict subset of the broad sibling's file set. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(narrow.CacheHit).To(BeFalse()) + + // (b) The sibling-present file must NOT be re-fetched: zero fetches. This is + // the assertion that is RED on master (one fetch) and GREEN on the branch. + Expect(resolver.fetchCount("a/first.bin")).To(Equal(0), + "a/first.bin must be reused from the committed broad sibling, not re-fetched") + + // (a) The narrowed tree's staged file is the same inode as the broad + // sibling's file (hard-link), not a freshly downloaded second copy. RED on + // master (separate file), GREEN on the branch (hard-link). + broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin") + narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin") + broadInfo, err := os.Stat(broadFile) + Expect(err).NotTo(HaveOccurred()) + narrowInfo, err := os.Stat(narrowFile) + Expect(err).NotTo(HaveOccurred()) + Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(), + "the narrowed request must hard-link the broad sibling's file rather than store a second copy") + + // The reused bytes are intact end to end. + Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"])) + }) + + // The broader-request case is the manifest file-set guard: a broader request + // against a narrower committed sibling must still fetch the files the sibling + // lacks and commit a complete tree. Sibling-reuse can never serve an incomplete + // model as complete, because each requested file is matched individually against + // the sibling's manifest. + It("fetches files missing from a narrower committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit a NARROW sibling first: only a/first.bin and b/second.bin. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin", "b/second.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + narrowPaths := make([]string, 0, len(narrow.Manifest.Files)) + for _, f := range narrow.Manifest.Files { + narrowPaths = append(narrowPaths, f.Path) + } + Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"})) + + resolver.resetFetches() + + // A BROADER request asks for all three files, including c/third.bin which the + // narrow sibling does not hold. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + + // The file the narrow sibling lacks MUST be fetched: sibling-reuse must not + // inherit a narrower tree's gaps as if the broad request were complete. + Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0), + "c/third.bin is absent from the narrow sibling and must be fetched, not served as complete") + + // The broad tree's manifest file set is exactly the full set — never the + // narrow sibling's subset. This file-set comparison proves no incomplete model + // is ever served as complete via sibling-reuse. + broadPaths := make([]string, 0, len(broad.Manifest.Files)) + for _, f := range broad.Manifest.Files { + broadPaths = append(broadPaths, f.Path) + } + Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"})) + + // Every file is present on disk with the right bytes after commit. + for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} { + Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))). + To(Equal(contents[p])) + } + }) +}) From 0e52bb657e8c58f68fba24c62589f46466958103 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:24 +0200 Subject: [PATCH 38/42] fix(responses): wait for complete JSON tool calls (#12001) Partial JSON parsing heals a name-only chunk into a tool call. The stream emits that call with empty arguments and skips later chunks. Require complete JSON before emitting terminal tool-call events. Preserve complete calls before an unfinished trailing call, and count only actual tool calls. Add split-chunk regression tests and docs. Refs #11635. The non-streaming report remains unconfirmed. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .../http/endpoints/openresponses/responses.go | 68 ++++++++----------- .../openresponses/stream_tool_calls.go | 35 ++++++++++ .../openresponses/stream_tool_calls_test.go | 44 ++++++++++++ docs/content/features/text-generation.md | 5 ++ 4 files changed, 112 insertions(+), 40 deletions(-) create mode 100644 core/http/endpoints/openresponses/stream_tool_calls.go create mode 100644 core/http/endpoints/openresponses/stream_tool_calls_test.go diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index 553c01558..6da7f2adc 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 return true } - // Try JSON parsing as fallback - jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true) - if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount { + // Only completed JSON calls can be emitted as completed SSE items. + jsonResults := parseStreamingJSONToolCalls(cleanedResult) + if len(jsonResults) > lastEmittedToolCallCount { for i := lastEmittedToolCallCount; i < len(jsonResults); i++ { - jsonObj := jsonResults[i] - if name, ok := jsonObj["name"].(string); ok && name != "" { - args := "{}" - if argsVal, ok := jsonObj["arguments"]; ok { - if argsStr, ok := argsVal.(string); ok { - args = argsStr - } else { - argsBytes, _ := json.Marshal(argsVal) - args = string(argsBytes) - } - } + tc := jsonResults[i] + toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) + outputIndex++ - toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) - outputIndex++ - - functionCallItem := &schema.ORItemField{ - Type: "function_call", - ID: toolCallID, - Status: "completed", - CallID: toolCallID, - Name: name, - Arguments: args, - } - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.added", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ - - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.done", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ + functionCallItem := &schema.ORItemField{ + Type: "function_call", + ID: toolCallID, + Status: "completed", + CallID: toolCallID, + Name: tc.Name, + Arguments: tc.Arguments, } + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.added", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ + + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.done", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ } lastEmittedToolCallCount = len(jsonResults) c.Response().Flush() diff --git a/core/http/endpoints/openresponses/stream_tool_calls.go b/core/http/endpoints/openresponses/stream_tool_calls.go new file mode 100644 index 000000000..b8f0f185d --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls.go @@ -0,0 +1,35 @@ +package openresponses + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/pkg/functions" +) + +func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults { + // Partial parsing heals unfinished arguments. The caller emits terminal + // events and never revisits emitted calls, so only accept complete JSON. + // Keep completed objects returned before an unfinished trailing object. + objects, _ := functions.ParseJSONIterative(text, false) + var calls []functions.FuncCallResults + for _, object := range objects { + name, ok := object["name"].(string) + if !ok || name == "" { + continue + } + arguments := "{}" + if value, ok := object["arguments"]; ok { + if s, ok := value.(string); ok { + arguments = s + } else { + data, err := json.Marshal(value) + if err != nil { + continue + } + arguments = string(data) + } + } + calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments}) + } + return calls +} diff --git a/core/http/endpoints/openresponses/stream_tool_calls_test.go b/core/http/endpoints/openresponses/stream_tool_calls_test.go new file mode 100644 index 000000000..1fa6bca2a --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls_test.go @@ -0,0 +1,44 @@ +package openresponses + +import ( + "github.com/mudler/LocalAI/pkg/functions" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Streaming JSON tool calls", func() { + It("waits for the arguments before completing a split call", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("does not complete a call at any intermediate token boundary", func() { + text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}` + for end := 1; end < len(text); end++ { + Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end]) + } + Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1)) + }) + + It("keeps completed calls while the next call is incomplete", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("preserves string arguments and calls that take no arguments", func() { + Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + {Name: "status", Arguments: `{}`}, + })) + }) + + It("does not count unrelated JSON objects as emitted calls", func() { + Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "status", Arguments: `{}`}, + })) + }) +}) diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 490877e21..286359339 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -434,6 +434,11 @@ curl http://localhost:8080/v1/responses \ }' ``` +For streaming requests with JSON tool output, LocalAI waits for the complete JSON +object before emitting a completed `function_call` item. Arguments can span +multiple tokens. Read the arguments from the `response.output_item.done` event +before executing the tool. + #### Reasoning Configuration Configure reasoning effort and summary style: From 5794495a37b0cdc186ac6a14648991c730042499 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:29 +0200 Subject: [PATCH 39/42] fix(responses): preserve streamed output items (#12048) Keep each message and reasoning item at its announced output index. Include the answer in completed responses with reasoning or fallback function calls, and retain reasoning supplied through backend deltas. Add regression coverage for stream indices, final output, plain text, and automatic tool parsing. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .../http/endpoints/openresponses/responses.go | 60 +++---- .../openresponses/responses_stream_test.go | 148 ++++++++++++++++++ docs/content/features/text-generation.md | 9 ++ 3 files changed, 176 insertions(+), 41 deletions(-) create mode 100644 core/http/endpoints/openresponses/responses_stream_test.go diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index 6da7f2adc..d62fa7534 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -2412,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Non-tool-call streaming path + messageOutputIndex := outputIndex + var reasoningOutputIndex int // Emit output_item.added for message currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String()) messageItem := &schema.ORItemField{ @@ -2424,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2436,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &emptyTextPart, }) @@ -2459,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Handle reasoning item - if extractor.Reasoning() != "" { + if extractor.Reasoning() != "" || reasoningDelta != "" { // Check if we need to create reasoning item if currentReasoningID == "" { outputIndex++ + reasoningOutputIndex = outputIndex currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String()) reasoningItem := &schema.ORItemField{ Type: "reasoning", @@ -2472,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2484,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &emptyPart, }) @@ -2497,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Delta: strPtr(reasoningDelta), Logprobs: emptyLogprobs(), @@ -2514,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Delta: strPtr(contentDelta), Logprobs: emptyLogprobs(), @@ -2583,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Text: strPtr(finalReasoning), Logprobs: emptyLogprobs(), @@ -2596,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &reasoningPart, }) @@ -2612,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2646,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Text: strPtr(result), Logprobs: logprobsPtr(mcpStreamLogprobs), @@ -2659,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &resultPart, }) @@ -2671,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2711,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 // Emit response.completed now := time.Now().Unix() - // Collect final output items (reasoning first, then messages, then tool calls) - var finalOutputItems []schema.ORItemField - // Add reasoning item if it exists - if currentReasoningID != "" && finalReasoning != "" { - finalOutputItems = append(finalOutputItems, schema.ORItemField{ - Type: "reasoning", - ID: currentReasoningID, - Status: "completed", - Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)}, - }) - } - // Add message item - if len(collectedOutputItems) > 0 { - // Use collected items (may include reasoning already) - for _, item := range collectedOutputItems { - if item.Type == "message" { - finalOutputItems = append(finalOutputItems, item) - } - } - } else { - finalOutputItems = append(finalOutputItems, *messageItem) - } - // Add function_call items from fallback - for _, item := range collectedOutputItems { - if item.Type == "function_call" { - finalOutputItems = append(finalOutputItems, item) - } - } + // The final output array must use the indices announced in the stream. + // The message is opened first, followed by reasoning and fallback calls. + finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...) responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{ InputTokens: noToolTokenUsage.Prompt, OutputTokens: noToolTokenUsage.Completion, diff --git a/core/http/endpoints/openresponses/responses_stream_test.go b/core/http/endpoints/openresponses/responses_stream_test.go new file mode 100644 index 000000000..13ddb10e1 --- /dev/null +++ b/core/http/endpoints/openresponses/responses_stream_test.go @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: MIT +package openresponses + +import ( + "context" + "encoding/json" + "net/http/httptest" + "strings" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/schema" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/LocalAI/pkg/model" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Responses stream item consistency", func() { + DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) { + originalInference := backend.ModelInferenceFunc + DeferCleanup(func() { backend.ModelInferenceFunc = originalInference }) + backend.ModelInferenceFunc = func( + ctx context.Context, prompt string, messages schema.Messages, + images, videos, audios []string, loader *model.ModelLoader, + cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig, + tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string, + logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string, + ) (func() (backend.LLMResponse, error), error) { + return func() (backend.LLMResponse, error) { + for i, token := range tokens { + usage := backend.TokenUsage{} + if len(chatDeltas) > 0 { + usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]} + } + if !tokenCallback(token, usage) { + break + } + } + return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil + }, nil + } + cfg := &config.ModelConfig{} + cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback + cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)(.*?)`} + recorder := httptest.NewRecorder() + request := httptest.NewRequest("POST", "/v1/responses", nil) + c := echo.New().NewContext(request, recorder) + input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true} + err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil) + Expect(err).NotTo(HaveOccurred()) + Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n")) + + var events []schema.ORStreamEvent + var completed *schema.ORResponseResource + for _, line := range strings.Split(recorder.Body.String(), "\n") { + if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" { + continue + } + var event schema.ORStreamEvent + Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed()) + Expect(event.Type).NotTo(Equal("error")) + events = append(events, event) + if event.Type == "response.completed" { + completed = event.Response + } + } + Expect(completed).NotTo(BeNil()) + wantCount := 1 + if wantReasoning != "" { + wantCount++ + } + if fallback { + wantCount++ + } + Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls") + + indices := map[string]int{} + done := map[string]int{} + deltas := map[string]string{} + for i, event := range events { + Expect(event.SequenceNumber).To(Equal(i)) + if event.Type == "response.output_item.added" { + Expect(event.Item).NotTo(BeNil()) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(indices).NotTo(HaveKey(event.Item.ID)) + Expect(*event.OutputIndex).To(Equal(len(indices))) + indices[event.Item.ID] = *event.OutputIndex + } + id := event.ItemID + if event.Item != nil { + id = event.Item.ID + } + if id == "" { + continue + } + Expect(indices).To(HaveKey(id)) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id) + Expect(completed.Output[indices[id]].ID).To(Equal(id)) + if event.Type == "response.output_item.done" { + done[id]++ + Expect(event.Item.Status).To(Equal("completed")) + Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type)) + if event.Item.Type == "function_call" { + Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name)) + Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments)) + } else { + Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content)) + } + } + if event.Type == "response.output_text.delta" { + deltas[id] += *event.Delta + } + } + Expect(indices).To(HaveLen(wantCount)) + for _, item := range completed.Output { + Expect(done[item.ID]).To(Equal(1)) + switch item.Type { + case "message", "reasoning": + want := wantAnswer + if item.Type == "reasoning" { + want = wantReasoning + } + parts, ok := item.Content.([]any) + Expect(ok).To(BeTrue()) + Expect(parts).To(HaveLen(1)) + Expect(parts[0].(map[string]any)["text"]).To(Equal(want)) + if !fallback { + Expect(deltas[item.ID]).To(Equal(want)) + } + case "function_call": + Expect(item.Name).To(Equal("get_weather")) + Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`)) + Expect(item.CallID).NotTo(BeEmpty()) + default: + Fail("unexpected output item type: " + item.Type) + } + } + }, + Entry("tagged reasoning and answer", []string{"", "Let me think.", "", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false), + Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false), + Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false), + Entry("automatic fallback tool call", []string{`{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "", "", true), + Entry("reasoning and automatic fallback tool call", []string{"", "Let me think.", "", `{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "Let me think.", "", true), + ) +}) diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 286359339..dadb5e9af 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -340,6 +340,15 @@ curl http://localhost:8080/v1/responses \ }' ``` +#### Streaming responses + +Set `"stream": true` to receive Server-Sent Events. Each `response.output_item.added` event assigns an `output_index` to an item. +Use that index and the item ID to associate later deltas and completion events with the same item. + +If a request without explicit tools produces reasoning, the stream uses separate items for reasoning and answer text. +Each item keeps its original index throughout the stream. +The `response.completed` event includes both items in the same index order, followed by any automatically parsed tool calls. + #### Background Processing Run requests in the background for long-running tasks: From f154bd990a720130959e02c8eeea4e571e27117b Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:34 +0200 Subject: [PATCH 40/42] feat(system): report per-model DRM VRAM (#12026) * feat(system): report per-model DRM VRAM Expose optional resident device memory for local backend process trees. Deduplicate DRM clients and omit unsupported or incomplete readings. Document accounting limits and preserve a measured zero in JSON. Closes #11970. Assisted-by: Codex:gpt-6 * fix(system): document trusted procfs reads Scope G304 annotations to paths built from the fixed procfs root, integer process IDs, and kernel directory entries. These reads accept no user-controlled path components. Assisted-by: Codex:GPT-6 gosec --------- Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- core/http/endpoints/localai/system.go | 6 + .../endpoints/localai/system_info_test.go | 49 ++++++ core/schema/localai.go | 3 + core/schema/system_info_test.go | 25 +++ docs/content/reference/system-info.md | 24 +++ pkg/xsysinfo/process_vram_linux.go | 165 ++++++++++++++++++ pkg/xsysinfo/process_vram_linux_test.go | 105 +++++++++++ pkg/xsysinfo/process_vram_other.go | 9 + swagger/docs.go | 4 + swagger/swagger.json | 4 + swagger/swagger.yaml | 5 + 11 files changed, 399 insertions(+) create mode 100644 core/http/endpoints/localai/system_info_test.go create mode 100644 core/schema/system_info_test.go create mode 100644 pkg/xsysinfo/process_vram_linux.go create mode 100644 pkg/xsysinfo/process_vram_linux_test.go create mode 100644 pkg/xsysinfo/process_vram_other.go diff --git a/core/http/endpoints/localai/system.go b/core/http/endpoints/localai/system.go index 996c9a781..9c4ba7503 100644 --- a/core/http/endpoints/localai/system.go +++ b/core/http/endpoints/localai/system.go @@ -8,6 +8,7 @@ import ( "github.com/mudler/LocalAI/core/schema" "github.com/mudler/LocalAI/core/services/monitoring" "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/xsysinfo" ) // SystemInformations returns the system informations @@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app entry.Process = proc } } + if pid, ok := localPID(m); ok { + if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok { + entry.SizeVRAM = &used + } + } sysmodels = append(sysmodels, entry) } if sampler != nil { diff --git a/core/http/endpoints/localai/system_info_test.go b/core/http/endpoints/localai/system_info_test.go new file mode 100644 index 000000000..83f7daa07 --- /dev/null +++ b/core/http/endpoints/localai/system_info_test.go @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: MIT +package localai_test + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/http/endpoints/localai" + "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/system" + process "github.com/mudler/go-processmanager" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SystemInformations memory", func() { + It("keeps model metadata and omits VRAM for remote or stopped backends", func() { + path, err := os.MkdirTemp("", "system-info-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, path) + configFile := filepath.Join(path, "remote.yaml") + Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed()) + cl := config.NewModelConfigLoader(path) + Expect(cl.ReadModelConfig(configFile)).To(Succeed()) + ml := model.NewModelLoader(&system.SystemState{}) + store := model.NewInMemoryModelStore() + store.Set("remote", model.NewModel("remote", "worker:50051", nil)) + store.Set("stopped", model.NewModel("stopped", "", &process.Process{})) + ml.SetModelStore(store) + app := echo.New() + app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil)) + rec := httptest.NewRecorder() + app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil)) + Expect(rec.Code).To(Equal(http.StatusOK)) + var response struct { + Models []map[string]any `json:"loaded_models"` + } + Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed()) + Expect(response.Models).To(ConsistOf( + map[string]any{"id": "remote", "backend": "llama-cpp"}, + map[string]any{"id": "stopped"}, + )) + }) +}) diff --git a/core/schema/localai.go b/core/schema/localai.go index 7e5d5e314..dc99a1dbe 100644 --- a/core/schema/localai.go +++ b/core/schema/localai.go @@ -208,6 +208,9 @@ type SysInfoModel struct { // when the model has no local process (a distributed worker holds it) or // the process could not be read. Process *SysInfoProcess `json:"process,omitempty"` + // SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + // the backend process tree has no complete supported reading. + SizeVRAM *uint64 `json:"size_vram,omitempty"` } // SysInfoProcess is a point-in-time reading of one backend process. diff --git a/core/schema/system_info_test.go b/core/schema/system_info_test.go new file mode 100644 index 000000000..79a1bdba1 --- /dev/null +++ b/core/schema/system_info_test.go @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: MIT +package schema_test + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/core/schema" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SysInfoModel memory", func() { + It("omits unavailable VRAM while preserving a measured zero", func() { + entry := schema.SysInfoModel{ID: "model"} + encoded, err := json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`)) + + zero := uint64(0) + entry.SizeVRAM = &zero + encoded, err = json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`)) + }) +}) diff --git a/docs/content/reference/system-info.md b/docs/content/reference/system-info.md index b825e4e06..ed4de00ff 100644 --- a/docs/content/reference/system-info.md +++ b/docs/content/reference/system-info.md @@ -28,6 +28,29 @@ Returns available backends and currently loaded models. | `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM | | `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against | | `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) | +| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes | + +### Per-model VRAM + +On Linux, `size_vram` reports resident device memory for the local backend +process and its child processes. LocalAI reads `drm-resident-local*` and +`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU. +Host-memory regions are excluded. The reading includes buffers attributed +to the backend, without separating weights, KV cache, and other allocations. +See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html) +for these counters. + +The field is omitted when accounting is unavailable or incomplete. This +includes external and distributed backends, macOS, proprietary NVIDIA +drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters, +and unreadable process information. +A present value of `0` means the supported counters report zero bytes. +Treat an absent field as unknown. + +This is a snapshot of driver accounting, not a memory reservation. Shared +buffers can appear in different clients' counters, and allocations can change +during collection. Do not treat the sum across models as exclusive physical +GPU usage. These readings do not replace capacity checks when scheduling work. ### Usage @@ -49,6 +72,7 @@ curl http://localhost:8080/system { "id": "my-llama-model", "backend": "llama-cpp", + "size_vram": 5368709120, "process": { "pid": 48213, "rss_bytes": 5368709120, diff --git a/pkg/xsysinfo/process_vram_linux.go b/pkg/xsysinfo/process_vram_linux.go new file mode 100644 index 000000000..3dd0a59a9 --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux.go @@ -0,0 +1,165 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "bufio" + "bytes" + "math" + "os" + "path/filepath" + "strconv" + "strings" +) + +// ProcessVRAM reports device-local resident bytes accounted to a process tree +// by DRM. Unsupported or incomplete accounting returns false, not a measured zero. +func ProcessVRAM(pid int) (uint64, bool) { + return processVRAM("/proc", pid) +} + +func processVRAM(procRoot string, pid int) (uint64, bool) { + if pid <= 0 { + return 0, false + } + clients := map[string]uint64{} + seen := map[int]bool{} + pending := []int{pid} + for len(pending) > 0 { + current := pending[len(pending)-1] + pending = pending[:len(pending)-1] + if seen[current] { + continue + } + seen[current] = true + base := filepath.Join(procRoot, strconv.Itoa(current)) + fds, err := os.ReadDir(filepath.Join(base, "fd")) + if err != nil { + return 0, false + } + for _, fd := range fds { + target, err := os.Readlink(filepath.Join(base, "fd", fd.Name())) + if err != nil { + return 0, false + } + // A mixed DRM/NVIDIA tree cannot provide a complete DRM reading. + if strings.HasPrefix(target, "/dev/nvidia") { + return 0, false + } + if !strings.HasPrefix(target, "/dev/dri/render") { + // Primary nodes can also own allocations. Until their device + // identity is resolved, omitting them would undercount the tree. + if strings.HasPrefix(target, "/dev/dri/") { + return 0, false + } + continue + } + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and fd.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name())) + if err != nil { + return 0, false + } + client, used, ok := drmResidentClient(data) + if !ok { + return 0, false + } + key := target + ":" + client + // dup() and fork() can expose the same client more than once. The + // snapshot is not atomic; retain its largest observed reading. + clients[key] = max(clients[key], used) + } + + // A worker may be spawned by any thread, not just the thread leader. + tasks, err := os.ReadDir(filepath.Join(base, "task")) + if err != nil || len(tasks) == 0 { + return 0, false + } + for _, task := range tasks { + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and task.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children")) + if err != nil { + return 0, false + } + for _, raw := range strings.Fields(string(data)) { + child, err := strconv.Atoi(raw) + if err != nil || child <= 0 { + return 0, false + } + pending = append(pending, child) + } + } + } + var total uint64 + for _, used := range clients { + if used > math.MaxUint64-total { + return 0, false + } + total += used + } + return total, len(clients) > 0 +} + +func drmResidentClient(data []byte) (string, uint64, bool) { + var client string + var total uint64 + found := false + scanner := bufio.NewScanner(bytes.NewReader(data)) + for scanner.Scan() { + key, value, ok := strings.Cut(scanner.Text(), ":") + if !ok { + continue + } + if key == "drm-client-id" { + id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + if err != nil { + return "", 0, false + } + client = strconv.FormatUint(id, 10) + } + region, resident := strings.CutPrefix(key, "drm-resident-") + if !resident || !isVRAMRegion(region) { + continue + } + used, ok := drmResidentBytes(value) + if !ok || used > math.MaxUint64-total { + return "", 0, false + } + total += used + found = true + } + return client, total, scanner.Err() == nil && client != "" && found +} + +func drmResidentBytes(value string) (uint64, bool) { + fields := strings.Fields(value) + if len(fields) == 0 || len(fields) > 2 { + return 0, false + } + n, err := strconv.ParseUint(fields[0], 10, 64) + if err != nil { + return 0, false + } + unit := uint64(1) + if len(fields) == 2 { + switch strings.ToLower(fields[1]) { + case "b": + case "kib": + unit = 1 << 10 + case "mib": + unit = 1 << 20 + case "gib": + unit = 1 << 30 + default: + return 0, false + } + } + if n > math.MaxUint64/unit { + return 0, false + } + return n * unit, true +} diff --git a/pkg/xsysinfo/process_vram_linux_test.go b/pkg/xsysinfo/process_vram_linux_test.go new file mode 100644 index 000000000..4de1f6cdd --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux_test.go @@ -0,0 +1,105 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "os" + "path/filepath" + "strconv" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("ProcessVRAM", func() { + var root string + write := func(path, contents string) { + Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed()) + Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed()) + } + addProcess := func(pid int, children string) { + base := filepath.Join(root, strconv.Itoa(pid)) + Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed()) + write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children) + } + addFD := func(pid, fd int, render, info string) { + base := filepath.Join(root, strconv.Itoa(pid)) + name := strconv.Itoa(fd) + Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed()) + write(filepath.Join(base, "fdinfo", name), info) + } + BeforeEach(func() { + var err error + root, err = os.MkdirTemp("", "process-vram-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, root) + addProcess(100, "") + }) + + It("sums resident device memory across GPUs and child processes without duplicate clients", func() { + write(filepath.Join(root, "100/task/101/children"), "200") + addProcess(200, "") + info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n" + addFD(100, 3, "renderD128", info) + addFD(100, 4, "renderD128", info) + addFD(200, 3, "renderD128", info) + addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(Equal(uint64(384 * 1024 * 1024))) + }) + + It("distinguishes a measured zero from unavailable accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(BeZero()) + }) + + DescribeTable("does not invent readings from unsupported or invalid accounting", + func(info string) { + addFD(100, 3, "renderD128", info) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"), + Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"), + Entry("no client identity", "drm-resident-vram0: 128 MiB\n"), + Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"), + Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"), + Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"), + ) + + It("omits a partial reading if a child cannot be inspected", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + write(filepath.Join(root, "100/task/100/children"), "200") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + It("omits a partial reading if another DRM client lacks accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + addFD(100, 4, "renderD129", "drm-client-id: 8\n") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + DescribeTable("omits mixed readings with unsupported GPU descriptors", + func(target string) { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed()) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("primary DRM node", "/dev/dri/card0"), + Entry("NVIDIA device", "/dev/nvidia0"), + ) + + It("returns unavailable for missing processes or no DRM descriptors", func() { + for _, pid := range []int{-1, 0, 100, 999} { + _, ok := processVRAM(root, pid) + Expect(ok).To(BeFalse()) + } + }) +}) diff --git a/pkg/xsysinfo/process_vram_other.go b/pkg/xsysinfo/process_vram_other.go new file mode 100644 index 000000000..06cc49186 --- /dev/null +++ b/pkg/xsysinfo/process_vram_other.go @@ -0,0 +1,9 @@ +//go:build !linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting. +func ProcessVRAM(pid int) (uint64, bool) { + return 0, false +} diff --git a/swagger/docs.go b/swagger/docs.go index dc2fcf063..c447043ae 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -7810,6 +7810,10 @@ const docTemplate = `{ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.json b/swagger/swagger.json index a1e71cbdb..b4e49b347 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -7807,6 +7807,10 @@ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 12de303a0..34fcfc8d7 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2654,6 +2654,11 @@ definitions: type: string id: type: string + size_vram: + description: |- + SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + the backend process tree has no complete supported reading. + type: integer type: object schema.SystemInformationResponse: properties: From 490b952d062dd13f1ba5bcc7eef297979cdc781d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sun, 27 Sep 2026 21:18:38 +0200 Subject: [PATCH 41/42] feat(gallery): publish signed OCI fallbacks (#12182) * feat(gallery): publish signed OCI fallbacks Publish both official gallery indexes with their local base configs so an outage of the HTTP and GitHub sources can fall back to Quay. Keep artifact signing policies separate from backend image policies, and expose each moving gallery tag only after its digest is signed. Assisted-by: Codex:gpt-6 * fix(gallery): confine packaged files to selected roots Use directory-scoped file access to reject symlink escapes during gallery packaging. Create private bundle files for the publishing runner. Assisted-by: Codex:GPT-6 --------- Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- .github/workflows/gallery_publish.yml | 78 ++++++++++++++++ core/config/gallery.go | 12 ++- core/config/gallery_test.go | 21 +++++ core/config/runtime_settings_startup.go | 4 +- core/config/runtime_settings_startup_test.go | 12 ++- core/gallery/entry_url.go | 2 +- core/gallery/gallery.go | 2 +- core/gallery/gallery_mirrors.go | 4 +- core/gallery/gallery_oci.go | 18 ++-- core/gallery/gallery_oci_test.go | 15 ++++ docs/content/features/backends.md | 2 + docs/content/features/model-gallery.md | 20 +++-- scripts/build/gallery/main.go | 94 ++++++++++++++++++++ scripts/build/gallery/main_test.go | 83 +++++++++++++++++ 14 files changed, 345 insertions(+), 22 deletions(-) create mode 100644 .github/workflows/gallery_publish.yml create mode 100644 scripts/build/gallery/main.go create mode 100644 scripts/build/gallery/main_test.go diff --git a/.github/workflows/gallery_publish.yml b/.github/workflows/gallery_publish.yml new file mode 100644 index 000000000..c27b3d282 --- /dev/null +++ b/.github/workflows/gallery_publish.yml @@ -0,0 +1,78 @@ +name: Publish official OCI galleries + +on: + push: + branches: [master] + paths: + - 'gallery/**' + - 'backend/index.yaml' + - 'scripts/build/gallery/**' + - '.github/workflows/gallery_publish.yml' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: publish-official-galleries + cancel-in-progress: false + +jobs: + publish: + if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master' + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write + env: + COSIGN_EXPERIMENTAL: '1' + GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends + strategy: + matrix: + include: + - source: gallery + tag: gallery-models + - source: backend + tag: gallery-backends + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-go@v6 + with: + go-version-file: go.mod + - name: Test and package gallery + env: + GALLERY_SOURCE: ${{ matrix.source }} + run: | + go test ./scripts/build/gallery -count=1 + go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery" + - uses: oras-project/setup-oras@v1 + with: + version: '1.3.0' + - uses: sigstore/cosign-installer@v3 + with: + cosign-release: 'v2.6.5' + - name: Login to Quay.io + uses: docker/login-action@v4 + with: + registry: quay.io + username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }} + password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }} + - name: Publish and sign gallery + shell: bash + env: + GALLERY_TAG: ${{ matrix.tag }} + run: | + set -euo pipefail + cd "$RUNNER_TEMP/gallery" + files=() + while IFS= read -r -d '' file; do + files+=("${file#./}:application/yaml") + done < <(find . -type f -print0 | sort -z) + # Publish an immutable revision, then expose latest only after signing. + ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA" + oras push --artifact-type application/vnd.localai.gallery.v1 \ + --format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json" + digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json") + cosign sign --yes --new-bundle-format \ + --registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest" + oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG" diff --git a/core/config/gallery.go b/core/config/gallery.go index e22cbc94f..3f6b31ab9 100644 --- a/core/config/gallery.go +++ b/core/config/gallery.go @@ -47,10 +47,13 @@ type Gallery struct { // fallback for availability, not a load-balancing pool: the primary is // always preferred, and a mirror is only consulted after the one before // it fails. Any URI the gallery loader understands works here - // (https://, github:, file://). + // (https://, github:, file://, oci://). Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"` Name string `json:"name" yaml:"name"` Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"` + // ArtifactVerification overrides Verification only for the gallery OCI artifact. + // Backend images keep their separate Verification policy. + ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"` } // Equal reports whether two gallery entries describe the same gallery. @@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool { if !slices.Equal(g.Mirrors, other.Mirrors) { return false } + if g.ArtifactVerification == nil || other.ArtifactVerification == nil { + if g.ArtifactVerification != other.ArtifactVerification { + return false + } + } else if *g.ArtifactVerification != *other.ArtifactVerification { + return false + } if g.Verification == nil || other.Verification == nil { return g.Verification == other.Verification } diff --git a/core/config/gallery_test.go b/core/config/gallery_test.go index 71f4f3a36..7e1848a36 100644 --- a/core/config/gallery_test.go +++ b/core/config/gallery_test.go @@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() { Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery")) }) }) + +var _ = Describe("Gallery artifact verification", func() { + It("compares artifact policies by value and preserves them in JSON and YAML", func() { + a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + Expect(a.Equal(b)).To(BeTrue()) + b.ArtifactVerification.Identity = "another-workflow" + Expect(a.Equal(b)).To(BeFalse()) + b.ArtifactVerification = nil + Expect(a.Equal(b)).To(BeFalse()) + raw, err := json.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + Expect(json.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + raw, err = yaml.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + b = config.Gallery{} + Expect(yaml.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + }) +}) diff --git a/core/config/runtime_settings_startup.go b/core/config/runtime_settings_startup.go index 9808c877b..e5d7e2a46 100644 --- a/core/config/runtime_settings_startup.go +++ b/core/config/runtime_settings_startup.go @@ -17,8 +17,8 @@ import ( // a caching mirror of the files below. The GitHub URI stays as a mirror so an // install still resolves its gallery unchanged whenever the primary is // unreachable - see the fallback chain in core/gallery/gallery_mirrors.go. -const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]` -const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]` +const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` +const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` func mustGalleries(jsonList string) []Gallery { var g []Gallery diff --git a/core/config/runtime_settings_startup_test.go b/core/config/runtime_settings_startup_test.go index 410e457d6..d49c53ac4 100644 --- a/core/config/runtime_settings_startup_test.go +++ b/core/config/runtime_settings_startup_test.go @@ -10,22 +10,22 @@ import ( ) var _ = Describe("default galleries", func() { - It("serves the model gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/models")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"})) }) - It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"})) }) // The mirror is the whole reason this default is safe to ship: if @@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() { Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed()) for _, g := range galleries { Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name) + Expect(g.ArtifactVerification).ToNot(BeNil()) + Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master")) + Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com")) + Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust") } } }) diff --git a/core/gallery/entry_url.go b/core/gallery/entry_url.go index 184ce2dd9..49f4f720d 100644 --- a/core/gallery/entry_url.go +++ b/core/gallery/entry_url.go @@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string { if !looksLikeOCIGallery(candidate) { continue } - dir := ociGalleryCacheDir(basePath, candidate, g.Verification) + dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g)) if dir == "" { continue } diff --git a/core/gallery/gallery.go b/core/gallery/gallery.go index 68d6de9e8..d0c0f43e0 100644 --- a/core/gallery/gallery.go +++ b/core/gallery/gallery.go @@ -646,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]() // would also point relative entry urls at an unpacked tree the new policy has // not produced yet, so they could not be installed. func galleryIndexCacheKey(g config.Gallery) string { - return g.Name + "-" + galleryCacheName(g.URL, g.Verification) + return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g)) } func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) { diff --git a/core/gallery/gallery_mirrors.go b/core/gallery/gallery_mirrors.go index 4058c0a45..083512417 100644 --- a/core/gallery/gallery_mirrors.go +++ b/core/gallery/gallery_mirrors.go @@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { if !looksLikeOCIGallery(g.URL) { return nil } - return g.Verification + return galleryArtifactPolicy(g) } // verifiableCandidates drops the candidates that cannot answer for a signed @@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { // at, and after a refusal it would turn "this artifact is not trusted" into // "use this other, unchecked copy instead". func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string { - if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) { + if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) { return candidates } out := make([]string, 0, len(candidates)) diff --git a/core/gallery/gallery_oci.go b/core/gallery/gallery_oci.go index 19d6d16c0..cbd649229 100644 --- a/core/gallery/gallery_oci.go +++ b/core/gallery/gallery_oci.go @@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) { // later fetch served would hand the user a truncated gallery with no sign that // anything went wrong. func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) { + policy := galleryArtifactPolicy(g) // Checked before the cache: a copy unpacked while strict integrity was // off was never verified, and turning strict integrity on must not keep // serving it for the rest of its TTL. - if g.Verification == nil && requireIntegrity { + if policy == nil && requireIntegrity { return nil, &galleryVerificationError{ strict: true, - err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate), + err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate), } } - cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification) + cacheDir := ociGalleryCacheDir(basePath, candidate, policy) if cacheDir == "" { return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate) } @@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base pullRef := downloader.URI(candidate).OCIReference() - if g.Verification != nil { + if policy != nil { // Resolve first, verify the digest, then pull that same digest. // Nothing has been fetched at this point beyond the manifest, so a // policy failure leaves no content anywhere. @@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base if err != nil { return nil, err } - if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil { + if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil { // Only a decision about the artifact is a refusal. The // verifier also reaches the Sigstore TUF mirror and the // registry, and a timeout or a 5xx there says nothing about @@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base return body, nil } + +func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification { + if g.ArtifactVerification != nil { + return g.ArtifactVerification + } + return g.Verification +} diff --git a/core/gallery/gallery_oci_test.go b/core/gallery/gallery_oci_test.go index 5672da123..4c038b5a2 100644 --- a/core/gallery/gallery_oci_test.go +++ b/core/gallery/gallery_oci_test.go @@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() { }) }) + It("uses the artifact policy without replacing backend image verification", func() { + srv, _, _ := ociRegistry() + url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}}) + backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"} + artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"} + var seen *config.GalleryVerification + stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil }) + g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy} + _, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true) + Expect(source).To(Equal(url)) + Expect(err).ToNot(HaveOccurred()) + Expect(seen).To(Equal(artifactPolicy)) + Expect(g.Verification).To(Equal(backendPolicy)) + }) + It("refuses an unsigned gallery in strict integrity mode", func() { srv, _, blobs := ociRegistry() url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{ diff --git a/docs/content/features/backends.md b/docs/content/features/backends.md index 6085cf536..c0818df85 100644 --- a/docs/content/features/backends.md +++ b/docs/content/features/backends.md @@ -82,6 +82,8 @@ tags: ### Verifying OCI Backends +The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details. + Backend galleries can require keyless Sigstore signatures for every OCI image they provide. Add a `verification` policy to the gallery configuration, then enable strict integrity mode: diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..58aa920f9 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -97,7 +97,7 @@ To use a gallery that needs authentication, such as a private GitHub repository A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted. -Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. +Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. ```json GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}] @@ -151,10 +151,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out ### Signature verification -An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: +An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: ```json -GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] +GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] ``` The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends. @@ -173,9 +173,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}). {{% notice warning %}} -With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. +`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities. + +With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. {{% /notice %}} +### Official gallery publishing + +The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA. + +To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title. + ### Private registries A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use: @@ -207,10 +215,10 @@ GALLERIES=[{"name":"", "url":" Date: Sun, 27 Sep 2026 21:38:15 +0200 Subject: [PATCH 42/42] fix(kokoros): add missing animate3_d stub to Backend trait impl (#12301) * fix(kokoros): add missing animate3_d stub to Backend trait impl #12095 added the Animate3D RPC to backend.proto, but the kokoros service never got a matching method. The tonic-generated Backend trait now requires it, so kokoros fails to build with E0046 whenever the full backend matrix runs. Return Unimplemented, as the other unsupported RPCs do. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] * fix(kokoros): fill new Result fields with defaults backend.proto added a metadata field to Result, so the struct literals in the kokoros service no longer name every field and fail to compile. Spread Default::default() into them, so later additive proto fields do not break the build again. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- backend/rust/kokoros/src/service.rs | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/backend/rust/kokoros/src/service.rs b/backend/rust/kokoros/src/service.rs index aeddbf107..a97cc4bf5 100644 --- a/backend/rust/kokoros/src/service.rs +++ b/backend/rust/kokoros/src/service.rs @@ -132,6 +132,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Kokoros TTS model loaded".into(), + ..Default::default() })) } @@ -180,11 +181,13 @@ impl Backend for KokorosService { return Ok(Response::new(backend::Result { success: false, message: format!("Failed to write WAV: {}", e), + ..Default::default() })); } Ok(Response::new(backend::Result { success: true, message: String::new(), + ..Default::default() })) } Err(e) => { @@ -192,6 +195,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: false, message: format!("TTS error: {}", e), + ..Default::default() })) } } @@ -292,6 +296,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Model freed".into(), + ..Default::default() })) } @@ -348,6 +353,13 @@ impl Backend for KokorosService { Err(Status::unimplemented("Not supported")) } + async fn animate3_d( + &self, + _: Request, + ) -> Result, Status> { + Err(Status::unimplemented("Not supported")) + } + async fn audio_transcription( &self, _: Request,