diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml
index 13e67c6fe..d4bb0bf59 100644
--- a/.github/workflows/backend.yml
+++ b/.github/workflows/backend.yml
@@ -355,7 +355,7 @@ jobs:
with:
backend: ${{ matrix.backend }}
build-type: ${{ matrix.build-type }}
- go-version: "1.25.x"
+ go-version: "1.27.x"
tag-suffix: ${{ matrix.tag-suffix }}
lang: ${{ matrix.lang || 'python' }}
use-pip: ${{ matrix.backend == 'diffusers' }}
diff --git a/.github/workflows/backend_build.yml b/.github/workflows/backend_build.yml
index 05d50cf82..3e3af89f0 100644
--- a/.github/workflows/backend_build.yml
+++ b/.github/workflows/backend_build.yml
@@ -252,7 +252,8 @@ jobs:
name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }}
path: /tmp/digests/*
if-no-files-found: error
- retention-days: 1
+ # Release matrices and their retries can outlive a one-day artifact.
+ retention-days: 7
- name: Build (PR)
uses: docker/build-push-action@v7
diff --git a/.github/workflows/backend_build_darwin.yml b/.github/workflows/backend_build_darwin.yml
index 6b8b2a89b..19952a5ef 100644
--- a/.github/workflows/backend_build_darwin.yml
+++ b/.github/workflows/backend_build_darwin.yml
@@ -22,7 +22,8 @@ on:
type: string
go-version:
description: 'Go version to use'
- default: '1.24.x'
+ # Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs.
+ default: '1.27.x'
type: string
tag-suffix:
description: 'Tag suffix for the built image'
diff --git a/.github/workflows/backend_pr.yml b/.github/workflows/backend_pr.yml
index c13c444c4..2626f87e8 100644
--- a/.github/workflows/backend_pr.yml
+++ b/.github/workflows/backend_pr.yml
@@ -281,7 +281,7 @@ jobs:
with:
backend: ${{ matrix.backend }}
build-type: ${{ matrix.build-type }}
- go-version: "1.25.x"
+ go-version: "1.27.x"
tag-suffix: ${{ matrix.tag-suffix }}
lang: ${{ matrix.lang || 'python' }}
use-pip: ${{ matrix.backend == 'diffusers' }}
diff --git a/.github/workflows/gallery_publish.yml b/.github/workflows/gallery_publish.yml
new file mode 100644
index 000000000..c27b3d282
--- /dev/null
+++ b/.github/workflows/gallery_publish.yml
@@ -0,0 +1,78 @@
+name: Publish official OCI galleries
+
+on:
+ push:
+ branches: [master]
+ paths:
+ - 'gallery/**'
+ - 'backend/index.yaml'
+ - 'scripts/build/gallery/**'
+ - '.github/workflows/gallery_publish.yml'
+ workflow_dispatch:
+
+permissions:
+ contents: read
+
+concurrency:
+ group: publish-official-galleries
+ cancel-in-progress: false
+
+jobs:
+ publish:
+ if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master'
+ runs-on: ubuntu-latest
+ permissions:
+ contents: read
+ id-token: write
+ env:
+ COSIGN_EXPERIMENTAL: '1'
+ GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends
+ strategy:
+ matrix:
+ include:
+ - source: gallery
+ tag: gallery-models
+ - source: backend
+ tag: gallery-backends
+ steps:
+ - uses: actions/checkout@v7
+ - uses: actions/setup-go@v6
+ with:
+ go-version-file: go.mod
+ - name: Test and package gallery
+ env:
+ GALLERY_SOURCE: ${{ matrix.source }}
+ run: |
+ go test ./scripts/build/gallery -count=1
+ go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery"
+ - uses: oras-project/setup-oras@v1
+ with:
+ version: '1.3.0'
+ - uses: sigstore/cosign-installer@v3
+ with:
+ cosign-release: 'v2.6.5'
+ - name: Login to Quay.io
+ uses: docker/login-action@v4
+ with:
+ registry: quay.io
+ username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+ password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+ - name: Publish and sign gallery
+ shell: bash
+ env:
+ GALLERY_TAG: ${{ matrix.tag }}
+ run: |
+ set -euo pipefail
+ cd "$RUNNER_TEMP/gallery"
+ files=()
+ while IFS= read -r -d '' file; do
+ files+=("${file#./}:application/yaml")
+ done < <(find . -type f -print0 | sort -z)
+ # Publish an immutable revision, then expose latest only after signing.
+ ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA"
+ oras push --artifact-type application/vnd.localai.gallery.v1 \
+ --format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json"
+ digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json")
+ cosign sign --yes --new-bundle-format \
+ --registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest"
+ oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG"
diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile
index 85da233bd..e8d27eb52 100644
--- a/backend/cpp/audio-cpp/Makefile
+++ b/backend/cpp/audio-cpp/Makefile
@@ -9,7 +9,7 @@
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
# rebuild and so the bump bot can see the pin.
-AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22
+AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile
index 0f2a84e7d..d6bfd7490 100644
--- a/backend/cpp/ik-llama-cpp/Makefile
+++ b/backend/cpp/ik-llama-cpp/Makefile
@@ -1,5 +1,5 @@
-IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc
+IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
CMAKE_ARGS?=
diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile
index b672f2d31..8bf0ff5c0 100644
--- a/backend/cpp/llama-cpp/Makefile
+++ b/backend/cpp/llama-cpp/Makefile
@@ -1,5 +1,5 @@
-LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4
+LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
CMAKE_ARGS?=
diff --git a/backend/cpp/turboquant/Makefile b/backend/cpp/turboquant/Makefile
index e3482d8db..7a8024ebf 100644
--- a/backend/cpp/turboquant/Makefile
+++ b/backend/cpp/turboquant/Makefile
@@ -1,7 +1,7 @@
# Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant.
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
-TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64
+TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d
LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant
CMAKE_ARGS?=
diff --git a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch b/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch
deleted file mode 100644
index 7dfb385c3..000000000
--- a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch
+++ /dev/null
@@ -1,52 +0,0 @@
-diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
-index 680fd12..ffd6604 100644
---- a/ggml/src/ggml-cuda/fattn-vec.cuh
-+++ b/ggml/src/ggml-cuda/fattn-vec.cuh
-@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0);
- extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16);
- extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0);
- extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16);
--extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
--extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
--extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
-diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
-index 5c614a9..d765cfc 100644
---- a/ggml/src/ggml-cuda/fattn.cu
-+++ b/ggml/src/ggml-cuda/fattn.cu
-@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16)
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16)
-- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0)
-- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0)
-- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0)
-
- #ifdef GGML_CUDA_FA_ALL_QUANTS
- FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16)
-diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
-index a93be56..3630d87 100644
---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
-+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
-@@ -5,4 +5,3 @@
- DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
- DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
- DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
--DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
-diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
-index 3c806c2..c8a4d9f 100644
---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
-+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
-@@ -5,4 +5,3 @@
- DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
- DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
- DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
--DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
-diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
-index 180902f..1646ef0 100644
---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
-+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
-@@ -5,4 +5,3 @@
- DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
- DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
- DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
--DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile
index 65a4a784d..f2155ffb9 100644
--- a/backend/go/crispasr/Makefile
+++ b/backend/go/crispasr/Makefile
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# CrispASR version (release tag)
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
-CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee
+CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
SO_TARGET?=libgocrispasr.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile
index 8fc14bcb8..e288f6fcc 100644
--- a/backend/go/parakeet-cpp/Makefile
+++ b/backend/go/parakeet-cpp/Makefile
@@ -1,6 +1,6 @@
# parakeet-cpp backend Makefile.
#
-# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
+# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
# (.github/bump_deps.sh) can find and update it - matches the
# whisper.cpp / ds4 / vibevoice-cpp convention.
#
@@ -15,7 +15,7 @@
# That's what the L0 smoke test uses. The default target below does the
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
-PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
+PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
GOCMD?=go
diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile
index d19d103b1..56a32dbaf 100644
--- a/backend/go/vllm-cpp/Makefile
+++ b/backend/go/vllm-cpp/Makefile
@@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e
# vllm.cpp version
VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp
-VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb
+VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c
# MLX GEMM provider (darwin/metal only; see the metal branch below for why).
# Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun
diff --git a/backend/python/coqui/requirements.txt b/backend/python/coqui/requirements.txt
index 305f95f12..35c72a10f 100644
--- a/backend/python/coqui/requirements.txt
+++ b/backend/python/coqui/requirements.txt
@@ -1,4 +1,4 @@
-grpcio==1.83.1
+grpcio==1.84.0
protobuf
certifi
packaging==26.3
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt
index 3e3206912..8c5a025e2 100644
--- a/backend/python/transformers/requirements-cpu.txt
+++ b/backend/python/transformers/requirements-cpu.txt
@@ -2,9 +2,9 @@ torch==2.7.1
llvmlite==0.49.0
numba==0.67.0
accelerate
-transformers>=5.15.1
+transformers>=5.17.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt
index 40bf331d4..388a2334c 100644
--- a/backend/python/transformers/requirements-cublas12.txt
+++ b/backend/python/transformers/requirements-cublas12.txt
@@ -2,9 +2,9 @@ torch==2.7.1
accelerate
llvmlite==0.49.0
numba==0.67.0
-transformers>=5.15.1
+transformers>=5.17.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt
index f394f98b1..aa49676b1 100644
--- a/backend/python/transformers/requirements-cublas13.txt
+++ b/backend/python/transformers/requirements-cublas13.txt
@@ -2,9 +2,9 @@
torch==2.9.0
llvmlite==0.49.0
numba==0.67.0
-transformers>=5.15.1
+transformers>=5.17.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt
index e4b1bba11..84b62042b 100644
--- a/backend/python/transformers/requirements-hipblas.txt
+++ b/backend/python/transformers/requirements-hipblas.txt
@@ -1,11 +1,11 @@
--extra-index-url https://download.pytorch.org/whl/rocm7.0
torch==2.10.0+rocm7.0
accelerate
-transformers>=5.15.1
+transformers>=5.17.0
llvmlite==0.49.0
numba==0.67.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt
index 54ee6ce67..b0b565bf1 100644
--- a/backend/python/transformers/requirements-intel.txt
+++ b/backend/python/transformers/requirements-intel.txt
@@ -3,9 +3,9 @@ torch
optimum[openvino]
llvmlite==0.49.0
numba==0.67.0
-transformers>=5.15.1
+transformers>=5.17.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
\ No newline at end of file
diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt
index ea8ba5ab0..9659c943f 100644
--- a/backend/python/transformers/requirements-mps.txt
+++ b/backend/python/transformers/requirements-mps.txt
@@ -2,9 +2,9 @@ torch==2.7.1
llvmlite==0.49.0
numba==0.67.0
accelerate
-transformers>=5.15.1
+transformers>=5.17.0
bitsandbytes
-sentence-transformers==5.7.0
+sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt
index d85aca02a..af5027f10 100644
--- a/backend/python/transformers/requirements.txt
+++ b/backend/python/transformers/requirements.txt
@@ -1,6 +1,6 @@
-grpcio==1.83.0
+grpcio==1.84.0
protobuf==7.36.1
certifi
setuptools
scipy==1.18.0
-numpy>=2.5.2
\ No newline at end of file
+numpy>=2.5.3
\ No newline at end of file
diff --git a/backend/rust/kokoros/src/service.rs b/backend/rust/kokoros/src/service.rs
index 66f17e65b..a97cc4bf5 100644
--- a/backend/rust/kokoros/src/service.rs
+++ b/backend/rust/kokoros/src/service.rs
@@ -132,6 +132,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: true,
message: "Kokoros TTS model loaded".into(),
+ ..Default::default()
}))
}
@@ -180,11 +181,13 @@ impl Backend for KokorosService {
return Ok(Response::new(backend::Result {
success: false,
message: format!("Failed to write WAV: {}", e),
+ ..Default::default()
}));
}
Ok(Response::new(backend::Result {
success: true,
message: String::new(),
+ ..Default::default()
}))
}
Err(e) => {
@@ -192,6 +195,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: false,
message: format!("TTS error: {}", e),
+ ..Default::default()
}))
}
}
@@ -292,6 +296,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: true,
message: "Model freed".into(),
+ ..Default::default()
}))
}
diff --git a/core/backend/global_admission.go b/core/backend/global_admission.go
index 9b216ea83..7ee69aeb1 100644
--- a/core/backend/global_admission.go
+++ b/core/backend/global_admission.go
@@ -11,8 +11,9 @@ import (
)
// BackendAdmissionError reports that the process-wide backend execution
-// ceiling is full. HTTP callers map it to 503; internal callers receive the
-// same typed error instead of silently queueing and growing in-flight state.
+// ceiling is full. HTTP callers map it to 429 (Too Many Requests) with a
+// Retry-After header; internal callers receive the same typed error instead
+// of silently queueing and growing in-flight state.
type BackendAdmissionError struct {
Limit int
RetryAfter time.Duration
diff --git a/core/config/gallery.go b/core/config/gallery.go
index e22cbc94f..3f6b31ab9 100644
--- a/core/config/gallery.go
+++ b/core/config/gallery.go
@@ -47,10 +47,13 @@ type Gallery struct {
// fallback for availability, not a load-balancing pool: the primary is
// always preferred, and a mirror is only consulted after the one before
// it fails. Any URI the gallery loader understands works here
- // (https://, github:, file://).
+ // (https://, github:, file://, oci://).
Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"`
Name string `json:"name" yaml:"name"`
Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"`
+ // ArtifactVerification overrides Verification only for the gallery OCI artifact.
+ // Backend images keep their separate Verification policy.
+ ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"`
}
// Equal reports whether two gallery entries describe the same gallery.
@@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool {
if !slices.Equal(g.Mirrors, other.Mirrors) {
return false
}
+ if g.ArtifactVerification == nil || other.ArtifactVerification == nil {
+ if g.ArtifactVerification != other.ArtifactVerification {
+ return false
+ }
+ } else if *g.ArtifactVerification != *other.ArtifactVerification {
+ return false
+ }
if g.Verification == nil || other.Verification == nil {
return g.Verification == other.Verification
}
diff --git a/core/config/gallery_test.go b/core/config/gallery_test.go
index 71f4f3a36..7e1848a36 100644
--- a/core/config/gallery_test.go
+++ b/core/config/gallery_test.go
@@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() {
Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery"))
})
})
+
+var _ = Describe("Gallery artifact verification", func() {
+ It("compares artifact policies by value and preserves them in JSON and YAML", func() {
+ a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
+ b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
+ Expect(a.Equal(b)).To(BeTrue())
+ b.ArtifactVerification.Identity = "another-workflow"
+ Expect(a.Equal(b)).To(BeFalse())
+ b.ArtifactVerification = nil
+ Expect(a.Equal(b)).To(BeFalse())
+ raw, err := json.Marshal(a)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(json.Unmarshal(raw, &b)).To(Succeed())
+ Expect(a.Equal(b)).To(BeTrue())
+ raw, err = yaml.Marshal(a)
+ Expect(err).ToNot(HaveOccurred())
+ b = config.Gallery{}
+ Expect(yaml.Unmarshal(raw, &b)).To(Succeed())
+ Expect(a.Equal(b)).To(BeTrue())
+ })
+})
diff --git a/core/config/runtime_settings_startup.go b/core/config/runtime_settings_startup.go
index 9808c877b..e5d7e2a46 100644
--- a/core/config/runtime_settings_startup.go
+++ b/core/config/runtime_settings_startup.go
@@ -17,8 +17,8 @@ import (
// a caching mirror of the files below. The GitHub URI stays as a mirror so an
// install still resolves its gallery unchanged whenever the primary is
// unreachable - see the fallback chain in core/gallery/gallery_mirrors.go.
-const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]`
-const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]`
+const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
+const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
func mustGalleries(jsonList string) []Gallery {
var g []Gallery
diff --git a/core/config/runtime_settings_startup_test.go b/core/config/runtime_settings_startup_test.go
index 410e457d6..d49c53ac4 100644
--- a/core/config/runtime_settings_startup_test.go
+++ b/core/config/runtime_settings_startup_test.go
@@ -10,22 +10,22 @@ import (
)
var _ = Describe("default galleries", func() {
- It("serves the model gallery from index.localai.io with GitHub as a mirror", func() {
+ It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() {
var galleries []config.Gallery
Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed())
Expect(galleries).To(HaveLen(1))
Expect(galleries[0].Name).To(Equal("localai"))
Expect(galleries[0].URL).To(Equal("https://index.localai.io/models"))
- Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"}))
+ Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"}))
})
- It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() {
+ It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() {
var galleries []config.Gallery
Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed())
Expect(galleries).To(HaveLen(1))
Expect(galleries[0].Name).To(Equal("localai"))
Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends"))
- Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"}))
+ Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"}))
})
// The mirror is the whole reason this default is safe to ship: if
@@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() {
Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed())
for _, g := range galleries {
Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name)
+ Expect(g.ArtifactVerification).ToNot(BeNil())
+ Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"))
+ Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com"))
+ Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust")
}
}
})
diff --git a/core/gallery/entry_url.go b/core/gallery/entry_url.go
index 184ce2dd9..49f4f720d 100644
--- a/core/gallery/entry_url.go
+++ b/core/gallery/entry_url.go
@@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string {
if !looksLikeOCIGallery(candidate) {
continue
}
- dir := ociGalleryCacheDir(basePath, candidate, g.Verification)
+ dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g))
if dir == "" {
continue
}
diff --git a/core/gallery/gallery.go b/core/gallery/gallery.go
index 68d6de9e8..d0c0f43e0 100644
--- a/core/gallery/gallery.go
+++ b/core/gallery/gallery.go
@@ -646,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]()
// would also point relative entry urls at an unpacked tree the new policy has
// not produced yet, so they could not be installed.
func galleryIndexCacheKey(g config.Gallery) string {
- return g.Name + "-" + galleryCacheName(g.URL, g.Verification)
+ return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g))
}
func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) {
diff --git a/core/gallery/gallery_mirrors.go b/core/gallery/gallery_mirrors.go
index 4058c0a45..083512417 100644
--- a/core/gallery/gallery_mirrors.go
+++ b/core/gallery/gallery_mirrors.go
@@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
if !looksLikeOCIGallery(g.URL) {
return nil
}
- return g.Verification
+ return galleryArtifactPolicy(g)
}
// verifiableCandidates drops the candidates that cannot answer for a signed
@@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
// at, and after a refusal it would turn "this artifact is not trusted" into
// "use this other, unchecked copy instead".
func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string {
- if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) {
+ if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) {
return candidates
}
out := make([]string, 0, len(candidates))
diff --git a/core/gallery/gallery_oci.go b/core/gallery/gallery_oci.go
index 19d6d16c0..cbd649229 100644
--- a/core/gallery/gallery_oci.go
+++ b/core/gallery/gallery_oci.go
@@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) {
// later fetch served would hand the user a truncated gallery with no sign that
// anything went wrong.
func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) {
+ policy := galleryArtifactPolicy(g)
// Checked before the cache: a copy unpacked while strict integrity was
// off was never verified, and turning strict integrity on must not keep
// serving it for the rest of its TTL.
- if g.Verification == nil && requireIntegrity {
+ if policy == nil && requireIntegrity {
return nil, &galleryVerificationError{
strict: true,
- err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
+ err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
}
}
- cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification)
+ cacheDir := ociGalleryCacheDir(basePath, candidate, policy)
if cacheDir == "" {
return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate)
}
@@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
pullRef := downloader.URI(candidate).OCIReference()
- if g.Verification != nil {
+ if policy != nil {
// Resolve first, verify the digest, then pull that same digest.
// Nothing has been fetched at this point beyond the manifest, so a
// policy failure leaves no content anywhere.
@@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
if err != nil {
return nil, err
}
- if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil {
+ if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil {
// Only a decision about the artifact is a refusal. The
// verifier also reaches the Sigstore TUF mirror and the
// registry, and a timeout or a 5xx there says nothing about
@@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
return body, nil
}
+
+func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification {
+ if g.ArtifactVerification != nil {
+ return g.ArtifactVerification
+ }
+ return g.Verification
+}
diff --git a/core/gallery/gallery_oci_test.go b/core/gallery/gallery_oci_test.go
index 5672da123..4c038b5a2 100644
--- a/core/gallery/gallery_oci_test.go
+++ b/core/gallery/gallery_oci_test.go
@@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() {
})
})
+ It("uses the artifact policy without replacing backend image verification", func() {
+ srv, _, _ := ociRegistry()
+ url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}})
+ backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"}
+ artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"}
+ var seen *config.GalleryVerification
+ stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil })
+ g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy}
+ _, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true)
+ Expect(source).To(Equal(url))
+ Expect(err).ToNot(HaveOccurred())
+ Expect(seen).To(Equal(artifactPolicy))
+ Expect(g.Verification).To(Equal(backendPolicy))
+ })
+
It("refuses an unsigned gallery in strict integrity mode", func() {
srv, _, blobs := ociRegistry()
url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{
diff --git a/core/http/admission_handler_test.go b/core/http/admission_handler_test.go
new file mode 100644
index 000000000..6a0a2201d
--- /dev/null
+++ b/core/http/admission_handler_test.go
@@ -0,0 +1,70 @@
+package http
+
+import (
+ "errors"
+ "fmt"
+ "net/http"
+ "net/http/httptest"
+ "time"
+
+ "github.com/labstack/echo/v4"
+ corebackend "github.com/mudler/LocalAI/core/backend"
+ "github.com/mudler/LocalAI/core/services/nodes"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("Backend admission", func() {
+ It("maps BackendAdmissionError to 429 with Retry-After", func() {
+ e := echo.New()
+ req := httptest.NewRequest(http.MethodPost, "/", nil)
+ rec := httptest.NewRecorder()
+ c := e.NewContext(req, rec)
+
+ err := &corebackend.BackendAdmissionError{Limit: 4, RetryAfter: 3 * time.Second}
+ code := applyBackendAdmission(err, http.StatusInternalServerError, c)
+
+ Expect(code).To(Equal(http.StatusTooManyRequests))
+ Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
+ })
+
+ It("passes through non-admission errors unchanged", func() {
+ e := echo.New()
+ req := httptest.NewRequest(http.MethodPost, "/", nil)
+ rec := httptest.NewRecorder()
+ c := e.NewContext(req, rec)
+
+ code := applyBackendAdmission(errors.New("some other error"), http.StatusInternalServerError, c)
+ Expect(code).To(Equal(http.StatusInternalServerError))
+ Expect(rec.Header().Get("Retry-After")).To(BeEmpty())
+ })
+})
+
+var _ = Describe("No available nodes", func() {
+ It("maps ErrNoAvailableNodes to 503", func() {
+ // The scheduler wraps the sentinel in fmt.Errorf chains and via
+ // errors.Join — errors.Is must still find it.
+ wrapped := fmt.Errorf("routing model foo: %w",
+ fmt.Errorf("no available nodes: %w",
+ fmt.Errorf("no healthy nodes available: %w",
+ errors.Join(nodes.ErrEvictionBusy, nodes.ErrNoAvailableNodes))))
+
+ code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
+ Expect(code).To(Equal(http.StatusServiceUnavailable))
+ })
+
+ It("maps selector-mismatch chain to 503", func() {
+ wrapped := fmt.Errorf("routing model bar: %w",
+ fmt.Errorf("no available nodes: %w",
+ fmt.Errorf("no healthy nodes match selector for model bar: {\"gpu.vendor\":\"tpu\"}: %w",
+ nodes.ErrNoAvailableNodes)))
+
+ code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
+ Expect(code).To(Equal(http.StatusServiceUnavailable))
+ })
+
+ It("passes through unrelated errors unchanged", func() {
+ code := applyNoAvailableNodes(errors.New("database timeout"), http.StatusInternalServerError)
+ Expect(code).To(Equal(http.StatusInternalServerError))
+ })
+})
diff --git a/core/http/app.go b/core/http/app.go
index c7bc083c3..b0cd299eb 100644
--- a/core/http/app.go
+++ b/core/http/app.go
@@ -88,7 +88,20 @@ func applyBackendAdmission(err error, code int, c echo.Context) int {
return code
}
c.Response().Header().Set("Retry-After", strconv.Itoa(int(capacityErr.RetryAfter.Seconds())))
- return http.StatusServiceUnavailable
+ return http.StatusTooManyRequests
+}
+
+// applyNoAvailableNodes maps scheduler "no available nodes" errors to 503.
+// When the cluster has no healthy node to serve a model — all are full, a
+// node selector excludes every candidate, or eviction could not free a slot —
+// the request is retryable, not a server bug. Without this the error fell
+// through to 500, which tells clients something is broken when they just
+// need to wait for a node.
+func applyNoAvailableNodes(err error, code int) int {
+ if errors.Is(err, nodes.ErrNoAvailableNodes) {
+ return http.StatusServiceUnavailable
+ }
+ return code
}
// respondModelLoading answers a request whose model is still cold-loading with
@@ -211,6 +224,7 @@ func API(application *application.Application) (*echo.Echo, error) {
}
code = applyModelLoadCooldown(err, code, c)
code = applyBackendAdmission(err, code, c)
+ code = applyNoAvailableNodes(err, code)
// Handle 404 errors: serve React SPA for HTML requests, JSON otherwise
if code == http.StatusNotFound {
@@ -227,8 +241,13 @@ func API(application *application.Application) (*echo.Echo, error) {
}
// Send custom error page
+ errType := ""
+ var capErr *corebackend.BackendAdmissionError
+ if errors.As(err, &capErr) {
+ errType = "rate_limit_error"
+ }
c.JSON(code, schema.ErrorResponse{
- Error: &schema.APIError{Message: err.Error(), Code: code},
+ Error: &schema.APIError{Message: err.Error(), Code: code, Type: errType},
})
}
} else {
@@ -243,6 +262,7 @@ func API(application *application.Application) (*echo.Echo, error) {
// Opaque errors deliberately withhold the body, so a still-loading
// model gets the status and Retry-After but no progress detail.
code = applyModelLoading(err, code, c)
+ code = applyNoAvailableNodes(err, code)
c.NoContent(code)
}
}
diff --git a/core/http/endpoints/localai/system.go b/core/http/endpoints/localai/system.go
index 996c9a781..9c4ba7503 100644
--- a/core/http/endpoints/localai/system.go
+++ b/core/http/endpoints/localai/system.go
@@ -8,6 +8,7 @@ import (
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/monitoring"
"github.com/mudler/LocalAI/pkg/model"
+ "github.com/mudler/LocalAI/pkg/xsysinfo"
)
// SystemInformations returns the system informations
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
entry.Process = proc
}
}
+ if pid, ok := localPID(m); ok {
+ if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
+ entry.SizeVRAM = &used
+ }
+ }
sysmodels = append(sysmodels, entry)
}
if sampler != nil {
diff --git a/core/http/endpoints/localai/system_info_test.go b/core/http/endpoints/localai/system_info_test.go
new file mode 100644
index 000000000..83f7daa07
--- /dev/null
+++ b/core/http/endpoints/localai/system_info_test.go
@@ -0,0 +1,49 @@
+// SPDX-License-Identifier: MIT
+package localai_test
+
+import (
+ "encoding/json"
+ "net/http"
+ "net/http/httptest"
+ "os"
+ "path/filepath"
+
+ "github.com/labstack/echo/v4"
+ "github.com/mudler/LocalAI/core/config"
+ "github.com/mudler/LocalAI/core/http/endpoints/localai"
+ "github.com/mudler/LocalAI/pkg/model"
+ "github.com/mudler/LocalAI/pkg/system"
+ process "github.com/mudler/go-processmanager"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("SystemInformations memory", func() {
+ It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
+ path, err := os.MkdirTemp("", "system-info-")
+ Expect(err).NotTo(HaveOccurred())
+ DeferCleanup(os.RemoveAll, path)
+ configFile := filepath.Join(path, "remote.yaml")
+ Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
+ cl := config.NewModelConfigLoader(path)
+ Expect(cl.ReadModelConfig(configFile)).To(Succeed())
+ ml := model.NewModelLoader(&system.SystemState{})
+ store := model.NewInMemoryModelStore()
+ store.Set("remote", model.NewModel("remote", "worker:50051", nil))
+ store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
+ ml.SetModelStore(store)
+ app := echo.New()
+ app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
+ rec := httptest.NewRecorder()
+ app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
+ Expect(rec.Code).To(Equal(http.StatusOK))
+ var response struct {
+ Models []map[string]any `json:"loaded_models"`
+ }
+ Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
+ Expect(response.Models).To(ConsistOf(
+ map[string]any{"id": "remote", "backend": "llama-cpp"},
+ map[string]any{"id": "stopped"},
+ ))
+ })
+})
diff --git a/core/http/endpoints/ollama/models.go b/core/http/endpoints/ollama/models.go
index 60e58b9ea..34c8b0d6e 100644
--- a/core/http/endpoints/ollama/models.go
+++ b/core/http/endpoints/ollama/models.go
@@ -3,6 +3,8 @@ package ollama
import (
"crypto/sha256"
"fmt"
+ "os"
+ "path/filepath"
"strings"
"time"
@@ -37,7 +39,7 @@ func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) ec
Name: ollamaName,
Model: ollamaName,
ModifiedAt: time.Now().UTC(),
- Size: 0,
+ Size: modelOnDiskSize(bcl, ml, name),
Digest: digest,
Details: details,
Capabilities: caps,
@@ -101,13 +103,15 @@ func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) e
details, caps := modelMetaFromConfig(bcl, name)
entry := schema.OllamaPsEntry{
- Name: ollamaName,
- Model: ollamaName,
- Size: 0,
- Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
- Details: details,
- ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
- SizeVRAM: 0,
+ Name: ollamaName,
+ Model: ollamaName,
+ Size: modelOnDiskSize(bcl, ml, name),
+ Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
+ Details: details,
+ ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
+ // SizeVRAM is left unset: LocalAI has no authoritative per-model
+ // VRAM figure to report, and a literal 0 is worse than omitting
+ // the field (clients treat 0 as "costs nothing").
Capabilities: caps,
}
models = append(models, entry)
@@ -143,6 +147,37 @@ func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.Oll
return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg)
}
+// modelOnDiskSize returns the on-disk byte size of a model's primary weight
+// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath.
+// Returns nil when the size is unknown so callers omit the JSON field instead
+// of emitting an authoritative 0 (issue #11969).
+func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 {
+ if ml == nil || ml.ModelPath == "" {
+ return nil
+ }
+
+ // List endpoints pass the stored model ID, including any configured tag.
+ configName := name
+ rel := configName
+ if bcl != nil {
+ if cfg, exists := bcl.GetModelConfig(configName); exists {
+ if fileName := cfg.ModelFileName(); fileName != "" {
+ rel = fileName
+ }
+ }
+ }
+ if rel == "" {
+ return nil
+ }
+
+ info, err := os.Stat(filepath.Join(ml.ModelPath, rel))
+ if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 {
+ return nil
+ }
+ size := info.Size()
+ return &size
+}
+
func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails {
family := cfg.Backend
details := schema.OllamaModelDetails{
diff --git a/core/http/endpoints/ollama/models_test.go b/core/http/endpoints/ollama/models_test.go
index c4d0d6b5e..bf3c64b90 100644
--- a/core/http/endpoints/ollama/models_test.go
+++ b/core/http/endpoints/ollama/models_test.go
@@ -13,6 +13,8 @@ import (
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/http/endpoints/ollama"
"github.com/mudler/LocalAI/core/schema"
+ "github.com/mudler/LocalAI/pkg/model"
+ "github.com/mudler/LocalAI/pkg/system"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
@@ -163,8 +165,165 @@ parameters:
})
Describe("ListModelsEndpoint", func() {
- It("includes capabilities and details for each listed model in /api/tags", func() {
- Skip("covered by per-entry tests; integration smoke test")
+ var (
+ tmpDir string
+ bcl *config.ModelConfigLoader
+ ml *model.ModelLoader
+ )
+
+ BeforeEach(func() {
+ var err error
+ tmpDir, err = os.MkdirTemp("", "ollama-tags-test-*")
+ Expect(err).ToNot(HaveOccurred())
+
+ systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
+ Expect(err).ToNot(HaveOccurred())
+ ml = model.NewModelLoader(systemState)
+ bcl = config.NewModelConfigLoader(tmpDir)
+ })
+
+ AfterEach(func() {
+ _ = os.RemoveAll(tmpDir)
+ })
+
+ writeConfig := func(name, yaml string) {
+ path := filepath.Join(tmpDir, name+".yaml")
+ Expect(os.WriteFile(path, []byte(yaml), 0o644)).To(Succeed())
+ Expect(bcl.ReadModelConfig(path)).To(Succeed())
+ }
+
+ callTags := func() (schema.OllamaListResponse, []byte) {
+ req := httptest.NewRequest(http.MethodGet, "/api/tags", nil)
+ rec := httptest.NewRecorder()
+ c := e.NewContext(req, rec)
+
+ handler := ollama.ListModelsEndpoint(bcl, ml)
+ Expect(handler(c)).To(Succeed())
+ Expect(rec.Code).To(Equal(http.StatusOK))
+
+ var resp schema.OllamaListResponse
+ Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
+ return resp, rec.Body.Bytes()
+ }
+
+ It("uses the exact configured name when a model has a tag", func() {
+ Expect(os.WriteFile(filepath.Join(tmpDir, "base.gguf"), []byte("base"), 0o644)).To(Succeed())
+ Expect(os.WriteFile(filepath.Join(tmpDir, "tagged.gguf"), []byte("tagged-weights"), 0o644)).To(Succeed())
+ writeConfig("chat", "name: chat\nparameters:\n model: base.gguf\n")
+ writeConfig("tagged", "name: chat:q8\nparameters:\n model: tagged.gguf\n")
+
+ resp, _ := callTags()
+ var tagged *int64
+ for _, entry := range resp.Models {
+ if entry.Name == "chat:q8" {
+ tagged = entry.Size
+ }
+ }
+ Expect(tagged).ToNot(BeNil())
+ Expect(*tagged).To(Equal(int64(len("tagged-weights"))))
+ })
+
+ It("reports on-disk size from ModelFileName+ModelPath and omits size when unknown", func() {
+ weight := []byte("fake-gguf-weights-0123456789")
+ Expect(os.WriteFile(filepath.Join(tmpDir, "Llama-3-8B-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
+ writeConfig("chat", `
+name: chat
+backend: llama-cpp
+template:
+ chat: "{{ .Input }}"
+parameters:
+ model: Llama-3-8B-Q4_K_M.gguf
+`)
+ writeConfig("missing-weights", `
+name: missing-weights
+backend: llama-cpp
+template:
+ chat: "{{ .Input }}"
+parameters:
+ model: does-not-exist.gguf
+`)
+
+ resp, raw := callTags()
+ Expect(resp.Models).To(HaveLen(2))
+
+ byName := map[string]schema.OllamaModelEntry{}
+ for _, m := range resp.Models {
+ byName[m.Name] = m
+ }
+
+ chat := byName["chat:latest"]
+ Expect(chat.Size).ToNot(BeNil())
+ Expect(*chat.Size).To(Equal(int64(len(weight))))
+ Expect(chat.Capabilities).To(ContainElement("completion"))
+ Expect(chat.Details.QuantizationLevel).To(Equal("Q4_K_M"))
+
+ missing := byName["missing-weights:latest"]
+ Expect(missing.Size).To(BeNil())
+ Expect(string(raw)).ToNot(ContainSubstring(`"size":0`))
+ })
+ })
+
+ Describe("ListRunningEndpoint", func() {
+ var (
+ tmpDir string
+ bcl *config.ModelConfigLoader
+ ml *model.ModelLoader
+ )
+
+ BeforeEach(func() {
+ var err error
+ tmpDir, err = os.MkdirTemp("", "ollama-ps-test-*")
+ Expect(err).ToNot(HaveOccurred())
+
+ systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
+ Expect(err).ToNot(HaveOccurred())
+ ml = model.NewModelLoader(systemState)
+ bcl = config.NewModelConfigLoader(tmpDir)
+ })
+
+ AfterEach(func() {
+ _ = os.RemoveAll(tmpDir)
+ })
+
+ It("reports on-disk size for loaded models and omits size_vram when unknown", func() {
+ weight := []byte("loaded-model-weights-abcdef")
+ Expect(os.WriteFile(filepath.Join(tmpDir, "granite-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
+
+ cfgPath := filepath.Join(tmpDir, "granite.yaml")
+ Expect(os.WriteFile(cfgPath, []byte(`
+name: granite
+backend: llama-cpp
+template:
+ chat: "{{ .Input }}"
+parameters:
+ model: granite-Q4_K_M.gguf
+`), 0o644)).To(Succeed())
+ Expect(bcl.ReadModelConfig(cfgPath)).To(Succeed())
+
+ store := model.NewInMemoryModelStore()
+ store.Set("granite", model.NewModel("granite", "addr", nil))
+ ml.SetModelStore(store)
+
+ req := httptest.NewRequest(http.MethodGet, "/api/ps", nil)
+ rec := httptest.NewRecorder()
+ c := e.NewContext(req, rec)
+
+ handler := ollama.ListRunningEndpoint(bcl, ml)
+ Expect(handler(c)).To(Succeed())
+ Expect(rec.Code).To(Equal(http.StatusOK))
+
+ raw := rec.Body.String()
+ Expect(raw).ToNot(ContainSubstring(`"size":0`))
+ Expect(raw).ToNot(ContainSubstring(`"size_vram"`))
+
+ var resp schema.OllamaPsResponse
+ Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
+ Expect(resp.Models).To(HaveLen(1))
+ Expect(resp.Models[0].Name).To(Equal("granite:latest"))
+ Expect(resp.Models[0].Size).ToNot(BeNil())
+ Expect(*resp.Models[0].Size).To(Equal(int64(len(weight))))
+ Expect(resp.Models[0].SizeVRAM).To(BeNil())
+ Expect(resp.Models[0].Details.QuantizationLevel).To(Equal("Q4_K_M"))
})
})
})
diff --git a/core/http/endpoints/openai/chat.go b/core/http/endpoints/openai/chat.go
index f5f0d0109..638be16a9 100644
--- a/core/http/endpoints/openai/chat.go
+++ b/core/http/endpoints/openai/chat.go
@@ -451,7 +451,7 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator
}
// Update input grammar or json_schema based on use_llama_grammar option
- jsStruct := funcs.ToJSONStructure(config.FunctionsConfig.FunctionNameKey, config.FunctionsConfig.FunctionNameKey)
+ jsStruct := config.FunctionsConfig.ToJSONStructure(funcs)
g, err := jsStruct.Grammar(config.FunctionsConfig.GrammarOptions()...)
if err == nil {
config.Grammar = g
diff --git a/core/http/endpoints/openai/list_capabilities.go b/core/http/endpoints/openai/list_capabilities.go
index 72a645ef6..27583385a 100644
--- a/core/http/endpoints/openai/list_capabilities.go
+++ b/core/http/endpoints/openai/list_capabilities.go
@@ -36,6 +36,12 @@ func ListModelCapabilitiesEndpoint(bcl *config.ModelConfigLoader, ml *model.Mode
for _, m := range modelNames {
entry := schema.ModelCapabilities{ID: m, Object: "model"}
if cfg, ok := modelConfigFor(bcl, m); ok {
+ // Mirror the request path: SetDefaults applies the application
+ // default only when the model leaves context_size unset. An
+ // explicit 0 or -1 falls through to the backend fallback there.
+ if cfg.ContextSize == nil && appConfig != nil && appConfig.ContextSize > 0 {
+ cfg.ContextSize = &appConfig.ContextSize
+ }
entry.Capabilities = cfg.Capabilities()
entry.ThreeDOperations = cfg.ThreeDOperations()
entry.InputModalities = cfg.InputModalities()
diff --git a/core/http/endpoints/openai/list_capabilities_test.go b/core/http/endpoints/openai/list_capabilities_test.go
index 0c8a5d4ff..ecfc85f81 100644
--- a/core/http/endpoints/openai/list_capabilities_test.go
+++ b/core/http/endpoints/openai/list_capabilities_test.go
@@ -160,6 +160,33 @@ parameters:
Expect(entry).NotTo(BeNil())
Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
})
+
+ It("uses application config context size when model context_size is unset", func() {
+ writeConfig("llm-app-default", `
+name: llm-app-default
+backend: llama-cpp
+parameters:
+ model: model.gguf
+`)
+ appConf.ContextSize = 8192
+ entry := entryFor(call(), "llm-app-default")
+ Expect(entry).NotTo(BeNil())
+ Expect(entry.ContextSize).To(Equal(8192))
+ })
+
+ It("keeps the backend fallback when the model sets a non-positive context_size", func() {
+ writeConfig("llm-explicit-zero", `
+name: llm-explicit-zero
+backend: llama-cpp
+context_size: 0
+parameters:
+ model: model.gguf
+`)
+ appConf.ContextSize = 8192
+ entry := entryFor(call(), "llm-explicit-zero")
+ Expect(entry).NotTo(BeNil())
+ Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
+ })
It("reports an alias with its target's capabilities and context_size", func() {
writeConfig("real-llm", `
name: real-llm
diff --git a/core/http/endpoints/openai/realtime_model.go b/core/http/endpoints/openai/realtime_model.go
index b525eee26..a02696a1f 100644
--- a/core/http/endpoints/openai/realtime_model.go
+++ b/core/http/endpoints/openai/realtime_model.go
@@ -293,7 +293,7 @@ func (m *wrappedModel) Predict(ctx context.Context, messages schema.Messages, im
}
// Generate grammar from function definitions
- jsStruct := functions.Functions(funcs).ToJSONStructure(turnCfg.FunctionsConfig.FunctionNameKey, turnCfg.FunctionsConfig.FunctionNameKey)
+ jsStruct := turnCfg.FunctionsConfig.ToJSONStructure(functions.Functions(funcs))
g, err := jsStruct.Grammar(turnCfg.FunctionsConfig.GrammarOptions()...)
if err == nil {
turnCfg.Grammar = g
diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go
index d52e79b3f..71164e291 100644
--- a/core/http/endpoints/openresponses/responses.go
+++ b/core/http/endpoints/openresponses/responses.go
@@ -204,7 +204,7 @@ func ResponsesEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, eval
}
// Generate grammar to constrain model output to valid function calls
- jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey)
+ jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction)
g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...)
if err == nil {
cfg.Grammar = g
@@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
return true
}
- // Try JSON parsing as fallback
- jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true)
- if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount {
+ // Only completed JSON calls can be emitted as completed SSE items.
+ jsonResults := parseStreamingJSONToolCalls(cleanedResult)
+ if len(jsonResults) > lastEmittedToolCallCount {
for i := lastEmittedToolCallCount; i < len(jsonResults); i++ {
- jsonObj := jsonResults[i]
- if name, ok := jsonObj["name"].(string); ok && name != "" {
- args := "{}"
- if argsVal, ok := jsonObj["arguments"]; ok {
- if argsStr, ok := argsVal.(string); ok {
- args = argsStr
- } else {
- argsBytes, _ := json.Marshal(argsVal)
- args = string(argsBytes)
- }
- }
+ tc := jsonResults[i]
+ toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
+ outputIndex++
- toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
- outputIndex++
-
- functionCallItem := &schema.ORItemField{
- Type: "function_call",
- ID: toolCallID,
- Status: "completed",
- CallID: toolCallID,
- Name: name,
- Arguments: args,
- }
- sendSSEEvent(c, &schema.ORStreamEvent{
- Type: "response.output_item.added",
- SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
- Item: functionCallItem,
- })
- sequenceNumber++
-
- sendSSEEvent(c, &schema.ORStreamEvent{
- Type: "response.output_item.done",
- SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
- Item: functionCallItem,
- })
- sequenceNumber++
+ functionCallItem := &schema.ORItemField{
+ Type: "function_call",
+ ID: toolCallID,
+ Status: "completed",
+ CallID: toolCallID,
+ Name: tc.Name,
+ Arguments: tc.Arguments,
}
+ sendSSEEvent(c, &schema.ORStreamEvent{
+ Type: "response.output_item.added",
+ SequenceNumber: sequenceNumber,
+ OutputIndex: &outputIndex,
+ Item: functionCallItem,
+ })
+ sequenceNumber++
+
+ sendSSEEvent(c, &schema.ORStreamEvent{
+ Type: "response.output_item.done",
+ SequenceNumber: sequenceNumber,
+ OutputIndex: &outputIndex,
+ Item: functionCallItem,
+ })
+ sequenceNumber++
}
lastEmittedToolCallCount = len(jsonResults)
c.Response().Flush()
@@ -2424,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
}
// Non-tool-call streaming path
+ messageOutputIndex := outputIndex
+ var reasoningOutputIndex int
// Emit output_item.added for message
currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String())
messageItem := &schema.ORItemField{
@@ -2436,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
Item: messageItem,
})
sequenceNumber++
@@ -2448,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.added",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
ContentIndex: ¤tContentIndex,
Part: &emptyTextPart,
})
@@ -2471,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
}
// Handle reasoning item
- if extractor.Reasoning() != "" {
+ if extractor.Reasoning() != "" || reasoningDelta != "" {
// Check if we need to create reasoning item
if currentReasoningID == "" {
outputIndex++
+ reasoningOutputIndex = outputIndex
currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String())
reasoningItem := &schema.ORItemField{
Type: "reasoning",
@@ -2484,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
Item: reasoningItem,
})
sequenceNumber++
@@ -2496,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.added",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
ContentIndex: ¤tReasoningContentIndex,
Part: &emptyPart,
})
@@ -2509,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.delta",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
ContentIndex: ¤tReasoningContentIndex,
Delta: strPtr(reasoningDelta),
Logprobs: emptyLogprobs(),
@@ -2526,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.delta",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
ContentIndex: ¤tContentIndex,
Delta: strPtr(contentDelta),
Logprobs: emptyLogprobs(),
@@ -2595,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.done",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
ContentIndex: ¤tReasoningContentIndex,
Text: strPtr(finalReasoning),
Logprobs: emptyLogprobs(),
@@ -2608,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.done",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
ContentIndex: ¤tReasoningContentIndex,
Part: &reasoningPart,
})
@@ -2624,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
+ OutputIndex: &reasoningOutputIndex,
Item: reasoningItem,
})
sequenceNumber++
@@ -2658,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.done",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
ContentIndex: ¤tContentIndex,
Text: strPtr(result),
Logprobs: logprobsPtr(mcpStreamLogprobs),
@@ -2671,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.done",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
ContentIndex: ¤tContentIndex,
Part: &resultPart,
})
@@ -2683,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
- OutputIndex: &outputIndex,
+ OutputIndex: &messageOutputIndex,
Item: messageItem,
})
sequenceNumber++
@@ -2723,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
// Emit response.completed
now := time.Now().Unix()
- // Collect final output items (reasoning first, then messages, then tool calls)
- var finalOutputItems []schema.ORItemField
- // Add reasoning item if it exists
- if currentReasoningID != "" && finalReasoning != "" {
- finalOutputItems = append(finalOutputItems, schema.ORItemField{
- Type: "reasoning",
- ID: currentReasoningID,
- Status: "completed",
- Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)},
- })
- }
- // Add message item
- if len(collectedOutputItems) > 0 {
- // Use collected items (may include reasoning already)
- for _, item := range collectedOutputItems {
- if item.Type == "message" {
- finalOutputItems = append(finalOutputItems, item)
- }
- }
- } else {
- finalOutputItems = append(finalOutputItems, *messageItem)
- }
- // Add function_call items from fallback
- for _, item := range collectedOutputItems {
- if item.Type == "function_call" {
- finalOutputItems = append(finalOutputItems, item)
- }
- }
+ // The final output array must use the indices announced in the stream.
+ // The message is opened first, followed by reasoning and fallback calls.
+ finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...)
responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{
InputTokens: noToolTokenUsage.Prompt,
OutputTokens: noToolTokenUsage.Completion,
diff --git a/core/http/endpoints/openresponses/responses_stream_test.go b/core/http/endpoints/openresponses/responses_stream_test.go
new file mode 100644
index 000000000..13ddb10e1
--- /dev/null
+++ b/core/http/endpoints/openresponses/responses_stream_test.go
@@ -0,0 +1,148 @@
+// SPDX-License-Identifier: MIT
+package openresponses
+
+import (
+ "context"
+ "encoding/json"
+ "net/http/httptest"
+ "strings"
+
+ "github.com/labstack/echo/v4"
+ "github.com/mudler/LocalAI/core/backend"
+ "github.com/mudler/LocalAI/core/config"
+ "github.com/mudler/LocalAI/core/schema"
+ pb "github.com/mudler/LocalAI/pkg/grpc/proto"
+ "github.com/mudler/LocalAI/pkg/model"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("Responses stream item consistency", func() {
+ DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) {
+ originalInference := backend.ModelInferenceFunc
+ DeferCleanup(func() { backend.ModelInferenceFunc = originalInference })
+ backend.ModelInferenceFunc = func(
+ ctx context.Context, prompt string, messages schema.Messages,
+ images, videos, audios []string, loader *model.ModelLoader,
+ cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig,
+ tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string,
+ logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string,
+ ) (func() (backend.LLMResponse, error), error) {
+ return func() (backend.LLMResponse, error) {
+ for i, token := range tokens {
+ usage := backend.TokenUsage{}
+ if len(chatDeltas) > 0 {
+ usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]}
+ }
+ if !tokenCallback(token, usage) {
+ break
+ }
+ }
+ return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil
+ }, nil
+ }
+ cfg := &config.ModelConfig{}
+ cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback
+ cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)(.*?)`}
+ recorder := httptest.NewRecorder()
+ request := httptest.NewRequest("POST", "/v1/responses", nil)
+ c := echo.New().NewContext(request, recorder)
+ input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true}
+ err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n"))
+
+ var events []schema.ORStreamEvent
+ var completed *schema.ORResponseResource
+ for _, line := range strings.Split(recorder.Body.String(), "\n") {
+ if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" {
+ continue
+ }
+ var event schema.ORStreamEvent
+ Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed())
+ Expect(event.Type).NotTo(Equal("error"))
+ events = append(events, event)
+ if event.Type == "response.completed" {
+ completed = event.Response
+ }
+ }
+ Expect(completed).NotTo(BeNil())
+ wantCount := 1
+ if wantReasoning != "" {
+ wantCount++
+ }
+ if fallback {
+ wantCount++
+ }
+ Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls")
+
+ indices := map[string]int{}
+ done := map[string]int{}
+ deltas := map[string]string{}
+ for i, event := range events {
+ Expect(event.SequenceNumber).To(Equal(i))
+ if event.Type == "response.output_item.added" {
+ Expect(event.Item).NotTo(BeNil())
+ Expect(event.OutputIndex).NotTo(BeNil())
+ Expect(indices).NotTo(HaveKey(event.Item.ID))
+ Expect(*event.OutputIndex).To(Equal(len(indices)))
+ indices[event.Item.ID] = *event.OutputIndex
+ }
+ id := event.ItemID
+ if event.Item != nil {
+ id = event.Item.ID
+ }
+ if id == "" {
+ continue
+ }
+ Expect(indices).To(HaveKey(id))
+ Expect(event.OutputIndex).NotTo(BeNil())
+ Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id)
+ Expect(completed.Output[indices[id]].ID).To(Equal(id))
+ if event.Type == "response.output_item.done" {
+ done[id]++
+ Expect(event.Item.Status).To(Equal("completed"))
+ Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type))
+ if event.Item.Type == "function_call" {
+ Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name))
+ Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments))
+ } else {
+ Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content))
+ }
+ }
+ if event.Type == "response.output_text.delta" {
+ deltas[id] += *event.Delta
+ }
+ }
+ Expect(indices).To(HaveLen(wantCount))
+ for _, item := range completed.Output {
+ Expect(done[item.ID]).To(Equal(1))
+ switch item.Type {
+ case "message", "reasoning":
+ want := wantAnswer
+ if item.Type == "reasoning" {
+ want = wantReasoning
+ }
+ parts, ok := item.Content.([]any)
+ Expect(ok).To(BeTrue())
+ Expect(parts).To(HaveLen(1))
+ Expect(parts[0].(map[string]any)["text"]).To(Equal(want))
+ if !fallback {
+ Expect(deltas[item.ID]).To(Equal(want))
+ }
+ case "function_call":
+ Expect(item.Name).To(Equal("get_weather"))
+ Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`))
+ Expect(item.CallID).NotTo(BeEmpty())
+ default:
+ Fail("unexpected output item type: " + item.Type)
+ }
+ }
+ },
+ Entry("tagged reasoning and answer", []string{"", "Let me think.", "", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false),
+ Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false),
+ Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false),
+ Entry("automatic fallback tool call", []string{`{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "", "", true),
+ Entry("reasoning and automatic fallback tool call", []string{"", "Let me think.", "", `{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "Let me think.", "", true),
+ )
+})
diff --git a/core/http/endpoints/openresponses/stream_tool_calls.go b/core/http/endpoints/openresponses/stream_tool_calls.go
new file mode 100644
index 000000000..b8f0f185d
--- /dev/null
+++ b/core/http/endpoints/openresponses/stream_tool_calls.go
@@ -0,0 +1,35 @@
+package openresponses
+
+import (
+ "encoding/json"
+
+ "github.com/mudler/LocalAI/pkg/functions"
+)
+
+func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults {
+ // Partial parsing heals unfinished arguments. The caller emits terminal
+ // events and never revisits emitted calls, so only accept complete JSON.
+ // Keep completed objects returned before an unfinished trailing object.
+ objects, _ := functions.ParseJSONIterative(text, false)
+ var calls []functions.FuncCallResults
+ for _, object := range objects {
+ name, ok := object["name"].(string)
+ if !ok || name == "" {
+ continue
+ }
+ arguments := "{}"
+ if value, ok := object["arguments"]; ok {
+ if s, ok := value.(string); ok {
+ arguments = s
+ } else {
+ data, err := json.Marshal(value)
+ if err != nil {
+ continue
+ }
+ arguments = string(data)
+ }
+ }
+ calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments})
+ }
+ return calls
+}
diff --git a/core/http/endpoints/openresponses/stream_tool_calls_test.go b/core/http/endpoints/openresponses/stream_tool_calls_test.go
new file mode 100644
index 000000000..1fa6bca2a
--- /dev/null
+++ b/core/http/endpoints/openresponses/stream_tool_calls_test.go
@@ -0,0 +1,44 @@
+package openresponses
+
+import (
+ "github.com/mudler/LocalAI/pkg/functions"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("Streaming JSON tool calls", func() {
+ It("waits for the arguments before completing a split call", func() {
+ Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty())
+ Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty())
+ Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{
+ {Name: "Bash", Arguments: `{"command":"ls -la"}`},
+ }))
+ })
+
+ It("does not complete a call at any intermediate token boundary", func() {
+ text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}`
+ for end := 1; end < len(text); end++ {
+ Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end])
+ }
+ Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1))
+ })
+
+ It("keeps completed calls while the next call is incomplete", func() {
+ Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{
+ {Name: "Bash", Arguments: `{"command":"ls -la"}`},
+ }))
+ })
+
+ It("preserves string arguments and calls that take no arguments", func() {
+ Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{
+ {Name: "Bash", Arguments: `{"command":"ls -la"}`},
+ {Name: "status", Arguments: `{}`},
+ }))
+ })
+
+ It("does not count unrelated JSON objects as emitted calls", func() {
+ Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{
+ {Name: "status", Arguments: `{}`},
+ }))
+ })
+})
diff --git a/core/http/endpoints/openresponses/websocket.go b/core/http/endpoints/openresponses/websocket.go
index 3a92f275a..f0cb4b331 100644
--- a/core/http/endpoints/openresponses/websocket.go
+++ b/core/http/endpoints/openresponses/websocket.go
@@ -381,7 +381,7 @@ func handleWSResponseCreate(connCtx context.Context, conn *lockedConn, connectio
funcsWithNoAction = funcsWithNoAction.Select(cfg.FunctionToCall())
}
- jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey)
+ jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction)
g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...)
if err == nil {
cfg.Grammar = g
diff --git a/core/http/middleware/admission.go b/core/http/middleware/admission.go
index c79066925..d6134b026 100644
--- a/core/http/middleware/admission.go
+++ b/core/http/middleware/admission.go
@@ -20,7 +20,7 @@ import (
// SERVED model — a router fanout that lands on a saturated downstream
// model gets rejected even though the requested router-model has slack.
//
-// On reject: HTTP 503, Retry-After header, error JSON. An audit row
+// On reject: HTTP 429, Retry-After header, error JSON. An audit row
// goes into the shared event store under KindAdmission so admins see
// rejection rates alongside PII and proxy events.
//
@@ -39,9 +39,10 @@ func AdmissionControl(limiter *admission.Limiter, events pii.EventStore) echo.Mi
retryAfter := admission.RetryAfter(cfg.Limits.RetryAfterSeconds)
recordAdmissionRejection(events, cfg.Name, retryAfter)
c.Response().Header().Set("Retry-After", strconv.Itoa(int(retryAfter.Seconds())))
- return c.JSON(http.StatusServiceUnavailable, map[string]any{
+ return c.JSON(http.StatusTooManyRequests, map[string]any{
"error": map[string]any{
- "type": "admission_rejected",
+ "type": "rate_limit_error",
+ "code": "admission_rejected",
"message": fmt.Sprintf("model %q is at capacity (max_concurrent=%d); retry after %s", cfg.Name, max, retryAfter),
},
})
@@ -61,7 +62,7 @@ func recordAdmissionRejection(events pii.EventStore, modelName string, retryAfte
if events == nil {
return
}
- statusCode := http.StatusServiceUnavailable
+ statusCode := http.StatusTooManyRequests
durMS := retryAfter.Milliseconds()
id := fmt.Sprintf("adm_%d_%s", admissionEventSeq.Add(1), randHex(4))
_ = events.Record(context.Background(), pii.PIIEvent{
diff --git a/core/http/middleware/admission_test.go b/core/http/middleware/admission_test.go
index 841a2dd47..1e8649c2f 100644
--- a/core/http/middleware/admission_test.go
+++ b/core/http/middleware/admission_test.go
@@ -60,7 +60,7 @@ var _ = Describe("Admission", func() {
It("rejects when full", func() {
// Saturate the limiter outside the middleware, then a request
- // at the same model gets 503 with a Retry-After header.
+ // at the same model gets 429 with a Retry-After header.
lim := admission.New()
release, ok := lim.Acquire("busy", 1)
Expect(ok).To(BeTrue(), "setup acquire should succeed")
@@ -75,7 +75,7 @@ var _ = Describe("Admission", func() {
return c.String(http.StatusOK, "ok")
})
Expect(err).NotTo(HaveOccurred())
- Expect(rec.Code).To(Equal(http.StatusServiceUnavailable))
+ Expect(rec.Code).To(Equal(http.StatusTooManyRequests))
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects")
Expect(rec.Body.String()).To(ContainSubstring("admission_rejected"))
diff --git a/core/http/middleware/route_model.go b/core/http/middleware/route_model.go
index dc01ac931..0cd5f23f4 100644
--- a/core/http/middleware/route_model.go
+++ b/core/http/middleware/route_model.go
@@ -65,6 +65,31 @@ type CorpusLoader interface {
EnsureLoaded(ctx context.Context, storeName, embeddingModel, embeddingFingerprint string, embedder backend.Embedder, store backend.VectorStore) (int, error)
}
+// reseedingVectorStore runs the corpus sync before every lookup. It
+// wraps the RAW store and hands that raw store to the loader, so the
+// loader's own probe lookup never re-enters this wrapper. A sync error
+// fails the lookup closed, like the build-time load does: a decision
+// taken on an index that could not be synced is exactly the blind-
+// router bug this guards against.
+type reseedingVectorStore struct {
+ backend.VectorStore
+ ensure func(ctx context.Context) error
+}
+
+func (s *reseedingVectorStore) SearchK(ctx context.Context, vec []float32, k int) ([]backend.Neighbor, error) {
+ if err := s.ensure(ctx); err != nil {
+ return nil, fmt.Errorf("router: knn corpus sync before lookup: %w", err)
+ }
+ return s.VectorStore.SearchK(ctx, vec, k)
+}
+
+func (s *reseedingVectorStore) Search(ctx context.Context, vec []float32) (float64, []byte, bool, error) {
+ if err := s.ensure(ctx); err != nil {
+ return 0, nil, false, fmt.Errorf("router: knn corpus sync before lookup: %w", err)
+ }
+ return s.VectorStore.Search(ctx, vec)
+}
+
// ClassifierDeps bundles the backend factories the router middleware
// needs to build a classifier and its optional L2 cache. Bundled into
// one struct because RouteModel already takes many positional
@@ -478,12 +503,23 @@ func buildClassifier(cfg *config.ModelConfig, deps ClassifierDeps) (router.Class
// Loading fails closed: a live index from a different embedding
// space may have the same vector width and return plausible but
// incorrect routes.
- if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, vstore); err != nil {
+ raw := vstore
+ if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw); err != nil {
return nil, fmt.Errorf("router classifier knn: load corpus %q: %w", storeName, err)
} else if n > 0 {
xlog.Info("router: knn corpus loaded",
"router_model", cfg.Name, "store", storeName, "entries", n)
}
+ // The classifier built below is cached for the process lifetime
+ // (GetOrBuildClassifier), so this sync would otherwise be the
+ // only one — while the local-store process behind the index can
+ // be evicted or idle-killed and relaunched EMPTY at any later
+ // request. Re-check on every lookup; the corpus loader probes
+ // the live index and re-seeds it from the file on a miss.
+ vstore = &reseedingVectorStore{VectorStore: raw, ensure: func(ctx context.Context) error {
+ _, err := deps.Corpus.EnsureLoaded(ctx, storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw)
+ return err
+ }}
}
knnClassifier := router.NewKNNClassifier(embedder, vstore, router.KNNClassifierOptions{
K: rc.KNN.K,
diff --git a/core/http/middleware/route_model_test.go b/core/http/middleware/route_model_test.go
index 97aa58580..524994957 100644
--- a/core/http/middleware/route_model_test.go
+++ b/core/http/middleware/route_model_test.go
@@ -581,6 +581,29 @@ var (
errTestKNNInsert = errors.New("knn classifier must never insert into the corpus")
)
+// reseedingCorpusLoader mirrors corpus.Manager's contract: EnsureLoaded
+// is a no-op while the index still answers, and re-seeds it when the
+// store came back empty. It insists on the RAW scripted store, so a
+// wrapper leaking into the loader (and recursing) fails the spec.
+type reseedingCorpusLoader struct {
+ seed []backend.Neighbor
+ calls, reseeds int
+}
+
+func (r *reseedingCorpusLoader) EnsureLoaded(_ context.Context, _, _, _ string, _ backend.Embedder, store backend.VectorStore) (int, error) {
+ r.calls++
+ s, ok := store.(*scriptedVectorStore)
+ if !ok {
+ return 0, errors.New("corpus loader must receive the raw store, not a wrapper")
+ }
+ if len(s.neighbors) == 0 {
+ s.neighbors = r.seed
+ r.reseeds++
+ return len(r.seed), nil
+ }
+ return 0, nil
+}
+
type failingCorpusLoader struct{ err error }
func (f failingCorpusLoader) EnsureLoaded(context.Context, string, string, string, backend.Embedder, backend.VectorStore) (int, error) {
@@ -710,6 +733,43 @@ var _ = Describe("RouteModel middleware (knn classifier)", func() {
Expect(err.Error()).To(ContainSubstring("knn"))
})
+ It("re-seeds a relaunched corpus index behind the cached classifier", func() {
+ // The classifier is built once and cached; the local-store
+ // process behind its index may be evicted or idle-killed and
+ // relaunched empty afterwards. Measured in production: every probe
+ // then fell back with similarity 0 while corpus/stats still
+ // reported the full count. The lookup path must re-seed.
+ routerCfg := newKNNRouterModel(modelDir, "smart-router")
+ writeCandidate(modelDir, "small-model")
+ writeCandidate(modelDir, "big-model")
+ seeded := []backend.Neighbor{
+ {Similarity: 0.92, Payload: corpusPayload("code-generation")},
+ {Similarity: 0.88, Payload: corpusPayload("code-generation")},
+ }
+ vstore.neighbors = seeded
+ corpus := &reseedingCorpusLoader{seed: seeded}
+ deps := knnDeps()
+ deps.EmbedderFingerprint = func(string) (string, error) { return "fp", nil }
+ deps.Corpus = corpus
+ registry := router.NewRegistry()
+
+ first, err := GetOrBuildClassifier(registry, routerCfg, deps)
+ Expect(err).NotTo(HaveOccurred())
+ d, err := first.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"})
+ Expect(err).NotTo(HaveOccurred())
+ Expect(d.Labels).To(ContainElement("code-generation"))
+ Expect(corpus.reseeds).To(Equal(0), "a healthy index is not re-seeded")
+
+ vstore.neighbors = nil // the store process was relaunched empty
+ again, err := GetOrBuildClassifier(registry, routerCfg, deps)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(again).To(BeIdenticalTo(first), "the classifier stays cached — the sync must live on the lookup path")
+ d, err = again.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"})
+ Expect(err).NotTo(HaveOccurred())
+ Expect(d.Labels).To(ContainElement("code-generation"), "lookup on a relaunched index re-seeds instead of falling back")
+ Expect(corpus.reseeds).To(Equal(1))
+ })
+
It("fails closed when the persisted corpus cannot sync into the live index", func() {
routerCfg := newKNNRouterModel(modelDir, "smart-router")
writeCandidate(modelDir, "small-model")
diff --git a/core/http/react-ui/src/pages/Middleware.jsx b/core/http/react-ui/src/pages/Middleware.jsx
index 34d55d049..363b35f61 100644
--- a/core/http/react-ui/src/pages/Middleware.jsx
+++ b/core/http/react-ui/src/pages/Middleware.jsx
@@ -931,7 +931,8 @@ function eventDetails(e) {
}
case 'admission': {
const retry = e.duration_ms != null ? `retry-after ${Math.round(e.duration_ms / 1000)}s` : ''
- return `HTTP 503 rejected · ${retry}`
+ // Older audit rows were recorded as 503; newer ones as 429.
+ return `HTTP ${e.status_code || 429} rejected · ${retry}`
}
default: {
const len = e.length != null ? `len ${e.length}` : ''
diff --git a/core/schema/localai.go b/core/schema/localai.go
index 7e5d5e314..dc99a1dbe 100644
--- a/core/schema/localai.go
+++ b/core/schema/localai.go
@@ -208,6 +208,9 @@ type SysInfoModel struct {
// when the model has no local process (a distributed worker holds it) or
// the process could not be read.
Process *SysInfoProcess `json:"process,omitempty"`
+ // SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
+ // the backend process tree has no complete supported reading.
+ SizeVRAM *uint64 `json:"size_vram,omitempty"`
}
// SysInfoProcess is a point-in-time reading of one backend process.
diff --git a/core/schema/ollama.go b/core/schema/ollama.go
index 8ea414dde..ee496508f 100644
--- a/core/schema/ollama.go
+++ b/core/schema/ollama.go
@@ -293,12 +293,14 @@ type OllamaModelDetails struct {
QuantizationLevel string `json:"quantization_level,omitempty"`
}
-// OllamaModelEntry represents a model in the list response
+// OllamaModelEntry represents a model in the list response.
+// Size is a pointer so an unknown on-disk size can be omitted instead of
+// serializing as the misleading literal 0 (see issue #11969).
type OllamaModelEntry struct {
Name string `json:"name"`
Model string `json:"model"`
ModifiedAt time.Time `json:"modified_at"`
- Size int64 `json:"size"`
+ Size *int64 `json:"size,omitempty"`
Digest string `json:"digest"`
Details OllamaModelDetails `json:"details"`
Capabilities []string `json:"capabilities,omitempty"`
@@ -309,15 +311,18 @@ type OllamaListResponse struct {
Models []OllamaModelEntry `json:"models"`
}
-// OllamaPsEntry represents a running model in the ps response
+// OllamaPsEntry represents a running model in the ps response.
+// Size and SizeVRAM are pointers so unknown values are omitted rather than
+// reported as authoritative zeros (see issue #11969). SizeVRAM is only set
+// when the runtime can provide a real VRAM figure.
type OllamaPsEntry struct {
Name string `json:"name"`
Model string `json:"model"`
- Size int64 `json:"size"`
+ Size *int64 `json:"size,omitempty"`
Digest string `json:"digest"`
Details OllamaModelDetails `json:"details"`
ExpiresAt time.Time `json:"expires_at"`
- SizeVRAM int64 `json:"size_vram"`
+ SizeVRAM *int64 `json:"size_vram,omitempty"`
Capabilities []string `json:"capabilities,omitempty"`
}
diff --git a/core/schema/openai.go b/core/schema/openai.go
index 2aa69969b..6f3717256 100644
--- a/core/schema/openai.go
+++ b/core/schema/openai.go
@@ -99,7 +99,7 @@ type OpenAIResponse struct {
// OpenAI-SDK consumers that filter on a truthy `result.usage`
// (continuedev/continue, Kilo Code, Roo Code, etc.).
Usage *OpenAIUsage `json:"usage,omitempty"`
- Metadata json.RawMessage `json:"metadata,omitempty"`
+ Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"`
}
// StreamOptions mirrors OpenAI's `stream_options` request field. The only
diff --git a/core/schema/system_info_test.go b/core/schema/system_info_test.go
new file mode 100644
index 000000000..79a1bdba1
--- /dev/null
+++ b/core/schema/system_info_test.go
@@ -0,0 +1,25 @@
+// SPDX-License-Identifier: MIT
+package schema_test
+
+import (
+ "encoding/json"
+
+ "github.com/mudler/LocalAI/core/schema"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("SysInfoModel memory", func() {
+ It("omits unavailable VRAM while preserving a measured zero", func() {
+ entry := schema.SysInfoModel{ID: "model"}
+ encoded, err := json.Marshal(entry)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
+
+ zero := uint64(0)
+ entry.SizeVRAM = &zero
+ encoded, err = json.Marshal(entry)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
+ })
+})
diff --git a/core/services/agents/config.go b/core/services/agents/config.go
index 0016cf51e..875eedd36 100644
--- a/core/services/agents/config.go
+++ b/core/services/agents/config.go
@@ -107,6 +107,13 @@ type AgentConfig struct {
LoopDetection int `json:"loop_detection"`
EnableAutoCompaction bool `json:"enable_auto_compaction"`
AutoCompactionThreshold int `json:"auto_compaction_threshold"`
+
+ // Tool policy (see toolpolicy.go)
+ RequiredToolBeforeFinish string `json:"required_tool_before_finish"`
+ RequiredToolBeforeFinishPrompt string `json:"required_tool_before_finish_prompt"`
+ RequiredToolBeforeFinishAttempts int `json:"required_tool_before_finish_attempts"`
+ AllowedTools ToolNames `json:"allowed_tools"`
+ ExcludedTools ToolNames `json:"excluded_tools"`
}
// ConnectorConfig defines a connector integration (Slack, Discord, etc.).
diff --git a/core/services/agents/configmeta.go b/core/services/agents/configmeta.go
index d767aaff8..9f36ed1a9 100644
--- a/core/services/agents/configmeta.go
+++ b/core/services/agents/configmeta.go
@@ -134,6 +134,14 @@ func defaultFields() []ConfigField {
{Name: "enable_reasoning_tool", Label: "Enable Reasoning for Tools", Type: FieldCheckbox, DefaultValue: true, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
{Name: "enable_reasoning_for_instruct", Label: "Enable Reasoning for Instruct Models", Type: FieldCheckbox, DefaultValue: false, HelpText: "Force structured reasoning before tool selection (recommended for instruct-tuned models)", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
{Name: "enable_guided_tools", Label: "Enable Guided Tools", Type: FieldCheckbox, DefaultValue: false, HelpText: "Filter tools through guidance using descriptions", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
+ {Name: "allowed_tools", Label: "Allowed Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "get_document_content, search",
+ HelpText: "Comma or newline separated tool names. When set, the agent is offered only these tools (actions, knowledge base tools and MCP tools). send_message, stop and update_state are always kept. Leave empty to offer every tool.",
+ Tags: ConfigFieldTags{Section: "AdvancedSettings"},
+ },
+ {Name: "excluded_tools", Label: "Excluded Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "search_memory",
+ HelpText: "Comma or newline separated tool names that are never offered to the agent, even if they are in Allowed Tools. send_message, stop and update_state cannot be excluded; use their own settings instead.",
+ Tags: ConfigFieldTags{Section: "AdvancedSettings"},
+ },
{Name: "enable_skills", Label: "Enable Skills", Type: FieldCheckbox, DefaultValue: false, HelpText: "Inject skills into the agent", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
{Name: "skills_mode", Label: "Skills Injection Mode", Type: FieldSelect, DefaultValue: "prompt",
Options: []ConfigFieldOption{
@@ -146,6 +154,18 @@ func defaultFields() []ConfigField {
},
{Name: "parallel_jobs", Label: "Parallel Jobs", Type: FieldNumber, DefaultValue: 5, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
{Name: "max_attempts", Label: "Max Attempts", Type: FieldNumber, DefaultValue: 2, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
+ {Name: "required_tool_before_finish", Label: "Required Tool Before Finish", Type: FieldText, DefaultValue: "", Placeholder: "check_policy",
+ HelpText: "Name of a tool the agent must call successfully (a JSON result with \"ok\": true) before it may send its final answer. Has no effect if the agent does not have this tool. Leave empty to disable.",
+ Tags: ConfigFieldTags{Section: "AdvancedSettings"},
+ },
+ {Name: "required_tool_before_finish_prompt", Label: "Required Tool Prompt", Type: FieldTextarea, DefaultValue: "",
+ HelpText: "Instruction sent to the model when it tries to finish before the required tool has passed. Leave empty to use a default that names the tool.",
+ Tags: ConfigFieldTags{Section: "AdvancedSettings"},
+ },
+ {Name: "required_tool_before_finish_attempts", Label: "Required Tool Attempts", Type: FieldNumber, DefaultValue: 3, Min: 1, Step: 1,
+ HelpText: "How many times the model is told to run the required tool before its answer is sent anyway",
+ Tags: ConfigFieldTags{Section: "AdvancedSettings"},
+ },
{Name: "max_iterations", Label: "Max Iterations", Type: FieldNumber, DefaultValue: 1, Min: 1, Step: 1, HelpText: "Maximum tool loop iterations per execution", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
// MCP
diff --git a/core/services/agents/executor.go b/core/services/agents/executor.go
index 2787aeabc..9fa3c336e 100644
--- a/core/services/agents/executor.go
+++ b/core/services/agents/executor.go
@@ -4,6 +4,7 @@ import (
"cmp"
"context"
"encoding/json"
+ "errors"
"fmt"
"strings"
"time"
@@ -181,6 +182,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
// Build cogito options
var cogitoOpts []cogito.Option
+ // Local tools are collected first so the tool filter applies to all of
+ // them at once; cogito only runs tools it offered, so filtering what is
+ // offered also filters the lookup of the model's tool calls.
+ var localTools []cogito.ToolDefinitionInterface
+ filter := newToolFilter(cfg.AllowedTools, cfg.ExcludedTools)
// MCP sessions
sessions, cleanup := setupMCPSessions(ctx, cfg)
@@ -188,7 +194,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
defer cleanup()
}
if len(sessions) > 0 {
- cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...))
+ cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...), cogito.WithMCPToolFilter(filter.mcpToolFilter()))
}
// KB tools (search_memory / add_memory) — when kb mode is "tools" or "both"
@@ -197,7 +203,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
if kbResults <= 0 {
kbResults = 5
}
- cogitoOpts = append(cogitoOpts, cogito.WithTools(
+ localTools = append(localTools,
cogito.NewToolDefinition(
KBSearchMemoryTool{APIURL: effectiveURL, APIKey: effectiveKey, Collection: cfg.Name, MaxResults: kbResults, UserID: userID, CitationCollector: kbCitations},
KBSearchMemoryArgs{},
@@ -210,7 +216,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
"add_memory",
"Store content in memory for later retrieval",
),
- ))
+ )
}
// Skill tools — when skills_mode is "tools" or "both"
@@ -220,18 +226,36 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
allSkills, _ := skillProvider.ListSkills()
filtered := FilterSkills(allSkills, cfg.SelectedSkills)
if len(filtered) > 0 {
- cogitoOpts = append(cogitoOpts, cogito.WithTools(
+ localTools = append(localTools,
cogito.NewToolDefinition(
RequestSkillTool{Skills: filtered},
RequestSkillArgs{},
"request_skill",
"Request a skill by name. Available skills: "+skillNames(filtered),
),
- ))
+ )
}
}
}
+ localTools = filter.filterTools(localTools)
+ if len(localTools) > 0 {
+ cogitoOpts = append(cogitoOpts, cogito.WithTools(localTools...))
+ }
+
+ // Required-tool gate: the agent must run the configured tool to success
+ // before its answer is final. It is enforced on the output because a
+ // model follows "always call X first" unreliably.
+ requiredTool := cfg.RequiredToolBeforeFinish
+ requiredPassed := false
+ requiredAttempts := 0
+ maxRequiredAttempts := cfg.RequiredToolBeforeFinishAttempts
+ if maxRequiredAttempts <= 0 {
+ maxRequiredAttempts = defaultRequiredFinishAttempts
+ }
+ requiredPrompt := requiredFinishPromptFor(requiredTool, cfg.RequiredToolBeforeFinishPrompt)
+ requiredAvailable := requiredTool != "" && requiredToolAvailable(ctx, requiredTool, localTools, sessions, filter)
+
// Sink state is always disabled — the agent responds directly when no tools match.
cogitoOpts = append(cogitoOpts, cogito.DisableSinkState)
@@ -250,8 +274,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
}
// Tool call result callback
- if cb.OnToolResult != nil || cb.OnToolCall != nil {
+ if cb.OnToolResult != nil || cb.OnToolCall != nil || requiredAvailable {
cogitoOpts = append(cogitoOpts, cogito.WithToolCallResultCallback(func(t cogito.ToolStatus) {
+ if requiredAvailable && t.Name == requiredTool && requiredToolResultOK(t.Result) {
+ requiredPassed = true
+ }
if isInternalCogitoTool(t.Name) {
return
}
@@ -327,6 +354,32 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
return "", fmt.Errorf("agent execution failed: %w", err)
}
+ for len(result.Messages) > 0 && textFinalizationNeedsRequiredTool(requiredAvailable, requiredPassed,
+ requiredAttempts, maxRequiredAttempts, result.LastMessage().Role, result.LastMessage().Content) {
+ requiredAttempts++
+ xlog.Info("required-tool gate: answer without the required tool, nudging",
+ "agent", cfg.Name, "tool", requiredTool, "attempt", requiredAttempts)
+ answered := result
+ next, err := cogito.ExecuteTools(llm, result.AddMessage(cogito.UserMessageRole, requiredPrompt), cogitoOpts...)
+ if err != nil && ctx.Err() != nil {
+ if cb.OnStatus != nil {
+ cb.OnStatus("error: " + err.Error())
+ }
+ return "", fmt.Errorf("agent execution failed: %w", err)
+ }
+ // A failed retry must not throw away the answer the model already gave.
+ if err != nil && !errors.Is(err, cogito.ErrNoToolSelected) {
+ xlog.Error("required-tool gate: retry failed, keeping the previous answer", "agent", cfg.Name, "error", err)
+ result = answered
+ break
+ }
+ result = next
+ }
+ if requiredAvailable && !requiredPassed && requiredAttempts >= maxRequiredAttempts {
+ xlog.Warn("required-tool gate: bypass after max attempts, answer finalized ungated",
+ "agent", cfg.Name, "tool", requiredTool)
+ }
+
// Extract response
response := ""
if len(result.Messages) > 0 {
diff --git a/core/services/agents/toolpolicy.go b/core/services/agents/toolpolicy.go
new file mode 100644
index 000000000..cb5d97cdd
--- /dev/null
+++ b/core/services/agents/toolpolicy.go
@@ -0,0 +1,220 @@
+package agents
+
+// Tool policy for the distributed executor: the allowed/excluded tool lists and
+// the required-tool-before-finish gate. The semantics mirror LocalAGI's
+// core/agent/toolfilter.go and the gate in core/agent/agent.go. Those helpers
+// are unexported there, so the small pieces below are kept in step by hand; the
+// meta parity spec in toolpolicy_test.go catches drift in the form fields.
+
+import (
+ "context"
+ "encoding/json"
+ "fmt"
+ "strings"
+
+ gomcp "github.com/modelcontextprotocol/go-sdk/mcp"
+ "github.com/mudler/cogito"
+ "github.com/mudler/xlog"
+)
+
+// ToolNames is a list of tool names. The agent form submits it as a comma or
+// newline separated string and the API as a JSON array, so both are accepted.
+type ToolNames []string
+
+// UnmarshalJSON accepts a JSON array of strings, a comma or newline separated
+// string, or null. Names are trimmed and empty entries dropped.
+func (t *ToolNames) UnmarshalJSON(data []byte) error {
+ var value any
+ if err := json.Unmarshal(data, &value); err != nil {
+ return err
+ }
+ var raw []string
+ switch v := value.(type) {
+ case nil:
+ *t = nil
+ return nil
+ case string:
+ raw = strings.FieldsFunc(v, func(r rune) bool { return r == ',' || r == '\n' || r == '\r' })
+ case []any:
+ for _, item := range v {
+ name, ok := item.(string)
+ if !ok {
+ return fmt.Errorf("expected a list of tool names, got %T", item)
+ }
+ raw = append(raw, name)
+ }
+ default:
+ return fmt.Errorf("expected a list of tool names or a comma separated string, got %T", value)
+ }
+ var names ToolNames
+ for _, n := range raw {
+ if n = strings.TrimSpace(n); n != "" {
+ names = append(names, n)
+ }
+ }
+ *t = names
+ return nil
+}
+
+// controlActionNames are LocalAGI's loop-driving actions. The distributed
+// executor does not offer them today, but an agent config is shared between
+// both modes, so the filter must treat them the same way in both.
+var controlActionNames = map[string]struct{}{
+ "send_message": {},
+ "stop": {},
+ "update_state": {},
+}
+
+// toolFilter is an allow/deny list over tool names. A nil *toolFilter allows
+// everything.
+type toolFilter struct {
+ allow map[string]struct{}
+ deny map[string]struct{}
+}
+
+func newToolFilter(allow, deny []string) *toolFilter {
+ f := &toolFilter{allow: toNameSet(allow), deny: toNameSet(deny)}
+ if len(f.allow) == 0 && len(f.deny) == 0 {
+ return nil
+ }
+ return f
+}
+
+func toNameSet(names []string) map[string]struct{} {
+ set := make(map[string]struct{}, len(names))
+ for _, n := range names {
+ if n = strings.TrimSpace(n); n != "" {
+ set[n] = struct{}{}
+ }
+ }
+ return set
+}
+
+func (f *toolFilter) allows(name string) bool {
+ if f == nil {
+ return true
+ }
+ if _, ok := controlActionNames[name]; ok {
+ return true
+ }
+ if _, denied := f.deny[name]; denied {
+ return false
+ }
+ if len(f.allow) == 0 {
+ return true
+ }
+ _, allowed := f.allow[name]
+ return allowed
+}
+
+func (f *toolFilter) filterTools(tools []cogito.ToolDefinitionInterface) []cogito.ToolDefinitionInterface {
+ if f == nil {
+ return tools
+ }
+ out := make([]cogito.ToolDefinitionInterface, 0, len(tools))
+ for _, t := range tools {
+ if f.allows(t.Tool().Function.Name) {
+ out = append(out, t)
+ }
+ }
+ return out
+}
+
+// mcpToolFilter is needed on top of filterTools because cogito discovers MCP
+// tools straight from the live sessions.
+func (f *toolFilter) mcpToolFilter() cogito.MCPToolFilter {
+ if f == nil {
+ return nil
+ }
+ return func(_ *gomcp.ClientSession, toolName string) bool {
+ return f.allows(toolName)
+ }
+}
+
+// defaultRequiredFinishAttempts bounds the reminders: a gate that can loop
+// forever is worse than one that gives up loudly.
+const defaultRequiredFinishAttempts = 3
+
+func requiredFinishPromptFor(tool, override string) string {
+ if override != "" {
+ return override
+ }
+ return "Before you send your final answer you MUST first call the tool " + tool +
+ " and it must succeed (ok:true). Call " + tool + " now; only send the final " +
+ "message after it passes."
+}
+
+// requiredToolResultOK reports whether a tool result is a JSON object with a
+// top-level "ok": true. When the result is not JSON as a whole (MCP content
+// may wrap it in text), each top-level object embedded in it is checked.
+func requiredToolResultOK(result string) bool {
+ trimmed := strings.TrimSpace(result)
+ if json.Valid([]byte(trimmed)) {
+ return jsonObjectOK([]byte(trimmed))
+ }
+ for i := 0; i < len(result); {
+ j := strings.IndexByte(result[i:], '{')
+ if j < 0 {
+ return false
+ }
+ start := i + j
+ dec := json.NewDecoder(strings.NewReader(result[start:]))
+ var raw json.RawMessage
+ if err := dec.Decode(&raw); err != nil {
+ i = start + 1
+ continue
+ }
+ if jsonObjectOK(raw) {
+ return true
+ }
+ // Skip the whole object so its nested objects are not checked on their own.
+ i = start + int(dec.InputOffset())
+ }
+ return false
+}
+
+func jsonObjectOK(data []byte) bool {
+ var obj map[string]json.RawMessage
+ if err := json.Unmarshal(data, &obj); err != nil {
+ return false
+ }
+ var ok bool
+ if err := json.Unmarshal(obj["ok"], &ok); err != nil {
+ return false
+ }
+ return ok
+}
+
+// requiredToolAvailable reports whether the model is offered the required
+// tool. The gate stays inert otherwise, so a pool-wide setting is harmless for
+// agents that lack the tool. MCP sessions are only listed when the tool is not
+// a local one.
+func requiredToolAvailable(ctx context.Context, name string, local []cogito.ToolDefinitionInterface, sessions []*gomcp.ClientSession, filter *toolFilter) bool {
+ if name == "" || !filter.allows(name) {
+ return false
+ }
+ if cogito.Tools(local).Find(name) != nil {
+ return true
+ }
+ for _, s := range sessions {
+ res, err := s.ListTools(ctx, nil)
+ if err != nil {
+ xlog.Warn("required-tool gate: failed to list MCP tools", "error", err)
+ continue
+ }
+ for _, t := range res.Tools {
+ if t.Name == name {
+ return true
+ }
+ }
+ }
+ return false
+}
+
+// textFinalizationNeedsRequiredTool reports whether the run ended with a
+// non-empty assistant answer although the required tool has not passed and
+// reminders are left.
+func textFinalizationNeedsRequiredTool(toolAvailable, toolPassed bool, attempts, max int, lastRole, lastContent string) bool {
+ return toolAvailable && !toolPassed && attempts < max &&
+ lastRole == "assistant" && strings.TrimSpace(lastContent) != ""
+}
diff --git a/core/services/agents/toolpolicy_test.go b/core/services/agents/toolpolicy_test.go
new file mode 100644
index 000000000..4a18d3f80
--- /dev/null
+++ b/core/services/agents/toolpolicy_test.go
@@ -0,0 +1,387 @@
+package agents
+
+import (
+ "context"
+ "encoding/json"
+ "net/http"
+ "net/http/httptest"
+ "sort"
+ "sync"
+ "sync/atomic"
+
+ "github.com/modelcontextprotocol/go-sdk/mcp"
+ "github.com/mudler/LocalAGI/core/state"
+ "github.com/mudler/cogito"
+ openai "github.com/sashabaranov/go-openai"
+
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+// mcpFixture serves an MCP server over SSE whose tools return fixed results,
+// so the executor reaches it through the same transport as a real agent.
+type mcpFixture struct {
+ server *httptest.Server
+ calls map[string]*atomic.Int32
+}
+
+func newMCPFixture(results map[string]string) *mcpFixture {
+ srv := mcp.NewServer(&mcp.Implementation{Name: "fixture", Version: "v0.0.1"}, nil)
+ fx := &mcpFixture{calls: map[string]*atomic.Int32{}}
+ for name, result := range results {
+ counter := &atomic.Int32{}
+ fx.calls[name] = counter
+ srv.AddTool(&mcp.Tool{
+ Name: name,
+ Description: "fixture tool " + name,
+ InputSchema: json.RawMessage(`{"type":"object","properties":{}}`),
+ }, func(context.Context, *mcp.CallToolRequest) (*mcp.CallToolResult, error) {
+ counter.Add(1)
+ return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: result}}}, nil
+ })
+ }
+ fx.server = httptest.NewServer(mcp.NewSSEHandler(func(*http.Request) *mcp.Server { return srv }, nil))
+ return fx
+}
+
+func (fx *mcpFixture) close() { fx.server.Close() }
+func (fx *mcpFixture) callCount(n string) int32 { return fx.calls[n].Load() }
+
+// policyLLM answers each chat completion through respond (plain text answer
+// when respond is nil) and records every
+// request, so specs can see which tools were offered and which messages the
+// executor added.
+type policyLLM struct {
+ mu sync.Mutex
+ requests []openai.ChatCompletionRequest
+ asked [][]openai.ChatCompletionMessage
+ respond func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage
+ answer string
+}
+
+func (m *policyLLM) Ask(_ context.Context, f cogito.Fragment) (cogito.Fragment, error) {
+ m.mu.Lock()
+ m.asked = append(m.asked, append([]openai.ChatCompletionMessage(nil), f.Messages...))
+ m.mu.Unlock()
+ return f.AddMessage(cogito.AssistantMessageRole, m.answer), nil
+}
+
+func (m *policyLLM) CreateChatCompletion(_ context.Context, req openai.ChatCompletionRequest) (cogito.LLMReply, cogito.LLMUsage, error) {
+ m.mu.Lock()
+ m.requests = append(m.requests, req)
+ m.mu.Unlock()
+ msg := openai.ChatCompletionMessage{Role: "assistant", Content: m.answer}
+ if m.respond != nil {
+ msg = m.respond(req)
+ }
+ return cogito.LLMReply{
+ ChatCompletionResponse: openai.ChatCompletionResponse{
+ Choices: []openai.ChatCompletionChoice{{Message: msg}},
+ },
+ }, cogito.LLMUsage{}, nil
+}
+
+// offeredTools returns the sorted tool names of the first request that
+// offered tools to the model.
+func (m *policyLLM) offeredTools() []string {
+ m.mu.Lock()
+ defer m.mu.Unlock()
+ for _, req := range m.requests {
+ if len(req.Tools) == 0 {
+ continue
+ }
+ names := []string{}
+ for _, t := range req.Tools {
+ if t.Function != nil {
+ names = append(names, t.Function.Name)
+ }
+ }
+ sort.Strings(names)
+ return names
+ }
+ return nil
+}
+
+// nudges counts the user messages carrying prompt in the longest conversation
+// the model saw, which is the number of times the gate sent it.
+func (m *policyLLM) nudges(prompt string) int {
+ m.mu.Lock()
+ defer m.mu.Unlock()
+ best := 0
+ count := func(msgs []openai.ChatCompletionMessage) {
+ n := 0
+ for _, msg := range msgs {
+ if msg.Role == "user" && msg.Content == prompt {
+ n++
+ }
+ }
+ if n > best {
+ best = n
+ }
+ }
+ for _, req := range m.requests {
+ count(req.Messages)
+ }
+ for _, msgs := range m.asked {
+ count(msgs)
+ }
+ return best
+}
+
+func toolCallMessage(name string) openai.ChatCompletionMessage {
+ return openai.ChatCompletionMessage{
+ Role: "assistant",
+ ToolCalls: []openai.ToolCall{{
+ ID: "call-" + name,
+ Type: openai.ToolTypeFunction,
+ Function: openai.FunctionCall{Name: name, Arguments: `{}`},
+ }},
+ }
+}
+
+func lastMessage(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
+ if len(req.Messages) == 0 {
+ return openai.ChatCompletionMessage{}
+ }
+ return req.Messages[len(req.Messages)-1]
+}
+
+var _ = Describe("tool policy settings", func() {
+ Describe("config parsing", func() {
+ It("accepts the tool lists as a comma or newline separated string", func() {
+ var cfg AgentConfig
+ Expect(ParseConfigJSON(`{"allowed_tools":"a, b\nc,,","excluded_tools":" d \r\n"}`, &cfg)).To(Succeed())
+ Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b", "c"}))
+ Expect([]string(cfg.ExcludedTools)).To(Equal([]string{"d"}))
+ })
+
+ It("accepts the tool lists as a JSON array", func() {
+ var cfg AgentConfig
+ Expect(ParseConfigJSON(`{"allowed_tools":["a"," b ",""],"excluded_tools":null}`, &cfg)).To(Succeed())
+ Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b"}))
+ Expect(cfg.ExcludedTools).To(BeEmpty())
+ })
+
+ It("rejects a list with non-string entries", func() {
+ var cfg AgentConfig
+ Expect(ParseConfigJSON(`{"allowed_tools":[1]}`, &cfg)).ToNot(Succeed())
+ })
+
+ It("keeps every setting when the config is stored through LocalAGI's config", func() {
+ // The REST handlers decode into state.AgentConfig and store its JSON;
+ // the distributed dispatcher decodes that JSON into AgentConfig.
+ var in state.AgentConfig
+ Expect(json.Unmarshal([]byte(`{
+ "name": "a",
+ "allowed_tools": "search, check_policy",
+ "excluded_tools": ["add_memory"],
+ "required_tool_before_finish": "check_policy",
+ "required_tool_before_finish_prompt": "run it",
+ "required_tool_before_finish_attempts": 4
+ }`), &in)).To(Succeed())
+ stored, err := json.Marshal(in)
+ Expect(err).ToNot(HaveOccurred())
+
+ var out AgentConfig
+ Expect(ParseConfigJSON(string(stored), &out)).To(Succeed())
+ Expect([]string(out.AllowedTools)).To(Equal([]string{"search", "check_policy"}))
+ Expect([]string(out.ExcludedTools)).To(Equal([]string{"add_memory"}))
+ Expect(out.RequiredToolBeforeFinish).To(Equal("check_policy"))
+ Expect(out.RequiredToolBeforeFinishPrompt).To(Equal("run it"))
+ Expect(out.RequiredToolBeforeFinishAttempts).To(Equal(4))
+
+ again, err := json.Marshal(out)
+ Expect(err).ToNot(HaveOccurred())
+ var back state.AgentConfig
+ Expect(json.Unmarshal(again, &back)).To(Succeed())
+ Expect(back.AllowedTools).To(Equal([]string{"search", "check_policy"}))
+ Expect(back.RequiredToolBeforeFinishAttempts).To(Equal(4))
+ })
+ })
+
+ Describe("config meta", func() {
+ It("describes the settings exactly like LocalAGI does", func() {
+ upstream := map[string]ConfigField{}
+ for _, f := range state.NewAgentConfigMeta(nil, nil, nil, nil).Fields {
+ upstream[f.Name] = ConfigField{
+ Name: f.Name, Type: string(f.Type), Label: f.Label, DefaultValue: f.DefaultValue,
+ Placeholder: f.Placeholder, HelpText: f.HelpText, Min: f.Min, Max: f.Max, Step: f.Step,
+ Tags: ConfigFieldTags{Section: f.Tags.Section},
+ }
+ }
+ local := map[string]ConfigField{}
+ for _, f := range DefaultConfigMeta().Fields {
+ local[f.Name] = f
+ }
+ for _, name := range []string{
+ "allowed_tools", "excluded_tools",
+ "required_tool_before_finish", "required_tool_before_finish_prompt", "required_tool_before_finish_attempts",
+ } {
+ Expect(upstream).To(HaveKey(name))
+ Expect(local).To(HaveKeyWithValue(name, upstream[name]), name)
+ }
+ })
+ })
+
+ Describe("tool filter", func() {
+ It("keeps the control actions even when they are excluded or not allowed", func() {
+ f := newToolFilter([]string{"search"}, []string{"send_message", "stop", "update_state", "search"})
+ for _, name := range []string{"send_message", "stop", "update_state"} {
+ Expect(f.allows(name)).To(BeTrue(), name)
+ }
+ Expect(f.allows("search")).To(BeFalse())
+ Expect(f.allows("other")).To(BeFalse())
+ })
+ })
+
+ Describe("ExecuteChatWithLLM", func() {
+ var fx *mcpFixture
+
+ BeforeEach(func() {
+ fx = newMCPFixture(map[string]string{
+ "check_policy": `{"ok":true}`,
+ "mcp_allowed": "allowed result",
+ "mcp_blocked": "blocked result",
+ })
+ })
+
+ AfterEach(func() { fx.close() })
+
+ baseConfig := func() *AgentConfig {
+ return &AgentConfig{
+ Name: "policy-agent",
+ Model: "test-model",
+ MCPServers: []MCPServer{{URL: fx.server.URL}},
+ EnableKnowledgeBase: true,
+ KBMode: KBModeTools,
+ }
+ }
+
+ Context("with allowed and excluded tools", func() {
+ It("offers the model only the allowed tools that are not excluded, MCP tools included", func() {
+ llm := &policyLLM{answer: "final"}
+ cfg := baseConfig()
+ cfg.AllowedTools = ToolNames{"mcp_allowed", "search_memory", "add_memory"}
+ cfg.ExcludedTools = ToolNames{"add_memory"}
+
+ _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(llm.offeredTools()).To(Equal([]string{"mcp_allowed", "search_memory"}))
+ })
+
+ It("offers every tool when no list is set", func() {
+ llm := &policyLLM{answer: "final"}
+ _, err := ExecuteChatWithLLM(context.Background(), llm, baseConfig(), "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(llm.offeredTools()).To(Equal([]string{"add_memory", "check_policy", "mcp_allowed", "mcp_blocked", "search_memory"}))
+ })
+
+ It("does not run a filtered MCP tool the model calls anyway", func() {
+ var calls atomic.Int32
+ llm := &policyLLM{answer: "final", respond: func(openai.ChatCompletionRequest) openai.ChatCompletionMessage {
+ if calls.Add(1) == 1 {
+ return toolCallMessage("mcp_blocked")
+ }
+ return openai.ChatCompletionMessage{Role: "assistant", Content: "done"}
+ }}
+ cfg := baseConfig()
+ cfg.ExcludedTools = ToolNames{"mcp_blocked"}
+
+ _, _ = ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(fx.callCount("mcp_blocked")).To(BeZero())
+ })
+ })
+
+ Context("with a required tool before finish", func() {
+ const prompt = "RUN check_policy NOW"
+
+ It("nudges the model until the required tool passes, then returns its answer", func() {
+ llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
+ if last := lastMessage(req); last.Role == "user" && last.Content == prompt {
+ return toolCallMessage("check_policy")
+ }
+ return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"}
+ }}
+ cfg := baseConfig()
+ cfg.RequiredToolBeforeFinish = "check_policy"
+ cfg.RequiredToolBeforeFinishPrompt = prompt
+
+ result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(result).To(Equal("final answer"))
+ Expect(fx.callCount("check_policy")).To(Equal(int32(1)))
+ Expect(llm.nudges(prompt)).To(Equal(1))
+ })
+
+ It("lets the answer through after the configured number of reminders", func() {
+ llm := &policyLLM{answer: "stubborn answer"}
+ cfg := baseConfig()
+ cfg.RequiredToolBeforeFinish = "check_policy"
+ cfg.RequiredToolBeforeFinishPrompt = prompt
+ cfg.RequiredToolBeforeFinishAttempts = 2
+
+ result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(result).To(Equal("stubborn answer"))
+ Expect(llm.nudges(prompt)).To(Equal(2))
+ Expect(fx.callCount("check_policy")).To(BeZero())
+ })
+
+ It("uses three reminders and a prompt naming the tool by default", func() {
+ llm := &policyLLM{answer: "stubborn answer"}
+ cfg := baseConfig()
+ cfg.RequiredToolBeforeFinish = "check_policy"
+
+ _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(llm.nudges(requiredFinishPromptFor("check_policy", ""))).To(Equal(3))
+ Expect(requiredFinishPromptFor("check_policy", "")).To(ContainSubstring("check_policy"))
+ })
+
+ It("keeps nudging when the required tool fails", func() {
+ fx.close()
+ fx = newMCPFixture(map[string]string{"check_policy": `{"ok":false}`})
+ llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
+ if last := lastMessage(req); last.Role == "user" && last.Content == prompt {
+ return toolCallMessage("check_policy")
+ }
+ return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"}
+ }}
+ cfg := baseConfig()
+ cfg.RequiredToolBeforeFinish = "check_policy"
+ cfg.RequiredToolBeforeFinishPrompt = prompt
+ cfg.RequiredToolBeforeFinishAttempts = 2
+
+ _, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(fx.callCount("check_policy")).To(Equal(int32(2)))
+ Expect(llm.nudges(prompt)).To(Equal(2))
+ })
+
+ It("does nothing when the agent does not have the required tool", func() {
+ llm := &policyLLM{answer: "final"}
+ cfg := baseConfig()
+ cfg.RequiredToolBeforeFinish = "check_policy"
+ cfg.RequiredToolBeforeFinishPrompt = prompt
+ cfg.ExcludedTools = ToolNames{"check_policy"}
+
+ result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
+ Expect(err).ToNot(HaveOccurred())
+ Expect(result).To(Equal("final"))
+ Expect(llm.nudges(prompt)).To(BeZero())
+ })
+ })
+ })
+})
+
+var _ = DescribeTable("requiredToolResultOK",
+ func(result string, want bool) {
+ Expect(requiredToolResultOK(result)).To(Equal(want))
+ },
+ Entry("top-level ok true", `{"ok":true}`, true),
+ Entry("top-level ok false", `{"ok":false}`, false),
+ Entry("ok as a string", `{"ok":"true"}`, false),
+ Entry("ok nested in another object", `{"data":{"ok":true}}`, false),
+ Entry("object embedded in text", `result: {"ok": true, "n": 1} done`, true),
+ Entry("plain text", `"ok": true`, false),
+)
diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go
index f37a60bf8..c1cc846bd 100644
--- a/core/services/nodes/router.go
+++ b/core/services/nodes/router.go
@@ -1060,7 +1060,7 @@ func (r *SmartRouter) resolveSelectorCandidates(ctx context.Context, modelID str
return nil, fmt.Errorf("looking up nodes for selector %s: %w", sched.NodeSelector, err)
}
if len(candidates) == 0 {
- return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s", modelID, sched.NodeSelector)
+ return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s: %w", modelID, sched.NodeSelector, ErrNoAvailableNodes)
}
return extractNodeIDs(candidates), nil
}
@@ -1273,9 +1273,9 @@ func (r *SmartRouter) scheduleNewModel(ctx context.Context, backendType, modelID
evictedNode, evictErr := r.evictLRUAndFreeNodeFrom(ctx, candidateNodeIDs)
if evictErr != nil {
if errors.Is(evictErr, ErrEvictionBusy) {
- return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", evictErr)
+ return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", errors.Join(evictErr, ErrNoAvailableNodes))
}
- return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", evictErr)
+ return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", errors.Join(evictErr, ErrNoAvailableNodes))
}
node = evictedNode
}
@@ -2247,6 +2247,13 @@ func (r *SmartRouter) EvictLRU(ctx context.Context, nodeID string) (string, erro
// and none can be evicted to make room.
var ErrEvictionBusy = errors.New("all models busy, cannot evict")
+// ErrNoAvailableNodes is returned when the scheduler cannot find any healthy
+// node to serve a model — all nodes are full and eviction cannot free a slot,
+// or a node selector excludes every candidate. The HTTP layer maps this to
+// 503 so clients treat it as a transient condition rather than a server bug
+// (which is what 500 would imply).
+var ErrNoAvailableNodes = errors.New("no available nodes")
+
// evictLRUAndFreeNode finds the globally least-recently-used model with zero in-flight,
// unloads it, and returns its node for reuse. If all models are busy, retries briefly.
//
diff --git a/core/services/nodes/router_test.go b/core/services/nodes/router_test.go
index 243de0f39..90a61a1f0 100644
--- a/core/services/nodes/router_test.go
+++ b/core/services/nodes/router_test.go
@@ -828,6 +828,26 @@ var _ = Describe("SmartRouter", func() {
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("no available nodes"))
})
+
+ It("wraps ErrNoAvailableNodes when all nodes are full and eviction cannot help", func() {
+ // gorm.ErrRecordNotFound is the registry's verdict that no node
+ // matches — the scheduler then falls through to eviction. With
+ // DB nil, eviction returns ErrEvictionBusy, and the scheduler
+ // wraps the error with ErrNoAvailableNodes so the HTTP layer can
+ // map it to 503 instead of 500.
+ reg.findIdleErr = errors.New("no idle")
+ reg.findLeastLoadedErr = gorm.ErrRecordNotFound
+
+ router := NewSmartRouter(reg, SmartRouterOptions{
+ Unloader: unloader,
+ ClientFactory: factory,
+ })
+
+ _, err := router.Route(context.Background(), "m5", "models/m5.gguf", "llama-cpp", "", nil, false)
+ Expect(err).To(HaveOccurred())
+ Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
+ Expect(errors.Is(err, ErrEvictionBusy)).To(BeTrue())
+ })
})
Describe("UnloadModel (mock-based)", func() {
@@ -955,6 +975,7 @@ var _ = Describe("SmartRouter", func() {
_, err := router.Route(context.Background(), "aliased-model", "models/aliased.gguf", "llama-cpp", "", nil, false)
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
+ Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
})
It("returns error when no nodes match selector", func() {
@@ -973,6 +994,7 @@ var _ = Describe("SmartRouter", func() {
_, err := router.Route(context.Background(), "no-match-model", "models/nomatch.gguf", "llama-cpp", "", nil, false)
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
+ Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
})
It("uses regular methods when model has no scheduling config", func() {
diff --git a/core/services/quantization/service.go b/core/services/quantization/service.go
index 34a587a60..7429ca462 100644
--- a/core/services/quantization/service.go
+++ b/core/services/quantization/service.go
@@ -627,6 +627,21 @@ func sanitizeQuantModelName(s string) string {
return strings.ToLower(s)
}
+// inferenceBackendFor returns the backend that can load what a quantization
+// backend produced.
+//
+// The gallery publishes a quantizer as a release channel of the engine that
+// runs its output: "llama-cpp-quantization" is llama.cpp's quantizer, and the
+// GGUF it writes is served by "llama-cpp". The suffix is a channel marker and
+// carries no engine information, so stripping it yields the backend to pin in
+// the imported model's config. Names that carry no channel suffix (a backend
+// that both quantizes and serves, such as "rocmfp4") are already the engine
+// name and pass through unchanged, as do pinned hardware variants
+// ("rocm-rocmfp4"), which are valid values for a config's `backend:`.
+func inferenceBackendFor(quantBackend string) string {
+ return strings.TrimSuffix(config.NormalizeBackendName(quantBackend), "-quantization")
+}
+
// ImportModel imports a quantized model into LocalAI asynchronously.
func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID string, req schema.QuantizationImportRequest) (string, error) {
s.mu.Lock()
@@ -725,6 +740,17 @@ func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID str
cfg.Name = modelName
+ // The importer detects the file format and defaults to llama-cpp for any
+ // GGUF. That is wrong for a model this service just quantized with a
+ // backend stock llama.cpp cannot read: the job knows which backend
+ // produced the file, so pin that one instead of the detected default.
+ if backend := inferenceBackendFor(job.Backend); backend != "" {
+ cfg.Backend = backend
+ }
+ if job.QuantizationType != "" {
+ cfg.Description = "Quantized model (" + job.QuantizationType + ", GGUF)"
+ }
+
// Write YAML config
yamlData, err := yaml.Marshal(cfg)
if err != nil {
diff --git a/core/services/quantization/service_test.go b/core/services/quantization/service_test.go
index 4b0c804d0..1631b79ca 100644
--- a/core/services/quantization/service_test.go
+++ b/core/services/quantization/service_test.go
@@ -353,6 +353,28 @@ var _ = Describe("QuantizationService", func() {
})
})
+ Describe("imported model backend", func() {
+ It("strips the quantization channel suffix so the config pins the serving engine", func() {
+ Expect(inferenceBackendFor("llama-cpp-quantization")).To(Equal("llama-cpp"))
+ })
+
+ It("leaves a backend that both quantizes and serves unchanged", func() {
+ Expect(inferenceBackendFor("rocmfp4")).To(Equal("rocmfp4"))
+ })
+
+ It("keeps a pinned hardware variant, which is a valid backend value", func() {
+ Expect(inferenceBackendFor("rocm-rocmfp4-quantization")).To(Equal("rocm-rocmfp4"))
+ })
+
+ It("normalizes dots the way gallery names are written", func() {
+ Expect(inferenceBackendFor("llama.cpp-quantization")).To(Equal("llama-cpp"))
+ })
+
+ It("returns empty for an unset backend so the detected default is kept", func() {
+ Expect(inferenceBackendFor("")).To(BeEmpty())
+ })
+ })
+
Describe("compile-time adapter contract", func() {
It("satisfies syncstate.Store for *distributed.QuantStore", func() {
// Guards against drift between the adapter and the component interface;
diff --git a/core/services/routing/admission/admission.go b/core/services/routing/admission/admission.go
index 168248181..f37273373 100644
--- a/core/services/routing/admission/admission.go
+++ b/core/services/routing/admission/admission.go
@@ -1,6 +1,6 @@
// Package admission is routing-module subsystem 5: per-model
// concurrency control + audit. The middleware acquires a slot
-// before the handler runs; on full, the request gets 503 with
+// before the handler runs; on full, the request gets 429 with
// Retry-After so clients back off rather than pile on. The audit
// row goes into the shared event store alongside PII and proxy
// rows so admins see a single timeline of routing pressure.
diff --git a/core/services/routing/corpus/manager.go b/core/services/routing/corpus/manager.go
index bac177011..6d3d92766 100644
--- a/core/services/routing/corpus/manager.go
+++ b/core/services/routing/corpus/manager.go
@@ -30,9 +30,10 @@ import (
"sync"
"time"
+ "github.com/mudler/xlog"
+
"github.com/mudler/LocalAI/core/backend"
"github.com/mudler/LocalAI/core/services/routing/router"
- "github.com/mudler/xlog"
)
// Entry is one labelled exemplar. Vector, EmbeddingModel, and
@@ -104,6 +105,9 @@ type storeState struct {
syncedFile fileFingerprint
needsSync bool
indexedEntries int
+ // probe is a vector we inserted ourselves; storeHolds uses it to
+ // tell a live index apart from a relaunched, empty one.
+ probe []float32
}
type cachedStats struct {
@@ -172,7 +176,17 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e
}
delete(m.states, storeName)
} else if !state.needsSync && state.syncedFile.equal(fileKey) {
- return 0, nil
+ if state.indexedEntries == 0 || store == nil || storeHolds(ctx, store, state.probe) {
+ return 0, nil
+ }
+ // The file is unchanged but the live index no longer answers
+ // for a vector we inserted: the store backend was relaunched
+ // (evicted by the active-backend cap or memory pressure, then
+ // started fresh and empty on this request). Fall through and
+ // re-seed it from the file — no re-embedding, the vectors are
+ // persisted.
+ xlog.Warn("router: knn corpus index came back empty, re-seeding from file",
+ "store", storeName, "entries", state.indexedEntries)
}
}
@@ -224,10 +238,40 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e
embeddingFingerprint: embeddingFingerprint,
syncedFile: fileKey,
indexedEntries: len(entries),
+ probe: entries[0].Vector,
}
return len(entries), nil
}
+// storeHolds reports whether the live vector index still contains the
+// corpus. The local-store backend is an in-memory gRPC process: when
+// the model loader evicts it (active-backend cap, memory pressure) and
+// relaunches it on the next request, it comes back EMPTY while the
+// manager still records the file as synced — from then on every probe
+// routes to the fallback with similarity 0, and corpus/stats keeps
+// reporting the full count because it reads the file. One nearest-
+// neighbour lookup with a vector we inserted ourselves tells the two
+// states apart. An index that cannot answer is treated as empty; the
+// re-seed that follows surfaces the real error.
+func storeHolds(ctx context.Context, store backend.VectorStore, probe []float32) bool {
+ if len(probe) == 0 {
+ return true
+ }
+ sim, _, ok, err := store.Search(ctx, probe)
+ return err == nil && ok && sim > 0.999
+}
+
+// firstVector returns the vector of the first entry across lists —
+// the probe storeHolds checks the live index with.
+func firstVector(lists ...[]Entry) []float32 {
+ for _, l := range lists {
+ if len(l) > 0 {
+ return l[0].Vector
+ }
+ }
+ return nil
+}
+
// Add validates, embeds, persists, and indexes new exemplars. Entries
// whose text is already in the corpus are skipped (an exemplar's
// labels are corrected via Clear + reseed, not silent overwrite).
@@ -288,6 +332,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF
embeddingFingerprint: embeddingFingerprint,
syncedFile: m.fingerprint(storeName),
indexedEntries: len(existing),
+ probe: firstVector(existing),
}
}
seen := make(map[string]struct{}, len(existing))
@@ -377,6 +422,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF
embeddingFingerprint: embeddingFingerprint,
syncedFile: m.fingerprint(storeName),
indexedEntries: entryCount,
+ probe: firstVector(current, added),
}
} else {
m.states[storeName] = storeState{embeddingFingerprint: embeddingFingerprint, needsSync: true, indexedEntries: entryCount}
diff --git a/core/services/routing/corpus/manager_test.go b/core/services/routing/corpus/manager_test.go
index 6c50c25a6..cd99322f2 100644
--- a/core/services/routing/corpus/manager_test.go
+++ b/core/services/routing/corpus/manager_test.go
@@ -31,17 +31,34 @@ func (e *countingEmbedder) Embed(_ context.Context, text string) ([]float32, err
return []float32{float32(len(text)), e.model}, nil
}
-// capturingStore records index mutations. Search/SearchK are
-// irrelevant to the manager and return clean misses.
+// capturingStore records index mutations. Search answers like a live
+// index for vectors that were inserted (the manager probes with one of
+// its own vectors to detect a relaunched, empty store); SearchK is
+// irrelevant to the manager and returns a clean miss.
type capturingStore struct {
mu sync.Mutex
+ vecs [][]float32
payloads [][]byte
batches int
deleted [][]float32
failBatches int
}
-func (s *capturingStore) Search(_ context.Context, _ []float32) (float64, []byte, bool, error) {
+func (s *capturingStore) Search(_ context.Context, vec []float32) (float64, []byte, bool, error) {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ for i, v := range s.vecs {
+ if len(v) == len(vec) && func() bool {
+ for j := range v {
+ if v[j] != vec[j] {
+ return false
+ }
+ }
+ return true
+ }() {
+ return 1, s.payloads[i], true, nil
+ }
+ }
return 0, nil, false, nil
}
@@ -49,9 +66,10 @@ func (s *capturingStore) SearchK(_ context.Context, _ []float32, _ int) ([]backe
return nil, nil
}
-func (s *capturingStore) Insert(_ context.Context, _ []float32, payload []byte) error {
+func (s *capturingStore) Insert(_ context.Context, vec []float32, payload []byte) error {
s.mu.Lock()
defer s.mu.Unlock()
+ s.vecs = append(s.vecs, vec)
s.payloads = append(s.payloads, payload)
return nil
}
@@ -64,8 +82,8 @@ func (s *capturingStore) InsertBatch(_ context.Context, vecs [][]float32, payloa
s.failBatches--
return errors.New("transient batch failure")
}
+ s.vecs = append(s.vecs, vecs...)
s.payloads = append(s.payloads, payloads...)
- _ = vecs
return nil
}
@@ -117,6 +135,30 @@ var _ = Describe("corpus.Manager", func() {
_ = os.RemoveAll(dir)
})
+ It("re-seeds the index when the store comes back empty under an unchanged file", func() {
+ // The local-store backend is an in-memory process the model loader
+ // may evict (active-backend cap) and relaunch empty on the next
+ // request. The file is untouched, so the file fingerprint alone
+ // says "synced" — and the router goes blind: every probe falls back
+ // with similarity 0 while corpus/stats still reports the full count.
+ _, _, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed)
+ Expect(err).NotTo(HaveOccurred())
+ n, err := mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, store)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(n).To(Equal(0), "live index holds the corpus: nothing to do")
+
+ relaunched := &capturingStore{}
+ n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(n).To(Equal(len(seed)), "empty index under an unchanged file is re-seeded")
+ Expect(relaunched.payloads).To(HaveLen(len(seed)))
+ Expect(embedder.calls).To(Equal(len(seed)), "vectors come from the file, nothing is re-embedded")
+
+ n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(n).To(Equal(0), "and the relaunched index counts as synced again")
+ })
+
It("adds entries: embeds, persists, and indexes them", func() {
added, skipped, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed)
Expect(err).NotTo(HaveOccurred())
diff --git a/core/services/routing/pii/types.go b/core/services/routing/pii/types.go
index c2e2510df..ad15ea462 100644
--- a/core/services/routing/pii/types.go
+++ b/core/services/routing/pii/types.go
@@ -109,7 +109,7 @@ const (
// model's MaxConcurrent ceiling is full. The Host field carries
// the model name (overloading the existing column rather than
// adding a new one — admins read it as "the thing that was
- // busy"); StatusCode is 503.
+ // busy"); StatusCode is 429.
KindAdmission EventKind = "admission"
)
diff --git a/docker-compose.yaml b/docker-compose.yaml
index ee137e83c..82b3c18b6 100644
--- a/docker-compose.yaml
+++ b/docker-compose.yaml
@@ -59,6 +59,7 @@ services:
# capabilities: [gpu, utility]
#
# For legacy NVIDIA driver (for older NVIDIA Container Toolkit):
+ # Request compute for CUDA libraries (libcuda.so.1) and utility for NVML.
# environment:
# NVIDIA_DRIVER_CAPABILITIES: "compute,utility"
# init: true
@@ -68,7 +69,7 @@ services:
# devices:
# - driver: nvidia
# count: 1
- # capabilities: [gpu, utility]
+ # capabilities: [gpu, compute, utility]
## Uncomment for PostgreSQL-backed knowledge base (see Agents docs)
# postgres:
diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md
index f7d9546cf..8580aa499 100644
--- a/docs/content/advanced/advanced-usage.md
+++ b/docs/content/advanced/advanced-usage.md
@@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model
local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master
```
-See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files.
+See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files.
### Prompt templates
diff --git a/docs/content/features/agents.md b/docs/content/features/agents.md
index cde92862e..4e9026942 100644
--- a/docs/content/features/agents.md
+++ b/docs/content/features/agents.md
@@ -188,6 +188,8 @@ Each agent has its own configuration that controls its behavior. Key settings in
- **Connectors** - external integrations (Slack, Discord, etc.)
- **Knowledge Base** - collections of documents for RAG
- **MCP Servers** - Model Context Protocol servers for additional tool access
+- **Allowed / Excluded Tools** (`allowed_tools`, `excluded_tools`) - limit the tools the agent can see, including MCP tools. The agent always keeps its control actions (`send_message`, `stop`, `update_state`). If a tool is in both lists, it is excluded.
+- **Required Tool Before Finish** (`required_tool_before_finish`) - a tool the agent must call successfully before it can give its final answer, for example a validation or policy check. `required_tool_before_finish_prompt` changes the reminder the model gets when it tries to finish early. `required_tool_before_finish_attempts` sets how many reminders it gets before the answer goes through anyway (default 3).
The pool-level defaults (API URL, API key, models) can be set via environment variables. Individual agents can further override these in their configuration, allowing them to use different LLM providers (OpenAI, other LocalAI instances, etc.) on a per-agent basis.
diff --git a/docs/content/features/api-discovery.md b/docs/content/features/api-discovery.md
index f45524423..932634a39 100644
--- a/docs/content/features/api-discovery.md
+++ b/docs/content/features/api-discovery.md
@@ -141,6 +141,11 @@ curl http://localhost:8080/api/instructions/config-management?format=json
An additive, LocalAI-specific superset of `/v1/models`. It returns the same set of models but enriches each entry with the **capabilities** the model supports and the **input/output modalities** it accepts and produces. Use it to decide, before sending a request, whether a given model can take an image, audio, or video attachment directly - or whether the input needs converting/transcribing first.
+The reported `context_size` uses a positive model-level value first.
+If the model does not set `context_size`, it uses **Settings → Performance → Default Context Size** when positive.
+Otherwise, it uses the backend fallback of 4096 tokens.
+For llama.cpp with separate KV caches, the reported value accounts for the number of parallel slots.
+
Because it is purely additive, clients that only understand `/v1/models` keep working unchanged; they simply never call this route.
```bash
diff --git a/docs/content/features/backends.md b/docs/content/features/backends.md
index 6085cf536..c0818df85 100644
--- a/docs/content/features/backends.md
+++ b/docs/content/features/backends.md
@@ -82,6 +82,8 @@ tags:
### Verifying OCI Backends
+The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details.
+
Backend galleries can require keyless Sigstore signatures for every OCI image
they provide. Add a `verification` policy to the gallery configuration, then
enable strict integrity mode:
diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md
index ca507703c..6fe983de2 100644
--- a/docs/content/features/distributed-mode.md
+++ b/docs/content/features/distributed-mode.md
@@ -952,8 +952,12 @@ usage is reported back to the frontend:
NVML library (and therefore `nvidia-smi`) is not available inside the
container. CUDA compute still works, but the worker cannot query free VRAM
and the Nodes page will show the node as fully used. Set
- `NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI
- runtime, list `capabilities: [gpu, utility]` on the device reservation).
+ `NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime.
+ For Docker Compose with `driver: nvidia`, use
+ `capabilities: [gpu, compute, utility]` on the device reservation.
+ Docker derives driver capabilities from this reservation, so include `compute`
+ for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone
+ enables monitoring but does not provide CUDA libraries.
- **Run the container with `init: true` (or `docker run --init`).** The
worker process becomes PID 1 in the container and cannot reap zombies on
diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md
index dd48e73c7..0bee9c890 100644
--- a/docs/content/features/model-gallery.md
+++ b/docs/content/features/model-gallery.md
@@ -39,6 +39,61 @@ Both views use the same model selection and store the view, search, filter, and
selection in the URL. Installing from Explore does not move you away from the
catalog; the entry updates in place when the operation finishes.
+## Cyber-Tiel-Coder
+
+Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp.
+The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector.
+To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`.
+Both configurations use the embedded chat template and default to 32,768 context tokens.
+The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license.
+
+## Qwen3.8-27B Agention Precision
+
+The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of
+Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image
+input and use a 32,768-token context by default.
+
+Install with automatic variant selection:
+
+```bash
+local-ai models install qwen3.8-27b-agention-iq4-xs
+```
+
+To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs`
+or `--variant qwen3.8-27b-agention-q4-k-m` to the same command.
+The files use standard llama.cpp quantization types and the Apache-2.0 license.
+See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF)
+for quantization details. These entries do not enable MTP speculative decoding.
+
+## Swift 1.5 Qwen3.8-27B GSQ-RCO
+
+Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp.
+The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model.
+To select IQ3_S explicitly, run:
+
+```bash
+local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
+```
+
+The configurations use the embedded chat template and default to 32,768 context tokens.
+These builds support text chat only: the publisher has no verified vision projector for this release.
+They use standard GGUF files without MTP decoding.
+See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms.
+
+## Sharp-Spark-X2.5-4B
+
+Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp.
+The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants.
+To select the publisher's recommended Q6 build, run:
+
+```bash
+local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6
+```
+
+All builds use a 32,768-token default context and the embedded Sharp-Spark chat template.
+That template adds a terseness instruction to the system prompt.
+See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details.
+
## MiMo-V2.6-Distill-Qwen-9B
Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp.
@@ -48,6 +103,35 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9
The configurations default to 32,768 context tokens and use the model's embedded chat template.
See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details.
+## Qwopus3.8 Flash V2
+
+Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection:
+
+```bash
+local-ai models install qwopus3.8-27b-flash-v2
+local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8
+```
+
+Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector.
+Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256.
+This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks.
+See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations.
+
+## ThinkingCap Qwen3.8-27B
+
+Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input.
+The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector.
+LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly:
+
+```bash
+local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8
+```
+
+Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings.
+MTP speculative decoding is not enabled by these entries.
+The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE).
+Review that license for permitted use.
+
## Hemmingway-1
Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants.
@@ -97,7 +181,7 @@ To use a gallery that needs authentication, such as a private GitHub repository
A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted.
-Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
+Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
```json
GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
@@ -151,10 +235,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out
### Signature verification
-An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
+An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
```json
-GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
+GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
```
The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends.
@@ -173,9 +257,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK
The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}).
{{% notice warning %}}
-With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
+`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities.
+
+With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
{{% /notice %}}
+### Official gallery publishing
+
+The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA.
+
+To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title.
+
### Private registries
A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use:
@@ -207,10 +299,10 @@ GALLERIES=[{"name":"", "url":"}}
{{% tab title="Apple" %}}
+To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later.
+Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts.
+Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables.
+
Install `xcode` from the App Store
```bash
diff --git a/docs/content/getting-started/models.md b/docs/content/getting-started/models.md
index 9c6ce32f7..2130dfc58 100644
--- a/docs/content/getting-started/models.md
+++ b/docs/content/getting-started/models.md
@@ -391,6 +391,8 @@ See the [Model Configuration]({{% relref "advanced/model-configuration" %}}) gui
### List Installed Models
+Ollama clients can list configured models with `GET /api/tags` and loaded models with `GET /api/ps`. Each entry includes `size` in bytes when LocalAI can resolve a non-empty primary weights file on disk. This is the size of that file, not the total size of a multi-file model or its memory use. Unknown sizes are omitted; `/api/ps` also omits `size_vram` because per-model VRAM use is not available.
+
```bash
# Via API
curl http://localhost:8080/v1/models
diff --git a/docs/content/operations/cloud-proxy.md b/docs/content/operations/cloud-proxy.md
index 02af25bd0..a258ece14 100644
--- a/docs/content/operations/cloud-proxy.md
+++ b/docs/content/operations/cloud-proxy.md
@@ -60,9 +60,9 @@ against - and two modes:
`proxy.provider` selects the auth scheme and (in translate mode) the wire
format. Supported values: `openai`, `anthropic`.
-API keys are loaded from either an environment variable (`api_key_env`) or a
-file (`api_key_file`). The key never appears in the config file or the admin
-UI; pick whichever fits your secret-management setup.
+If the upstream requires an API key, configure either an environment variable
+(`api_key_env`) or a file (`api_key_file`). The key never appears in the config
+file or the admin UI. If the upstream requires no API key, omit both fields.
### OpenAI passthrough
@@ -126,7 +126,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with
Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak
the OpenAI chat-completions wire format. Use `provider: openai` with the
-provider's URL and API key:
+provider's URL and, if required, its API key:
```yaml
name: llama-3-70b-via-together
@@ -140,6 +140,37 @@ proxy:
upstream_model: meta-llama/Llama-3-70b-chat-hf
```
+### Upstreams without an API key
+
+For an OpenAI-compatible upstream that accepts requests without authentication,
+omit both `api_key_env` and `api_key_file`:
+
+```yaml
+name: internal-chat-proxy
+backend: cloud-proxy
+
+proxy:
+ mode: passthrough
+ provider: openai
+ upstream_url: http://inference.internal:8000/v1/chat/completions
+ upstream_model: my-model
+```
+
+Replace the example URL and model name with your upstream's values. LocalAI
+loads this configuration without resolving a key and adds no upstream
+`Authorization` header. This also applies to OpenAI-compatible upstreams in
+translate mode.
+
+Omitting both fields differs from setting `api_key_env` to an empty or unset
+environment variable: the latter causes a backend load error.
+
+LocalAI's client authentication is separate. Clients must still authenticate
+to LocalAI when its authentication is enabled. LocalAI does not forward their
+`Authorization` header to the upstream.
+
+An upstream without API keys can still require another authentication or
+payment protocol. Omitting these fields does not implement that protocol.
+
### Translate mode
In translate mode the cloud-proxy backend converts LocalAI's internal proto
diff --git a/docs/content/operations/middleware.md b/docs/content/operations/middleware.md
index 43ab02dc1..ac97a10cd 100644
--- a/docs/content/operations/middleware.md
+++ b/docs/content/operations/middleware.md
@@ -558,7 +558,12 @@ The corpus is persisted as one JSONL file per router under
`/router-corpus/` (text, labels, vector, embedding-model name,
and embedding fingerprint) — **the file is the source of truth** and
survives restarts; the local-store index is rebuilt from it at classifier
-build time without re-embedding. The fingerprint follows the effective
+build time without re-embedding. Before each KNN lookup, LocalAI checks a stored
+vector against the live index. If the store restarts empty after eviction or
+an idle timeout, LocalAI restores the index from the file without re-embedding.
+A synchronization error fails the lookup.
+
+The fingerprint follows the effective
embedding-model config and local artifact identity, so changing the model or
replacing its local weights re-embeds the corpus on the next process load.
For remote embedding services whose weights can change invisibly, bump
diff --git a/docs/content/reference/api-errors.md b/docs/content/reference/api-errors.md
index 9bd9dea02..20f63786f 100644
--- a/docs/content/reference/api-errors.md
+++ b/docs/content/reference/api-errors.md
@@ -88,7 +88,9 @@ The `/v1/responses` endpoint returns errors with this structure:
| 404 | Not Found | Model or resource does not exist |
| 409 | Conflict | Resource already exists (e.g., duplicate token) |
| 422 | Unprocessable Entity | Validation failed (e.g., invalid parameter range) |
+| 429 | Too Many Requests | All backends are saturated (per-model `max_concurrent` or process-wide `--max-concurrent-backend-requests` ceiling reached). Includes a `Retry-After` header and `type: "rate_limit_error"` so OpenAI-compatible clients and harnesses back off automatically |
| 500 | Internal Server Error | Backend inference failure, unexpected server errors |
+| 503 | Service Unavailable | No healthy node available to serve the model (cluster is full, eviction cannot free a slot, or a `node_selector` excludes all candidates). Also used during model-load cooldown and while a model is still cold-loading. Retryable |
## Global Error Handling
diff --git a/docs/content/reference/cli-reference.md b/docs/content/reference/cli-reference.md
index c825c1ca2..92e8b2458 100644
--- a/docs/content/reference/cli-reference.md
+++ b/docs/content/reference/cli-reference.md
@@ -95,7 +95,7 @@ For more information on VRAM management, see [VRAM and Memory Management]({{%rel
| Parameter | Default | Description | Environment Variable |
|-----------|---------|-------------|----------------------|
| `--address` | `:8080` | Bind address for the API server | `$LOCALAI_ADDRESS`, `$ADDRESS` |
-| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 503 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
+| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 429 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
| `--cors` | `false` | Enable CORS (Cross-Origin Resource Sharing) | `$LOCALAI_CORS`, `$CORS` |
| `--cors-allow-origins` | | Comma-separated list of allowed CORS origins | `$LOCALAI_CORS_ALLOW_ORIGINS`, `$CORS_ALLOW_ORIGINS` |
| `--disable-csrf` | `false` | Disable CSRF middleware (enabled by default) | `$LOCALAI_DISABLE_CSRF` |
diff --git a/docs/content/reference/nvidia-l4t.md b/docs/content/reference/nvidia-l4t.md
index 2adac3a84..e3b54020a 100644
--- a/docs/content/reference/nvidia-l4t.md
+++ b/docs/content/reference/nvidia-l4t.md
@@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things:
NVML work inside the container. With `--gpus all` alone (or
`--runtime nvidia` without extra flags) only `compute` is wired in on
some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility`
- to your `docker run`, or `capabilities: [gpu, utility]` in compose /
- Kubernetes device reservations.
+ to your `docker run`. For Docker Compose with `driver: nvidia`, use
+ `capabilities: [gpu, compute, utility]` on the device reservation.
+ Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility`
+ alone only provides monitoring libraries and tools.
2. Pass `--init` to `docker run` (or `init: true` in compose) so the
container has a proper PID 1 reaper - otherwise short-lived child
processes like `nvidia-smi` can intermittently fail with
diff --git a/docs/content/reference/runtime-errors.md b/docs/content/reference/runtime-errors.md
index 1e14dc0b5..fd5ef2bd8 100644
--- a/docs/content/reference/runtime-errors.md
+++ b/docs/content/reference/runtime-errors.md
@@ -21,8 +21,9 @@ The left column is the literal string as it appears in the LocalAI server log (o
| `grpc service not ready` | The backend process was spawned but its gRPC server did not become healthy in time (slow start, crash on startup, or the process died while loading). When a local backend has already exited, the error includes its exit code and last stderr line. | Use the included stderr diagnostic when present; otherwise check the log lines just above. A crash here often means out of memory, a missing shared library, or an incompatible CPU (see `SIGILL`). Increase available RAM/VRAM or pick a smaller quantization. |
| `failed to load model: ...` | Returned by the load endpoints and several feature paths (voice, realtime, audio transform) when the model config could not be resolved or the backend load failed. | Confirm the model name exists (`local-ai models list`) and its YAML is valid. The trailing text carries the specific reason. |
| HTTP `503` with a `Retry-After` header, after a load failed | Model-load failure cooldown. After a model fails to load, LocalAI refuses new load attempts for that model for a short window so a client that keeps polling a broken model does not respawn a crashing backend on every request. The window starts at `--model-load-failure-cooldown` (default `10s`) and doubles per consecutive failure up to 5m; it resets on the first success. | Fix the underlying load failure (see the rows above), then wait out the `Retry-After` seconds before retrying, or restart LocalAI to clear the cooldown. Set `--model-load-failure-cooldown 0` (or `LOCALAI_MODEL_LOAD_FAILURE_COOLDOWN=0`) to disable the cooldown entirely. See {{% relref "reference/cli-reference" %}}. |
-| HTTP `503` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `503` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
-| HTTP `503` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
+| HTTP `503` with `no available nodes` or `no healthy nodes match selector` | The scheduler could not find any healthy node to serve the model. All nodes are full and eviction cannot free a slot, or a `node_selector` in the model's scheduling config excludes every candidate. | Retry after a node becomes available or an in-flight request completes and frees a slot. In a cluster, add nodes or replicas. If a selector is set, confirm at least one healthy node matches it. |
+| HTTP `429` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `429` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
+| HTTP `429` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
| `invalid pitch` (with CUDA) | The prompt exceeded the model's context size. | Reduce the prompt length, or raise the model's context size (`context_size:` in the model YAML). |
| `SIGILL` (illegal instruction) on startup | The prebuilt backend binary uses CPU instructions your CPU does not have (for example AVX512, AVX2, F16C, FMA). | Rebuild the backend for your CPU. In a container, set `REBUILD=true` and disable the unsupported instructions, for example `CMAKE_ARGS="-DGGML_F16C=OFF -DGGML_AVX512=OFF -DGGML_AVX2=OFF -DGGML_FMA=OFF" make build`. |
| CUDA / VRAM out of memory (backend log shows `out of memory`, `CUDA error: out of memory`, or the process is killed loading) | The model plus its KV cache does not fit in GPU memory. | Use a smaller quantization, reduce `context_size:`, offload fewer layers to the GPU (lower `gpu_layers:`), or free VRAM held by other processes. On multi-GPU hosts, confirm the model is not trying to load entirely onto one device. |
diff --git a/docs/content/reference/system-info.md b/docs/content/reference/system-info.md
index b825e4e06..ed4de00ff 100644
--- a/docs/content/reference/system-info.md
+++ b/docs/content/reference/system-info.md
@@ -28,6 +28,29 @@ Returns available backends and currently loaded models.
| `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM |
| `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against |
| `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) |
+| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes |
+
+### Per-model VRAM
+
+On Linux, `size_vram` reports resident device memory for the local backend
+process and its child processes. LocalAI reads `drm-resident-local*` and
+`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU.
+Host-memory regions are excluded. The reading includes buffers attributed
+to the backend, without separating weights, KV cache, and other allocations.
+See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html)
+for these counters.
+
+The field is omitted when accounting is unavailable or incomplete. This
+includes external and distributed backends, macOS, proprietary NVIDIA
+drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters,
+and unreadable process information.
+A present value of `0` means the supported counters report zero bytes.
+Treat an absent field as unknown.
+
+This is a snapshot of driver accounting, not a memory reservation. Shared
+buffers can appear in different clients' counters, and allocations can change
+during collection. Do not treat the sum across models as exclusive physical
+GPU usage. These readings do not replace capacity checks when scheduling work.
### Usage
@@ -49,6 +72,7 @@ curl http://localhost:8080/system
{
"id": "my-llama-model",
"backend": "llama-cpp",
+ "size_vram": 5368709120,
"process": {
"pid": 48213,
"rss_bytes": 5368709120,
diff --git a/gallery/index.yaml b/gallery/index.yaml
index 84733ea18..f6fa9361c 100644
--- a/gallery/index.yaml
+++ b/gallery/index.yaml
@@ -1,4 +1,96 @@
---
+- name: "ternary-bonsai-2-27b"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
+ - https://github.com/PrismML-Eng/llama.cpp
+ description: |
+ Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary
+ transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits
+ per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a
+ Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's
+ llama.cpp fork) instead of stock llama.cpp.
+ license: "apache-2.0"
+ tags:
+ - llm
+ - gguf
+ - reasoning
+ - vision
+ - multimodal
+ icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg
+ overrides:
+ backend: bonsai
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
+ options:
+ - use_jinja:true
+ parameters:
+ model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
+ sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3
+ uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf
+ - filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
+ sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903
+ uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
+- name: "swift-qwen3.8-27b"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/ukisai/Swift-Qwen3.8-27b
+ - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF
+ description: |
+ Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B.
+ The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss.
+ This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding.
+ The weights use the Swift Open License v1.0.
+ license: "swift-open-license-1.0"
+ tags:
+ - llm
+ - gguf
+ - reasoning
+ - vision
+ - multimodal
+ - mtp
+ overrides:
+ backend: llama-cpp
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
+ options:
+ - use_jinja:true
+ - spec_type:draft-mtp
+ - spec_n_max:6
+ - spec_p_min:0.75
+ parameters:
+ min_p: 0
+ model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
+ presence_penalty: 1.5
+ repeat_penalty: 1
+ temperature: 0.7
+ top_k: 20
+ top_p: 0.8
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
+ sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab
+ uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf
+ - filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
+ sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a
+ uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf
- name: "ornith-1.5-9b-uncensored"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
@@ -205,7 +297,101 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
- sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7
+ sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
+- name: "qwopus3.8-27b-flash-v2"
+ variants:
+ - model: qwopus3.8-27b-flash-v2-q8
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
+ - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
+ description: |
+ Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
+ workloads. This Q4_K_M GGUF includes the F32 vision projector and uses
+ llama.cpp's embedded chat template with MTP speculative decoding.
+ license: "apache-2.0"
+ tags:
+ - llm
+ - gguf
+ - qwen
+ - qwen3
+ - vision
+ - multimodal
+ - instruction-tuned
+ - reasoning
+ - mtp
+ icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
+ overrides:
+ backend: llama-cpp
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
+ options:
+ - use_jinja:true
+ - spec_type:draft-mtp
+ - spec_n_max:6
+ - spec_p_min:0.75
+ parameters:
+ model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
+ uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
+ sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e
+ - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
+ uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
+ sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
+- name: "qwopus3.8-27b-flash-v2-q8"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
+ - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
+ description: |
+ Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
+ workloads. This Q8_0 GGUF includes the F32 vision projector and uses
+ llama.cpp's embedded chat template with MTP speculative decoding.
+ license: "apache-2.0"
+ tags:
+ - llm
+ - gguf
+ - qwen
+ - qwen3
+ - vision
+ - multimodal
+ - instruction-tuned
+ - reasoning
+ - mtp
+ icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
+ overrides:
+ backend: llama-cpp
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
+ options:
+ - use_jinja:true
+ - spec_type:draft-mtp
+ - spec_n_max:6
+ - spec_p_min:0.75
+ parameters:
+ model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
+ uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
+ sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78
+ - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
+ uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
+ sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
- name: "qwopus3.8-27b-flash"
variants:
- model: qwopus3.8-27b-flash-q8
@@ -388,6 +574,102 @@
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
+- name: thinkingcap-qwen3.8-27b
+ variants:
+ - model: thinkingcap-qwen3.8-27b-q8
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
+ - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
+ description: |
+ ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
+ This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
+ Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
+ license: polyform-small-business-1.0.0
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - vision
+ - multimodal
+ - reasoning
+ last_checked: "2026-09-27"
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
+ options:
+ - use_jinja:true
+ template:
+ use_tokenizer_template: true
+ parameters:
+ model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
+ temperature: 1.0
+ top_p: 0.95
+ top_k: 20
+ min_p: 0.0
+ files:
+ - filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
+ sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299
+ uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
+ - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
+ sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
+ uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
+- name: thinkingcap-qwen3.8-27b-q8
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
+ - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
+ description: |
+ ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
+ This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
+ Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
+ license: polyform-small-business-1.0.0
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - vision
+ - multimodal
+ - reasoning
+ last_checked: "2026-09-27"
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
+ options:
+ - use_jinja:true
+ template:
+ use_tokenizer_template: true
+ parameters:
+ model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
+ temperature: 1.0
+ top_p: 0.95
+ top_k: 20
+ min_p: 0.0
+ files:
+ - filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
+ sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7
+ uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf
+ - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
+ sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
+ uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
- name: hemmingway-1
variants:
- model: hemmingway-1-q8
@@ -4767,6 +5049,110 @@
- filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf
uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf
sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41
+- name: cyber-tiel-coder-35b-a3b-q4-mtp
+ variants:
+ - model: cyber-tiel-coder-35b-a3b-q8-mtp
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ license: mit
+ urls:
+ - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
+ - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
+ description: |
+ Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
+ based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes
+ MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - qwen
+ - moe
+ - coding
+ - tools
+ - vision
+ - multimodal
+ - mtp
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
+ options:
+ - use_jinja:true
+ - spec_type:draft-mtp
+ parameters:
+ model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
+ temperature: 0.6
+ top_p: 0.95
+ top_k: 20
+ min_p: 0.0
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
+ sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a
+ - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
+ sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
+- name: cyber-tiel-coder-35b-a3b-q8-mtp
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ license: mit
+ urls:
+ - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
+ - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
+ description: |
+ Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
+ based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes
+ MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - qwen
+ - moe
+ - coding
+ - tools
+ - vision
+ - multimodal
+ - mtp
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ - vision
+ mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
+ options:
+ - use_jinja:true
+ - spec_type:draft-mtp
+ parameters:
+ model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
+ temperature: 0.6
+ top_p: 0.95
+ top_k: 20
+ min_p: 0.0
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
+ sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961
+ - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
+ sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
- &tiel-coder-35b-a3b
name: "tiel-coder-35b-a3b-q4"
variants:
@@ -5247,6 +5633,104 @@
- filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf
uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf
sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545
+- name: qwen3.8-27b-agention-iq4-xs
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ variants:
+ - model: qwen3.8-27b-agention-q4-k-m
+ urls:
+ - https://huggingface.co/Qwen/Qwen3.8-27B
+ - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
+ license: apache-2.0
+ description: |
+ Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp.
+ This 27B reasoning model supports text and image input. The download
+ includes the BF16 vision projector and uses the embedded chat template.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - qwen
+ - reasoning
+ - vision
+ - multimodal
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ known_usecases:
+ - chat
+ - vision
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
+ options:
+ - use_jinja:true
+ parameters:
+ model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
+ temperature: 1
+ top_p: 0.95
+ top_k: 20
+ min_p: 0
+ repeat_penalty: 1
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
+ uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf
+ sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67
+ - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
+ uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
+ sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
+- name: qwen3.8-27b-agention-q4-k-m
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ urls:
+ - https://huggingface.co/Qwen/Qwen3.8-27B
+ - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
+ license: apache-2.0
+ description: |
+ Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp.
+ This 27B reasoning model supports text and image input. The download
+ includes the BF16 vision projector and uses the embedded chat template.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - qwen
+ - reasoning
+ - vision
+ - multimodal
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ known_usecases:
+ - chat
+ - vision
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
+ options:
+ - use_jinja:true
+ parameters:
+ model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
+ temperature: 1
+ top_p: 0.95
+ top_k: 20
+ min_p: 0
+ repeat_penalty: 1
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
+ uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf
+ sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e
+ - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
+ uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
+ sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
- &qwen3-8-27b
name: "qwen3.8-27b-q4"
variants:
@@ -5533,6 +6017,178 @@
- filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf
uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf
sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2
+- name: "swift-1.5-qwen3.8-27b-gsq-rco"
+ variants:
+ - model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s
+ - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs
+ - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
+ license: "swift-open-license-1.0"
+ description: |
+ Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
+ This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
+ Text chat only; the publisher provides no verified vision projector for this release.
+ The weights use the Swift Open License v1.0.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - reasoning
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ min_p: 0
+ model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
+ presence_penalty: 0
+ repeat_penalty: 1
+ temperature: 1
+ top_k: 20
+ top_p: 0.95
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
+ sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8
+ uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
+- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
+ license: "swift-open-license-1.0"
+ description: |
+ Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
+ This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
+ Text chat only; the publisher provides no verified vision projector for this release.
+ The weights use the Swift Open License v1.0.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - reasoning
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ min_p: 0
+ model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
+ presence_penalty: 0
+ repeat_penalty: 1
+ temperature: 1
+ top_k: 20
+ top_p: 0.95
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
+ sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455
+ uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
+- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
+ license: "swift-open-license-1.0"
+ description: |
+ Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
+ This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
+ Text chat only; the publisher provides no verified vision projector for this release.
+ The weights use the Swift Open License v1.0.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - reasoning
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ min_p: 0
+ model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
+ presence_penalty: 0
+ repeat_penalty: 1
+ temperature: 1
+ top_k: 20
+ top_p: 0.95
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
+ sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2
+ uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
+- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s"
+ url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
+ urls:
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
+ - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
+ license: "swift-open-license-1.0"
+ description: |
+ Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
+ This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
+ Text chat only; the publisher provides no verified vision projector for this release.
+ The weights use the Swift Open License v1.0.
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - reasoning
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ function:
+ automatic_tool_parsing_fallback: true
+ grammar:
+ disable: true
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ min_p: 0
+ model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
+ presence_penalty: 0
+ repeat_penalty: 1
+ temperature: 1
+ top_k: 20
+ top_p: 0.95
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
+ sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786
+ uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
- !!merge <<: *qwen3-8-27b
name: "qwen3.8-27b-gsq-rco-iq2-xs"
variants: []
@@ -5725,6 +6381,120 @@
- filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf
uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf
sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442
+- name: sharp-spark-x2.5-4b
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ variants:
+ - model: sharp-spark-x2.5-4b-q5
+ - model: sharp-spark-x2.5-4b-q6
+ urls:
+ - https://huggingface.co/XHToken/Spark-X2.5-4B
+ - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
+ description: |
+ Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
+ with an adjusted chat template for coding. This Q4_K_XL build uses the
+ embedded Sharp-Spark template and a 32K-token default context.
+ license: apache-2.0
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - coding
+ - reasoning
+ last_checked: "2026-09-26"
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
+ temperature: 0.6
+ top_p: 0.95
+ top_k: 20
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
+ sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
+
+- name: sharp-spark-x2.5-4b-q5
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ urls:
+ - https://huggingface.co/XHToken/Spark-X2.5-4B
+ - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
+ description: |
+ Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
+ with an adjusted chat template for coding. This Q5_K_XL build uses the
+ embedded Sharp-Spark template and a 32K-token default context.
+ license: apache-2.0
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - coding
+ - reasoning
+ last_checked: "2026-09-26"
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
+ temperature: 0.6
+ top_p: 0.95
+ top_k: 20
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
+ sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
+
+- name: sharp-spark-x2.5-4b-q6
+ url: github:mudler/LocalAI/gallery/virtual.yaml@master
+ urls:
+ - https://huggingface.co/XHToken/Spark-X2.5-4B
+ - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
+ description: |
+ Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
+ with an adjusted chat template for coding. This Q6_K_XL build uses the
+ embedded Sharp-Spark template and a 32K-token default context.
+ license: apache-2.0
+ tags:
+ - llm
+ - gguf
+ - cpu
+ - gpu
+ - coding
+ - reasoning
+ last_checked: "2026-09-26"
+ overrides:
+ backend: llama-cpp
+ context_size: 32768
+ known_usecases:
+ - chat
+ options:
+ - use_jinja:true
+ parameters:
+ model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
+ temperature: 0.6
+ top_p: 0.95
+ top_k: 20
+ template:
+ use_tokenizer_template: true
+ files:
+ - filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
+ uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
+ sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
+
- &spark-x2-5-4b
name: "spark-x2.5-4b-q4"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
@@ -45212,6 +45982,12 @@
sha256: ""
uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors
- name: kimodo-soma-rp
+ variants:
+ - model: kimodo-soma-rp-bf16
+ - model: kimodo-soma-rp-q4_k
+ - model: kimodo-soma-rp-q4_k_m
+ - model: kimodo-soma-rp-q5_k
+ - model: kimodo-soma-rp-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45399,6 +46175,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-soma-seed
+ variants:
+ - model: kimodo-soma-seed-bf16
+ - model: kimodo-soma-seed-q4_k
+ - model: kimodo-soma-seed-q4_k_m
+ - model: kimodo-soma-seed-q5_k
+ - model: kimodo-soma-seed-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45586,6 +46368,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-g1-rp
+ variants:
+ - model: kimodo-g1-rp-bf16
+ - model: kimodo-g1-rp-q4_k
+ - model: kimodo-g1-rp-q4_k_m
+ - model: kimodo-g1-rp-q5_k
+ - model: kimodo-g1-rp-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45773,6 +46561,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-g1-seed
+ variants:
+ - model: kimodo-g1-seed-bf16
+ - model: kimodo-g1-seed-q4_k
+ - model: kimodo-g1-seed-q4_k_m
+ - model: kimodo-g1-seed-q5_k
+ - model: kimodo-g1-seed-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
diff --git a/go.mod b/go.mod
index 404ae2347..8d434b5b5 100644
--- a/go.mod
+++ b/go.mod
@@ -83,6 +83,7 @@ require (
require (
filippo.io/bigmod v0.1.1-0.20260103110540-f8a47775ebe5 // indirect
+ filippo.io/edwards25519 v1.1.0 // indirect
filippo.io/keygen v0.0.0-20260114151900-8e2790ea4c5b // indirect
github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 // indirect
github.com/atotto/clipboard v0.1.4 // indirect
@@ -106,12 +107,10 @@ require (
github.com/cenkalti/backoff/v5 v5.0.3 // indirect
github.com/charmbracelet/bubbles v0.21.0 // indirect
github.com/charmbracelet/bubbletea v1.3.10 // indirect
- github.com/chasefleming/elem-go v0.30.0 // indirect
github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc // indirect
github.com/chromedp/chromedp v0.15.1 // indirect
github.com/chromedp/sysutil v1.1.0 // indirect
github.com/cyberphone/json-canonicalization v0.0.0-20241213102144-19d51d7fe467 // indirect
- github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 // indirect
github.com/digitorus/pkcs7 v0.0.0-20230818184609-3a137a874352 // indirect
github.com/digitorus/timestamp v0.0.0-20231217203849-220c5c2851b7 // indirect
github.com/dunglas/httpsfv v1.1.0 // indirect
@@ -140,21 +139,17 @@ require (
github.com/gobwas/httphead v0.1.0 // indirect
github.com/gobwas/pool v0.2.1 // indirect
github.com/gobwas/ws v1.4.0 // indirect
- github.com/gofiber/template v1.8.3 // indirect
- github.com/gofiber/template/html/v2 v2.1.3 // indirect
- github.com/gofiber/utils v1.1.0 // indirect
github.com/google/certificate-transparency-go v1.3.2 // indirect
github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0 // indirect
github.com/in-toto/attestation v1.1.2 // indirect
github.com/in-toto/in-toto-golang v0.9.0 // indirect
- github.com/inconshreveable/mousetrap v1.1.0 // indirect
github.com/invopop/jsonschema v0.13.0 // indirect
github.com/jinzhu/inflection v1.0.0 // indirect
github.com/jinzhu/now v1.1.5 // indirect
github.com/jolestar/go-commons-pool/v2 v2.1.2 // indirect
github.com/klippa-app/go-pdfium v1.19.2 // indirect
github.com/mattn/go-localereader v0.0.1 // indirect
- github.com/mattn/go-sqlite3 v1.14.28 // indirect
+ github.com/mattn/go-sqlite3 v1.14.32 // indirect
github.com/moby/moby/api v1.54.2 // indirect
github.com/moby/moby/client v0.4.1 // indirect
github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect
@@ -167,8 +162,6 @@ require (
github.com/sigstore/rekor-tiles/v2 v2.0.1 // indirect
github.com/sigstore/sigstore v1.10.0 // indirect
github.com/sigstore/timestamp-authority/v2 v2.0.3 // indirect
- github.com/spf13/cobra v1.10.2 // indirect
- github.com/spf13/pflag v1.0.10 // indirect
github.com/standard-webhooks/standard-webhooks/libraries v0.0.0-20260508151727-1282bb917829 // indirect
github.com/stretchr/testify v1.11.1 // indirect
github.com/sv-tools/openapi v0.2.1 // indirect
@@ -242,7 +235,7 @@ require (
github.com/kevinburke/ssh_config v1.2.0 // indirect
github.com/labstack/gommon v0.4.2 // indirect
github.com/mschoch/smat v0.2.0 // indirect
- github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1
+ github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca
github.com/mudler/localrecall v0.6.5 // indirect
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145
github.com/olekukonko/tablewriter v0.0.5 // indirect
@@ -250,7 +243,7 @@ require (
github.com/philippgille/chromem-go v0.7.0 // indirect
github.com/pion/transport/v4 v4.0.1 // indirect
github.com/pjbgf/sha1cd v0.6.0 // indirect
- github.com/rs/zerolog v1.31.0 // indirect
+ github.com/rs/zerolog v1.34.0 // indirect
github.com/saintfish/chardet v0.0.0-20230101081208-5e3ef4b5456d // indirect
github.com/segmentio/asm v1.1.3 // indirect
github.com/segmentio/encoding v0.5.4 // indirect
@@ -270,14 +263,13 @@ require (
github.com/valyala/fasttemplate v1.2.2 // indirect
github.com/xanzy/ssh-agent v0.3.3 // indirect
go.etcd.io/bbolt v1.4.3 // indirect
- go.mau.fi/util v0.3.0 // indirect
+ go.mau.fi/util v0.9.2 // indirect
go.starlark.net v0.0.0-20250417143717-f57e51f710eb // indirect
google.golang.org/appengine v1.6.8 // indirect
gopkg.in/warnings.v0 v0.1.2 // indirect
gopkg.in/yaml.v2 v2.4.0 // indirect
jaytaylor.com/html2text v0.0.0-20230321000545-74c2419ad056 // indirect
- maunium.net/go/maulogger/v2 v2.4.1 // indirect
- maunium.net/go/mautrix v0.17.0 // indirect
+ maunium.net/go/mautrix v0.25.2 // indirect
mvdan.cc/xurls/v2 v2.6.0 // indirect
)
diff --git a/go.sum b/go.sum
index 823bfa280..1eba26de7 100644
--- a/go.sum
+++ b/go.sum
@@ -277,8 +277,6 @@ github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf h1:rLG0Y
github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf/go.mod h1:B3UgsnsBZS/eX42BlaNiJkD1pPOUa+oF1IYC6Yd2CEU=
github.com/charmbracelet/x/term v0.2.1 h1:AQeHeLZ1OqSXhrAWpYUtZyX1T3zVxfpZuEQMIQaGIAQ=
github.com/charmbracelet/x/term v0.2.1/go.mod h1:oQ4enTYFV7QN4m0i9mzHrViD7TQKvNEEkHUMCmsxdUg=
-github.com/chasefleming/elem-go v0.30.0 h1:BlhV1ekv1RbFiM8XZUQeln1Ikb4D+bu2eDO4agREvok=
-github.com/chasefleming/elem-go v0.30.0/go.mod h1:hz73qILBIKnTgOujnSMtEj20/epI+f6vg71RUilJAA4=
github.com/chengxilo/virtualterm v1.0.4 h1:Z6IpERbRVlfB8WkOmtbHiDbBANU7cimRIof7mk9/PwM=
github.com/chengxilo/virtualterm v1.0.4/go.mod h1:DyxxBZz/x1iqJjFxTFcr6/x+jSpqN0iwWCOK1q10rlY=
github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc h1:wkN/LMi5vc60pBRWx6qpbk/aEvq3/ZVNpnMvsw8PVVU=
@@ -321,7 +319,6 @@ github.com/cpuguy83/dockercfg v0.3.2 h1:DlJTyZGBDlXqUZ2Dk2Q3xHs/FtnooJJVaad2S9GK
github.com/cpuguy83/dockercfg v0.3.2/go.mod h1:sugsbF4//dDlL/i+S+rtpIWp+5h0BHJHfjj5/jFyUJc=
github.com/cpuguy83/go-md2man/v2 v2.0.0-20190314233015-f79a8a8ca69d/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU=
github.com/cpuguy83/go-md2man/v2 v2.0.0/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU=
-github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
github.com/creachadair/mds v0.21.3 h1:RRgEAPIb52cU0q7UxGyN+13QlCVTZIL4slRr0cYYQfA=
github.com/creachadair/mds v0.21.3/go.mod h1:1ltMWZd9yXhaHEoZwBialMaviWVUpRPvMwVP7saFAzM=
github.com/creachadair/otp v0.5.0 h1:q3Th7CXm2zlmCdBjw5tEPFOj4oWJMnVL5HXlq0sNKS0=
@@ -334,8 +331,6 @@ github.com/cyphar/filepath-securejoin v0.6.1 h1:5CeZ1jPXEiYt3+Z6zqprSAgSWiggmpVy
github.com/cyphar/filepath-securejoin v0.6.1/go.mod h1:A8hd4EnAeyujCJRrICiOWqjS1AX0a9kM5XL+NwKoYSc=
github.com/danieljoos/wincred v1.2.2 h1:774zMFJrqaeYCK2W57BgAem/MLi6mtSE47MB6BOJ0i0=
github.com/danieljoos/wincred v1.2.2/go.mod h1:w7w4Utbrz8lqeMbDAK0lkNJUv5sAOkFi7nd/ogr0Uh8=
-github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 h1:flLYmnQFZNo04x2NPehMbf30m7Pli57xwZ0NFqR/hb0=
-github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2/go.mod h1:NtWqRzAp/1tw+twkW8uuBenEVVYndEAZACWU3F3xdoQ=
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
@@ -556,12 +551,6 @@ github.com/godbus/dbus/v5 v5.1.0 h1:4KLkAxT3aOY8Li4FRJe/KvhoNFFxo0m6fNuFUO8QJUk=
github.com/godbus/dbus/v5 v5.1.0/go.mod h1:xhWf0FNVPg57R7Z0UbKHbJfkEywrmjJnf7w5xrFpKfA=
github.com/gofiber/fiber/v2 v2.52.13 h1:TOKP64iqC9b5P49VrBW5tHhUOvDyrtJ0xePEfzJbCbk=
github.com/gofiber/fiber/v2 v2.52.13/go.mod h1:YEcBbO/FB+5M1IZNBP9FO3J9281zgPAreiI1oqg8nDw=
-github.com/gofiber/template v1.8.3 h1:hzHdvMwMo/T2kouz2pPCA0zGiLCeMnoGsQZBTSYgZxc=
-github.com/gofiber/template v1.8.3/go.mod h1:bs/2n0pSNPOkRa5VJ8zTIvedcI/lEYxzV3+YPXdBvq8=
-github.com/gofiber/template/html/v2 v2.1.3 h1:n1LYBtmr9C0V/k/3qBblXyMxV5B0o/gpb6dFLp8ea+o=
-github.com/gofiber/template/html/v2 v2.1.3/go.mod h1:U5Fxgc5KpyujU9OqKzy6Kn6Qup6Tm7zdsISR+VpnHRE=
-github.com/gofiber/utils v1.1.0 h1:vdEBpn7AzIUJRhe+CiTOJdUcTg4Q9RK+pEa0KPbLdrM=
-github.com/gofiber/utils v1.1.0/go.mod h1:poZpsnhBykfnY1Mc0KeEa6mSHrS3dV0+oBWyeQmb2e0=
github.com/gofrs/flock v0.13.0 h1:95JolYOvGMqeH31+FC7D2+uULf6mG61mEZ/A8dRYMzw=
github.com/gofrs/flock v0.13.0/go.mod h1:jxeyy9R1auM5S6JYDBhDt+E2TCo7DkratH4Pgi8P+Z0=
github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q=
@@ -928,8 +917,8 @@ github.com/mattn/go-runewidth v0.0.9/go.mod h1:H031xJmbD/WCDINGzjvQ9THkh0rPKHF+m
github.com/mattn/go-runewidth v0.0.12/go.mod h1:RAqKPSqVFrSLVXbA8x7dzmKdmGzieGRCM46jaSJTDAk=
github.com/mattn/go-runewidth v0.0.17 h1:78v8ZlW0bP43XfmAfPsdXcoNCelfMHsDmd/pkENfrjQ=
github.com/mattn/go-runewidth v0.0.17/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
-github.com/mattn/go-sqlite3 v1.14.28 h1:ThEiQrnbtumT+QMknw63Befp/ce/nUPgBPMlRFEum7A=
-github.com/mattn/go-sqlite3 v1.14.28/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y=
+github.com/mattn/go-sqlite3 v1.14.32 h1:JD12Ag3oLy1zQA+BNn74xRgaBbdhbNIDYvQUEuuErjs=
+github.com/mattn/go-sqlite3 v1.14.32/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y=
github.com/mdelapenya/tlscert v0.2.0 h1:7H81W6Z/4weDvZBNOfQte5GpIMo0lGYEeWbkGp5LJHI=
github.com/mdelapenya/tlscert v0.2.0/go.mod h1:O4njj3ELLnJjGdkN7M/vIVCpZ+Cf0L6muqOG4tLSl8o=
github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3vE=
@@ -1005,10 +994,8 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI=
github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8=
github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM=
github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw=
-github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxFoYsAObvAuOqXBakRPMD0PWxWG5EE=
-github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c=
-github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc=
-github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o=
+github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca h1:bHlzSuOc5cKvHGF21ZjFFUV7IlkS3wt9YkBsWtkcaC4=
+github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4=
github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU=
@@ -1017,8 +1004,6 @@ github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc h1:RxwneJl1VgvikiX
github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc/go.mod h1:O7SwdSWMilAWhBZMK9N9Y/oBDyMMzshE3ju8Xkexwig=
github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6 h1:/nFm1Ttf8g1BnWtEth986JR34pCh9rzae5A2vKBZosc=
github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6/go.mod h1:h6kmHUZeafr+k5hRYpGLMzJFH4hItHffgpRo2QIkP+o=
-github.com/mudler/localrecall v0.6.3 h1:uXOrP9JmetzxgVKzSrawviyBHZfAcvPBBIrvVUdZjDA=
-github.com/mudler/localrecall v0.6.3/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU=
github.com/mudler/localrecall v0.6.5 h1:Q0atTJFFAyumKZG5dbGSrvQ+wsuA88hywIOfHdxpBEU=
github.com/mudler/localrecall v0.6.5/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU=
github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 h1:Ry8RiWy8fZ6Ff4E7dPmjRsBrnHOnPeOOj2LhCgyjQu0=
@@ -1202,13 +1187,12 @@ github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6L
github.com/rogpeppe/go-internal v1.3.0/go.mod h1:M8bDsm7K2OlrFYOpmOWEs/qY81heoFRclV5y23lUDJ4=
github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ=
github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc=
-github.com/rs/xid v1.5.0/go.mod h1:trrq9SKmegXys3aeAKXMUTdJsYXVwGY3RLcfgqegfbg=
-github.com/rs/zerolog v1.31.0 h1:FcTR3NnLWW+NnTwwhFWiJSZr4ECLpqCm6QsEnyvbV4A=
-github.com/rs/zerolog v1.31.0/go.mod h1:/7mN4D5sKwJLZQ2b/znpjC3/GQWY/xaDXUM0kKWRHss=
+github.com/rs/xid v1.6.0/go.mod h1:7XoLgs4eV+QndskICGsho+ADou8ySMSjJKDIan90Nz0=
+github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY=
+github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ=
github.com/russross/blackfriday v1.6.0 h1:KqfZb0pUVN2lYqZUYRddxF4OR8ZMURnJIG5Y3VRLtww=
github.com/russross/blackfriday v1.6.0/go.mod h1:ti0ldHuxg49ri4ksnFxlkCfN+hvslNlmVHqNRXXJNAY=
github.com/russross/blackfriday/v2 v2.0.1/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
-github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
github.com/ruudk/golang-pdf417 v0.0.0-20181029194003-1af4ab5afa58/go.mod h1:6lfFZQK844Gfx8o5WFuvpxWRwnSoipWe/p622j1v06w=
github.com/ryanuber/columnize v0.0.0-20160712163229-9b3edd62028f/go.mod h1:sm1tb6uqfes/u+d4ooFouqFdy9/2g9QGwK3SQygK0Ts=
github.com/ryanuber/go-glob v1.0.0 h1:iQh3xXAumdQ+4Ufa5b25cRpC5TYKlno6hsv6Cb3pkBk=
@@ -1301,7 +1285,6 @@ github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4=
github.com/spf13/jwalterweatherman v1.1.0/go.mod h1:aNWZUN0dPAAO/Ljvb5BEdw96iTZ0EXowPYD95IqWIGo=
github.com/spf13/pflag v1.0.5/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
-github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk=
github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
github.com/spf13/viper v1.8.1/go.mod h1:o0Pch8wJ9BVSWGQMbra6iw0oQ5oktSIBaujf1rJH9Ns=
@@ -1449,8 +1432,8 @@ go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
go.etcd.io/etcd/api/v3 v3.5.0/go.mod h1:cbVKeC6lCfl7j/8jBhAK6aIYO9XOjdptoxU/nLQcPvs=
go.etcd.io/etcd/client/pkg/v3 v3.5.0/go.mod h1:IJHfcCEKxYu1Os13ZdwCwIUTUVGYTSAM3YSwc9/Ac1g=
go.etcd.io/etcd/client/v2 v2.305.0/go.mod h1:h9puh54ZTgAKtEbut2oe9P4L/oqKCVB6xsXlzd7alYQ=
-go.mau.fi/util v0.3.0 h1:Lt3lbRXP6ZBqTINK0EieRWor3zEwwwrDT14Z5N8RUCs=
-go.mau.fi/util v0.3.0/go.mod h1:9dGsBCCbZJstx16YgnVMVi3O2bOizELoKpugLD4FoGs=
+go.mau.fi/util v0.9.2 h1:+S4Z03iCsGqU2WY8X2gySFsFjaLlUHFRDVCYvVwynKM=
+go.mau.fi/util v0.9.2/go.mod h1:055elBBCJSdhRsmub7ci9hXZPgGr1U6dYg44cSgRgoU=
go.mongodb.org/mongo-driver v1.17.6 h1:87JUG1wZfWsr6rIz3ZmpH90rL5tea7O3IHuSwHUpsss=
go.mongodb.org/mongo-driver v1.17.6/go.mod h1:Hy04i7O2kC4RS06ZrhPRqj/u4DTYkFDAAccj+rVKqgQ=
go.opencensus.io v0.21.0/go.mod h1:mSImk1erAIZhrmZN+AvHh14ztQfjbGwt4TtuofqLduU=
@@ -2005,10 +1988,8 @@ k8s.io/klog/v2 v2.130.1 h1:n9Xl7H1Xvksem4KFG4PYbdQCQxqc/tTUyrgXaOhHSzk=
k8s.io/klog/v2 v2.130.1/go.mod h1:3Jpz1GvMt720eyJH1ckRHK1EDfpxISzJ7I9OYgaDtPE=
lukechampine.com/blake3 v1.4.1 h1:I3Smz7gso8w4/TunLKec6K2fn+kyKtDxr/xcQEN84Wg=
lukechampine.com/blake3 v1.4.1/go.mod h1:QFosUxmjB8mnrWFSNwKmvxHpfY72bmD2tQ0kBMM3kwo=
-maunium.net/go/maulogger/v2 v2.4.1 h1:N7zSdd0mZkB2m2JtFUsiGTQQAdP0YeFWT7YMc80yAL8=
-maunium.net/go/maulogger/v2 v2.4.1/go.mod h1:omPuYwYBILeVQobz8uO3XC8DIRuEb5rXYlQSuqrbCho=
-maunium.net/go/mautrix v0.17.0 h1:scc1qlUbzPn+wc+3eAPquyD+3gZwwy/hBANBm+iGKK8=
-maunium.net/go/mautrix v0.17.0/go.mod h1:j+puTEQCEydlVxhJ/dQP5chfa26TdvBO7X6F3Ataav8=
+maunium.net/go/mautrix v0.25.2 h1:CUG23zp754yGOTMh9Q4mVSENS9FyweE/G+6ZsPDMCUU=
+maunium.net/go/mautrix v0.25.2/go.mod h1:EWgYyp2iFZP7pnSm+rufHlO8YVnA2KnoNBDpwekiAwI=
mvdan.cc/xurls/v2 v2.6.0 h1:3NTZpeTxYVWNSokW3MKeyVkz/j7uYXYiMtXRUfmjbgI=
mvdan.cc/xurls/v2 v2.6.0/go.mod h1:bCvEZ1XvdA6wDnxY7jPPjEmigDtvtvPXAD/Exa9IMSk=
oras.land/oras-go/v2 v2.6.0 h1:X4ELRsiGkrbeox69+9tzTu492FMUu7zJQW6eJU+I2oc=
diff --git a/pkg/functions/functions.go b/pkg/functions/functions.go
index 0686e4f9d..3e604d9df 100644
--- a/pkg/functions/functions.go
+++ b/pkg/functions/functions.go
@@ -89,6 +89,11 @@ func (f Functions) ToJSONStructure(name, args string) JSONFunctionStructure {
return js
}
+// ToJSONStructure converts functions using the configured property keys.
+func (c FunctionsConfig) ToJSONStructure(functions Functions) JSONFunctionStructure {
+ return functions.ToJSONStructure(c.FunctionNameKey, c.FunctionArgumentsKey)
+}
+
// Select returns a list of functions containing the function with the given name
func (f Functions) Select(name string) Functions {
var funcs Functions
diff --git a/pkg/functions/functions_test.go b/pkg/functions/functions_test.go
index e0952c13f..7c10da473 100644
--- a/pkg/functions/functions_test.go
+++ b/pkg/functions/functions_test.go
@@ -65,6 +65,33 @@ var _ = Describe("LocalAI grammar functions", func() {
Expect(fnName.Const).To(Equal("search"))
Expect(fnArgs.Properties["query"].(map[string]any)["type"]).To(Equal("string"))
})
+
+ It("keeps the name and the arguments in separate properties when both keys are customized", func() {
+ var functions Functions = []Function{
+ {
+ Name: "get_weather",
+ Parameters: map[string]any{
+ "properties": map[string]any{
+ "city": map[string]any{
+ "type": "string",
+ },
+ },
+ },
+ },
+ }
+
+ config := FunctionsConfig{
+ FunctionNameKey: "function",
+ FunctionArgumentsKey: "parameters",
+ }
+ js := config.ToJSONStructure(functions)
+ Expect(js.OneOf[0].Properties).To(HaveLen(2))
+
+ fnName := js.OneOf[0].Properties["function"].(FunctionName)
+ fnArgs := js.OneOf[0].Properties["parameters"].(Argument)
+ Expect(fnName.Const).To(Equal("get_weather"))
+ Expect(fnArgs.Properties["city"].(map[string]any)["type"]).To(Equal("string"))
+ })
})
Context("Select()", func() {
It("selects one of the functions and returns a list containing only the selected one", func() {
diff --git a/pkg/modelartifacts/materializer.go b/pkg/modelartifacts/materializer.go
index a3e2a9e8e..87036f756 100644
--- a/pkg/modelartifacts/materializer.go
+++ b/pkg/modelartifacts/materializer.go
@@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
skippedFiles := 0
skippedBytes := int64(0)
tasks := make([]downloader.FileTask, 0, len(snapshot.Files))
+ // Sibling manifests are read once, before the staging loop, so the
+ // per-file reuse lookups below never re-read or re-parse them.
+ siblings := loadSiblingCandidates(modelsPath, spec, layout)
for index, file := range snapshot.Files {
if err := ctx.Err(); err != nil {
return Result{}, err
@@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
skippedBytes += file.Size
continue
}
+ // Before reaching for the network, consult committed sibling trees for the
+ // same Source (type+endpoint+repo+revision). A narrower allow_patterns
+ // request gets a different CacheKey, so committedResult misses even though a
+ // broader sibling already holds this exact file; reusing it avoids a
+ // redundant re-download of tens of gigabytes. The match is re-verified
+ // through verifyDownloadedFile (full SHA-256), never size-only, and a broader
+ // request can never inherit a narrower sibling's gaps because each file is
+ // matched individually against the sibling's manifest.
+ if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok {
+ manifest.Files[taskIndex] = entry
+ completedBytes.Add(file.Size)
+ skippedFiles++
+ skippedBytes += file.Size
+ continue
+ }
nameSum := sha256.Sum256([]byte(file.Path))
blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:]))
blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel))
@@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest
return entry, true
}
+// siblingCandidate is one committed sibling artifact tree that shares this
+// request's Source (type+endpoint+repo+revision), with its manifest files
+// indexed by path.
+type siblingCandidate struct {
+ final string
+ filesByPath map[string][]ManifestFile
+}
+
+// loadSiblingCandidates reads the committed sibling manifest set once, before
+// the staging loop. Doing it per file instead would re-read and re-parse every
+// sibling manifest for every file — 20 committed siblings and a 300-file
+// snapshot means 6000 manifest reads before the first byte is fetched.
+//
+// The current artifact's own committed tree is excluded: it is either absent
+// (the reason materializeLocked is running) or already handled by
+// committedResult's exact-key fast path.
+func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate {
+ if spec.Resolved == nil || layout.Final == "" {
+ return nil
+ }
+ siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface")
+ entries, err := os.ReadDir(siblingsRoot)
+ if err != nil {
+ return nil
+ }
+ var candidates []siblingCandidate
+ for _, entry := range entries {
+ if !entry.IsDir() {
+ continue
+ }
+ siblingFinal := filepath.Join(siblingsRoot, entry.Name())
+ if siblingFinal == layout.Final {
+ continue
+ }
+ siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json"))
+ if err != nil {
+ continue
+ }
+ siblingArtifact := siblingManifest.Artifact
+ if siblingArtifact.Resolved == nil ||
+ siblingArtifact.Source.Type != spec.Source.Type ||
+ siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint ||
+ siblingArtifact.Source.Repo != spec.Source.Repo ||
+ siblingArtifact.Resolved.Revision != spec.Resolved.Revision {
+ continue
+ }
+ byPath := make(map[string][]ManifestFile, len(siblingManifest.Files))
+ for _, f := range siblingManifest.Files {
+ byPath[f.Path] = append(byPath[f.Path], f)
+ }
+ candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath})
+ }
+ return candidates
+}
+
+// reuseFromCommittedSibling looks for a file already committed under a sibling
+// artifact tree — same Source (type+endpoint+repo+revision), different
+// allow/ignore patterns — and stages it for this writer instead of fetching.
+// A narrower allow_patterns request gets a different CacheKey (path.go:62), so
+// committedResult misses and materializeLocked would otherwise re-download
+// files an already-committed broader sibling already holds.
+//
+// The match is never size-only: the sibling file is re-hashed through the
+// shared verifyDownloadedFile against the current request's SnapshotFile (its
+// LFS or git blob OID), so the staged entry is byte-for-byte identical to a
+// fresh download. A broader request can never stand in for files a narrower
+// sibling lacks, because each requested file is matched individually against
+// the sibling's manifest file set. Hard-link keeps the shared models volume
+// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the
+// kernel cannot hard-link.
+func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) {
+ snapshotRel := path.Join("snapshot", file.Path)
+ snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel))
+ for _, sibling := range candidates {
+ for _, siblingFile := range sibling.filesByPath[file.Path] {
+ if siblingFile.Size != file.Size {
+ continue
+ }
+ siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path))
+ verified, err := verifyDownloadedFile(siblingPath, file)
+ if err != nil {
+ continue
+ }
+ if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil {
+ return ManifestFile{}, false
+ }
+ _ = root.Remove(snapshotRel)
+ if err := linkOrCopy(siblingPath, snapshotAbs); err != nil {
+ return ManifestFile{}, false
+ }
+ return verified, true
+ }
+ }
+ return ManifestFile{}, false
+}
+
+// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only
+// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking
+// keeps the shared models volume neutral — a narrowed request does not double
+// the storage of a broad sibling's files.
+func linkOrCopy(src, dst string) error {
+ if err := os.Link(src, dst); err == nil {
+ return nil
+ } else if !errors.Is(err, syscall.EXDEV) {
+ return err
+ }
+ in, err := os.Open(src)
+ if err != nil {
+ return err
+ }
+ defer func() { _ = in.Close() }()
+ out, err := os.Create(dst)
+ if err != nil {
+ return err
+ }
+ if _, err := io.Copy(out, in); err != nil {
+ _ = out.Close()
+ _ = os.Remove(dst)
+ return err
+ }
+ return out.Close()
+}
+
func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) {
file, err := os.Open(fileName)
if err != nil {
diff --git a/pkg/modelartifacts/materializer_sibling_reuse_test.go b/pkg/modelartifacts/materializer_sibling_reuse_test.go
new file mode 100644
index 000000000..572b255d8
--- /dev/null
+++ b/pkg/modelartifacts/materializer_sibling_reuse_test.go
@@ -0,0 +1,230 @@
+package modelartifacts_test
+
+import (
+ "context"
+ "crypto/sha256"
+ "encoding/hex"
+ "net/http"
+ "net/http/httptest"
+ "os"
+ "path/filepath"
+ "strconv"
+ "strings"
+ "sync"
+
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+
+ hfapi "github.com/mudler/LocalAI/pkg/huggingface-api"
+ "github.com/mudler/LocalAI/pkg/modelartifacts"
+)
+
+const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567"
+
+// recordingResolver serves a fixed full file set filtered by each request's
+// allow/ignore patterns, so a narrower request genuinely resolves to a strict
+// subset of a broader sibling's files. The HTTP server behind it records every
+// fetch, which is the signal the sibling-reuse fix is verified through. The
+// function under fix is never mocked: a real Manager drives the real staging +
+// commit path against this stub collaborator.
+type recordingResolver struct {
+ endpoint string
+ repo string
+ files []hfapi.SnapshotFile
+ server *httptest.Server
+
+ mu sync.Mutex
+ fetched map[string]int
+}
+
+func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver {
+ r := &recordingResolver{
+ endpoint: "https://huggingface.co",
+ repo: "owner/repo",
+ files: files,
+ fetched: map[string]int{},
+ }
+ r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) {
+ name := strings.TrimPrefix(req.URL.Path, "/file/")
+ body, ok := contents[name]
+ if !ok {
+ w.WriteHeader(http.StatusNotFound)
+ return
+ }
+ r.mu.Lock()
+ r.fetched[name]++
+ r.mu.Unlock()
+ w.Header().Set("Content-Length", strconv.Itoa(len(body)))
+ _, _ = w.Write(body)
+ }))
+ return r
+}
+
+func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) {
+ files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns)
+ if err != nil {
+ return hfapi.Snapshot{}, err
+ }
+ out := make([]hfapi.SnapshotFile, len(files))
+ for i, f := range files {
+ f.URL = r.server.URL + "/file/" + f.Path
+ out[i] = f
+ }
+ return hfapi.Snapshot{
+ Endpoint: r.endpoint, Repo: r.repo,
+ RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out,
+ }, nil
+}
+
+func (r *recordingResolver) fetchCount(path string) int {
+ r.mu.Lock()
+ defer r.mu.Unlock()
+ return r.fetched[path]
+}
+
+func (r *recordingResolver) resetFetches() {
+ r.mu.Lock()
+ defer r.mu.Unlock()
+ r.fetched = map[string]int{}
+}
+
+func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile {
+ paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"}
+ files := make([]hfapi.SnapshotFile, 0, len(paths))
+ for _, p := range paths {
+ sum := sha256.Sum256(contents[p])
+ files = append(files, hfapi.SnapshotFile{
+ Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]),
+ })
+ }
+ return files
+}
+
+// The narrow-request case proves the fix for #11047:
+// a request with narrower allow_patterns (a strict subset) reuses files an
+// already-committed broader sibling holds, hard-linking instead of re-fetching.
+//
+// On master this is RED: a narrower allow_patterns set hashes to a different
+// CacheKey (path.go:62), so committedResult misses and materializeLocked
+// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On
+// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling,
+// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0,
+// os.SameFile true).
+var _ = Describe("committed sibling reuse", func() {
+ It("reuses files from a broader committed sibling", func() {
+ contents := map[string][]byte{
+ "a/first.bin": []byte("first-file-bytes"),
+ "b/second.bin": []byte("second-file-bytes-longer"),
+ "c/third.bin": []byte("third-file"),
+ }
+ resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
+ defer resolver.server.Close()
+
+ modelsPath := GinkgoT().TempDir()
+ manager := modelartifacts.NewManager(resolver,
+ modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
+
+ // Commit the broad sibling: all three files, fetched from the resolver.
+ broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
+ Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
+ }}
+ broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(broad.CacheHit).To(BeFalse())
+ Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0),
+ "the broad sibling must have fetched a/first.bin to commit it")
+
+ resolver.resetFetches()
+
+ // Narrowed request: a strict subset of the broad sibling's file set.
+ narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
+ Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
+ AllowPatterns: []string{"a/first.bin"},
+ }}
+ narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(narrow.CacheHit).To(BeFalse())
+
+ // (b) The sibling-present file must NOT be re-fetched: zero fetches. This is
+ // the assertion that is RED on master (one fetch) and GREEN on the branch.
+ Expect(resolver.fetchCount("a/first.bin")).To(Equal(0),
+ "a/first.bin must be reused from the committed broad sibling, not re-fetched")
+
+ // (a) The narrowed tree's staged file is the same inode as the broad
+ // sibling's file (hard-link), not a freshly downloaded second copy. RED on
+ // master (separate file), GREEN on the branch (hard-link).
+ broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin")
+ narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin")
+ broadInfo, err := os.Stat(broadFile)
+ Expect(err).NotTo(HaveOccurred())
+ narrowInfo, err := os.Stat(narrowFile)
+ Expect(err).NotTo(HaveOccurred())
+ Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(),
+ "the narrowed request must hard-link the broad sibling's file rather than store a second copy")
+
+ // The reused bytes are intact end to end.
+ Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"]))
+ })
+
+ // The broader-request case is the manifest file-set guard: a broader request
+ // against a narrower committed sibling must still fetch the files the sibling
+ // lacks and commit a complete tree. Sibling-reuse can never serve an incomplete
+ // model as complete, because each requested file is matched individually against
+ // the sibling's manifest.
+ It("fetches files missing from a narrower committed sibling", func() {
+ contents := map[string][]byte{
+ "a/first.bin": []byte("first-file-bytes"),
+ "b/second.bin": []byte("second-file-bytes-longer"),
+ "c/third.bin": []byte("third-file"),
+ }
+ resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
+ defer resolver.server.Close()
+
+ modelsPath := GinkgoT().TempDir()
+ manager := modelartifacts.NewManager(resolver,
+ modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
+
+ // Commit a NARROW sibling first: only a/first.bin and b/second.bin.
+ narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
+ Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
+ AllowPatterns: []string{"a/first.bin", "b/second.bin"},
+ }}
+ narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
+ Expect(err).NotTo(HaveOccurred())
+ narrowPaths := make([]string, 0, len(narrow.Manifest.Files))
+ for _, f := range narrow.Manifest.Files {
+ narrowPaths = append(narrowPaths, f.Path)
+ }
+ Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"}))
+
+ resolver.resetFetches()
+
+ // A BROADER request asks for all three files, including c/third.bin which the
+ // narrow sibling does not hold.
+ broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
+ Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
+ }}
+ broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
+ Expect(err).NotTo(HaveOccurred())
+
+ // The file the narrow sibling lacks MUST be fetched: sibling-reuse must not
+ // inherit a narrower tree's gaps as if the broad request were complete.
+ Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0),
+ "c/third.bin is absent from the narrow sibling and must be fetched, not served as complete")
+
+ // The broad tree's manifest file set is exactly the full set — never the
+ // narrow sibling's subset. This file-set comparison proves no incomplete model
+ // is ever served as complete via sibling-reuse.
+ broadPaths := make([]string, 0, len(broad.Manifest.Files))
+ for _, f := range broad.Manifest.Files {
+ broadPaths = append(broadPaths, f.Path)
+ }
+ Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"}))
+
+ // Every file is present on disk with the right bytes after commit.
+ for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} {
+ Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))).
+ To(Equal(contents[p]))
+ }
+ })
+})
diff --git a/pkg/xsysinfo/process_vram_linux.go b/pkg/xsysinfo/process_vram_linux.go
new file mode 100644
index 000000000..3dd0a59a9
--- /dev/null
+++ b/pkg/xsysinfo/process_vram_linux.go
@@ -0,0 +1,165 @@
+//go:build linux
+
+// SPDX-License-Identifier: MIT
+package xsysinfo
+
+import (
+ "bufio"
+ "bytes"
+ "math"
+ "os"
+ "path/filepath"
+ "strconv"
+ "strings"
+)
+
+// ProcessVRAM reports device-local resident bytes accounted to a process tree
+// by DRM. Unsupported or incomplete accounting returns false, not a measured zero.
+func ProcessVRAM(pid int) (uint64, bool) {
+ return processVRAM("/proc", pid)
+}
+
+func processVRAM(procRoot string, pid int) (uint64, bool) {
+ if pid <= 0 {
+ return 0, false
+ }
+ clients := map[string]uint64{}
+ seen := map[int]bool{}
+ pending := []int{pid}
+ for len(pending) > 0 {
+ current := pending[len(pending)-1]
+ pending = pending[:len(pending)-1]
+ if seen[current] {
+ continue
+ }
+ seen[current] = true
+ base := filepath.Join(procRoot, strconv.Itoa(current))
+ fds, err := os.ReadDir(filepath.Join(base, "fd"))
+ if err != nil {
+ return 0, false
+ }
+ for _, fd := range fds {
+ target, err := os.Readlink(filepath.Join(base, "fd", fd.Name()))
+ if err != nil {
+ return 0, false
+ }
+ // A mixed DRM/NVIDIA tree cannot provide a complete DRM reading.
+ if strings.HasPrefix(target, "/dev/nvidia") {
+ return 0, false
+ }
+ if !strings.HasPrefix(target, "/dev/dri/render") {
+ // Primary nodes can also own allocations. Until their device
+ // identity is resolved, omitting them would undercount the tree.
+ if strings.HasPrefix(target, "/dev/dri/") {
+ return 0, false
+ }
+ continue
+ }
+ // #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
+ // base adds an integer PID, and fd.Name comes from os.ReadDir.
+ // The kernel supplies these path components, not request input.
+ data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name()))
+ if err != nil {
+ return 0, false
+ }
+ client, used, ok := drmResidentClient(data)
+ if !ok {
+ return 0, false
+ }
+ key := target + ":" + client
+ // dup() and fork() can expose the same client more than once. The
+ // snapshot is not atomic; retain its largest observed reading.
+ clients[key] = max(clients[key], used)
+ }
+
+ // A worker may be spawned by any thread, not just the thread leader.
+ tasks, err := os.ReadDir(filepath.Join(base, "task"))
+ if err != nil || len(tasks) == 0 {
+ return 0, false
+ }
+ for _, task := range tasks {
+ // #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
+ // base adds an integer PID, and task.Name comes from os.ReadDir.
+ // The kernel supplies these path components, not request input.
+ data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children"))
+ if err != nil {
+ return 0, false
+ }
+ for _, raw := range strings.Fields(string(data)) {
+ child, err := strconv.Atoi(raw)
+ if err != nil || child <= 0 {
+ return 0, false
+ }
+ pending = append(pending, child)
+ }
+ }
+ }
+ var total uint64
+ for _, used := range clients {
+ if used > math.MaxUint64-total {
+ return 0, false
+ }
+ total += used
+ }
+ return total, len(clients) > 0
+}
+
+func drmResidentClient(data []byte) (string, uint64, bool) {
+ var client string
+ var total uint64
+ found := false
+ scanner := bufio.NewScanner(bytes.NewReader(data))
+ for scanner.Scan() {
+ key, value, ok := strings.Cut(scanner.Text(), ":")
+ if !ok {
+ continue
+ }
+ if key == "drm-client-id" {
+ id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64)
+ if err != nil {
+ return "", 0, false
+ }
+ client = strconv.FormatUint(id, 10)
+ }
+ region, resident := strings.CutPrefix(key, "drm-resident-")
+ if !resident || !isVRAMRegion(region) {
+ continue
+ }
+ used, ok := drmResidentBytes(value)
+ if !ok || used > math.MaxUint64-total {
+ return "", 0, false
+ }
+ total += used
+ found = true
+ }
+ return client, total, scanner.Err() == nil && client != "" && found
+}
+
+func drmResidentBytes(value string) (uint64, bool) {
+ fields := strings.Fields(value)
+ if len(fields) == 0 || len(fields) > 2 {
+ return 0, false
+ }
+ n, err := strconv.ParseUint(fields[0], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+ unit := uint64(1)
+ if len(fields) == 2 {
+ switch strings.ToLower(fields[1]) {
+ case "b":
+ case "kib":
+ unit = 1 << 10
+ case "mib":
+ unit = 1 << 20
+ case "gib":
+ unit = 1 << 30
+ default:
+ return 0, false
+ }
+ }
+ if n > math.MaxUint64/unit {
+ return 0, false
+ }
+ return n * unit, true
+}
diff --git a/pkg/xsysinfo/process_vram_linux_test.go b/pkg/xsysinfo/process_vram_linux_test.go
new file mode 100644
index 000000000..4de1f6cdd
--- /dev/null
+++ b/pkg/xsysinfo/process_vram_linux_test.go
@@ -0,0 +1,105 @@
+//go:build linux
+
+// SPDX-License-Identifier: MIT
+package xsysinfo
+
+import (
+ "os"
+ "path/filepath"
+ "strconv"
+
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("ProcessVRAM", func() {
+ var root string
+ write := func(path, contents string) {
+ Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed())
+ Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed())
+ }
+ addProcess := func(pid int, children string) {
+ base := filepath.Join(root, strconv.Itoa(pid))
+ Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed())
+ write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children)
+ }
+ addFD := func(pid, fd int, render, info string) {
+ base := filepath.Join(root, strconv.Itoa(pid))
+ name := strconv.Itoa(fd)
+ Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed())
+ write(filepath.Join(base, "fdinfo", name), info)
+ }
+ BeforeEach(func() {
+ var err error
+ root, err = os.MkdirTemp("", "process-vram-")
+ Expect(err).NotTo(HaveOccurred())
+ DeferCleanup(os.RemoveAll, root)
+ addProcess(100, "")
+ })
+
+ It("sums resident device memory across GPUs and child processes without duplicate clients", func() {
+ write(filepath.Join(root, "100/task/101/children"), "200")
+ addProcess(200, "")
+ info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n"
+ addFD(100, 3, "renderD128", info)
+ addFD(100, 4, "renderD128", info)
+ addFD(200, 3, "renderD128", info)
+ addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n")
+ used, ok := processVRAM(root, 100)
+ Expect(ok).To(BeTrue())
+ Expect(used).To(Equal(uint64(384 * 1024 * 1024)))
+ })
+
+ It("distinguishes a measured zero from unavailable accounting", func() {
+ addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n")
+ used, ok := processVRAM(root, 100)
+ Expect(ok).To(BeTrue())
+ Expect(used).To(BeZero())
+ })
+
+ DescribeTable("does not invent readings from unsupported or invalid accounting",
+ func(info string) {
+ addFD(100, 3, "renderD128", info)
+ _, ok := processVRAM(root, 100)
+ Expect(ok).To(BeFalse())
+ },
+ Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"),
+ Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"),
+ Entry("no client identity", "drm-resident-vram0: 128 MiB\n"),
+ Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"),
+ Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"),
+ Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"),
+ )
+
+ It("omits a partial reading if a child cannot be inspected", func() {
+ addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
+ write(filepath.Join(root, "100/task/100/children"), "200")
+ _, ok := processVRAM(root, 100)
+ Expect(ok).To(BeFalse())
+ })
+
+ It("omits a partial reading if another DRM client lacks accounting", func() {
+ addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
+ addFD(100, 4, "renderD129", "drm-client-id: 8\n")
+ _, ok := processVRAM(root, 100)
+ Expect(ok).To(BeFalse())
+ })
+
+ DescribeTable("omits mixed readings with unsupported GPU descriptors",
+ func(target string) {
+ addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
+ Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed())
+ _, ok := processVRAM(root, 100)
+ Expect(ok).To(BeFalse())
+ },
+ Entry("primary DRM node", "/dev/dri/card0"),
+ Entry("NVIDIA device", "/dev/nvidia0"),
+ )
+
+ It("returns unavailable for missing processes or no DRM descriptors", func() {
+ for _, pid := range []int{-1, 0, 100, 999} {
+ _, ok := processVRAM(root, pid)
+ Expect(ok).To(BeFalse())
+ }
+ })
+})
diff --git a/pkg/xsysinfo/process_vram_other.go b/pkg/xsysinfo/process_vram_other.go
new file mode 100644
index 000000000..06cc49186
--- /dev/null
+++ b/pkg/xsysinfo/process_vram_other.go
@@ -0,0 +1,9 @@
+//go:build !linux
+
+// SPDX-License-Identifier: MIT
+package xsysinfo
+
+// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting.
+func ProcessVRAM(pid int) (uint64, bool) {
+ return 0, false
+}
diff --git a/scripts/build/gallery/main.go b/scripts/build/gallery/main.go
new file mode 100644
index 000000000..5e749eaaa
--- /dev/null
+++ b/scripts/build/gallery/main.go
@@ -0,0 +1,94 @@
+// SPDX-License-Identifier: MIT
+// Package the official index and its repository-local base configurations.
+package main
+
+import (
+ "fmt"
+ "os"
+ "path/filepath"
+ "strings"
+
+ "gopkg.in/yaml.v3"
+)
+
+func main() {
+ if len(os.Args) != 4 {
+ fmt.Fprintln(os.Stderr, "usage: gallery REPOSITORY {gallery|backend} OUTPUT")
+ os.Exit(1)
+ }
+ if err := packageGallery(os.Args[1], os.Args[2], os.Args[3]); err != nil {
+ fmt.Fprintln(os.Stderr, err)
+ os.Exit(1)
+ }
+}
+
+func packageGallery(root, source, output string) error {
+ if source != "gallery" && source != "backend" {
+ return fmt.Errorf("unsupported gallery directory %q", source)
+ }
+ repository, err := os.OpenRoot(root)
+ if err != nil {
+ return err
+ }
+ defer func() { _ = repository.Close() }()
+ body, err := repository.ReadFile(filepath.Join(source, "index.yaml"))
+ if err != nil {
+ return err
+ }
+ var doc yaml.Node
+ if err := yaml.Unmarshal(body, &doc); err != nil {
+ return err
+ }
+ // The build operator explicitly selects the output directory via the CLI.
+ if err := os.MkdirAll(output, 0700); err != nil { // #nosec G703 -- caller-selected output root
+ return err
+ }
+ destination, err := os.OpenRoot(output)
+ if err != nil {
+ return err
+ }
+ defer func() { _ = destination.Close() }()
+ // Keep the tree relative to the repository root so repeated base configs
+ // share a layer, even when an index refers outside its own directory.
+ const prefix = "github:mudler/LocalAI/"
+ var walk func(*yaml.Node) error
+ walk = func(n *yaml.Node) error {
+ if n.Kind == yaml.MappingNode {
+ for i := 0; i < len(n.Content); i += 2 {
+ value := n.Content[i+1]
+ if n.Content[i].Value != "url" || value.Kind != yaml.ScalarNode || !strings.HasPrefix(value.Value, prefix) || !strings.HasSuffix(value.Value, "@master") {
+ continue
+ }
+ path := strings.TrimSuffix(strings.TrimPrefix(value.Value, prefix), "@master")
+ if !filepath.IsLocal(path) {
+ return fmt.Errorf("base config escapes repository: %q", path)
+ }
+ config, err := repository.ReadFile(path)
+ if err != nil {
+ return err
+ }
+ if err := destination.MkdirAll(filepath.Dir(path), 0700); err != nil {
+ return err
+ }
+ if err := destination.WriteFile(path, config, 0600); err != nil {
+ return err
+ }
+ value.Value = filepath.ToSlash(path)
+ }
+ }
+ for _, child := range n.Content {
+ if err := walk(child); err != nil {
+ return err
+ }
+ }
+ return nil
+ }
+ if err := walk(&doc); err != nil {
+ return err
+ }
+ body, err = yaml.Marshal(&doc)
+ if err != nil {
+ return err
+ }
+ return destination.WriteFile("index.yaml", body, 0600)
+}
diff --git a/scripts/build/gallery/main_test.go b/scripts/build/gallery/main_test.go
new file mode 100644
index 000000000..646499b4b
--- /dev/null
+++ b/scripts/build/gallery/main_test.go
@@ -0,0 +1,83 @@
+// SPDX-License-Identifier: MIT
+package main
+
+import (
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+ "gopkg.in/yaml.v3"
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+)
+
+func TestGalleryPackage(t *testing.T) { RegisterFailHandler(Fail); RunSpecs(t, "Gallery packaging") }
+
+var _ = Describe("Gallery packaging", func() {
+ It("packages both official indexes with every repository-local base available offline", func() {
+ for _, source := range []string{"gallery", "backend"} {
+ out := GinkgoT().TempDir()
+ Expect(packageGallery("../../..", source, out)).To(Succeed())
+ body, err := os.ReadFile(filepath.Join(out, "index.yaml"))
+ Expect(err).ToNot(HaveOccurred())
+ var entries []map[string]any
+ Expect(yaml.Unmarshal(body, &entries)).To(Succeed())
+ Expect(entries).ToNot(BeEmpty())
+ for _, entry := range entries {
+ url, _ := entry["url"].(string)
+ Expect(url).ToNot(HavePrefix("github:mudler/LocalAI/"))
+ if strings.HasPrefix(url, "gallery/") {
+ Expect(filepath.Join(out, url)).To(BeAnExistingFile())
+ }
+ }
+ }
+ })
+ It("bundles local base configs and preserves external URLs and YAML aliases", func() {
+ root := GinkgoT().TempDir()
+ Expect(os.MkdirAll(filepath.Join(root, "gallery"), 0755)).To(Succeed())
+ Expect(os.WriteFile(filepath.Join(root, "gallery/base.yaml"), []byte("backend: llama-cpp\n"), 0644)).To(Succeed())
+ Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- &base\n name: first\n url: github:mudler/LocalAI/gallery/base.yaml@master\n- <<: *base\n name: second\n- name: external\n url: https://example.com/config.yaml\n"), 0644)).To(Succeed())
+ out := filepath.Join(root, "out")
+ Expect(packageGallery(root, "gallery", out)).To(Succeed())
+ data, err := os.ReadFile(filepath.Join(out, "index.yaml"))
+ Expect(err).ToNot(HaveOccurred())
+ var entries []map[string]any
+ Expect(yaml.Unmarshal(data, &entries)).To(Succeed())
+ Expect(entries[0]["url"]).To(Equal("gallery/base.yaml"))
+ Expect(entries[1]["url"]).To(Equal("gallery/base.yaml"))
+ Expect(entries[2]["url"]).To(Equal("https://example.com/config.yaml"))
+ body, err := os.ReadFile(filepath.Join(out, "gallery/base.yaml"))
+ Expect(err).ToNot(HaveOccurred())
+ Expect(string(body)).To(Equal("backend: llama-cpp\n"))
+ })
+ It("fails if a referenced config is missing or escapes the repository", func() {
+ for _, ref := range []string{"missing.yaml", "../../outside.yaml"} {
+ root := GinkgoT().TempDir()
+ Expect(os.Mkdir(filepath.Join(root, "gallery"), 0755)).To(Succeed())
+ Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- name: broken\n url: github:mudler/LocalAI/gallery/"+ref+"@master\n"), 0644)).To(Succeed())
+ Expect(packageGallery(root, "gallery", filepath.Join(root, "out"))).ToNot(Succeed())
+ }
+ })
+ It("rejects symlink escapes when reading configs or writing the bundle", func() {
+ for _, location := range []string{"source", "output"} {
+ root, out, outside := GinkgoT().TempDir(), GinkgoT().TempDir(), GinkgoT().TempDir()
+ for _, dir := range []string{filepath.Join(root, "gallery"), filepath.Join(out, "gallery")} {
+ Expect(os.Mkdir(dir, 0700)).To(Succeed())
+ }
+ index := []byte("- name: test\n url: github:mudler/LocalAI/gallery/base.yaml@master\n")
+ Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), index, 0600)).To(Succeed())
+ outsideFile := filepath.Join(outside, "base.yaml")
+ Expect(os.WriteFile(outsideFile, []byte("outside"), 0600)).To(Succeed())
+ link := filepath.Join(root, "gallery/base.yaml")
+ if location == "output" {
+ Expect(os.WriteFile(link, []byte("inside"), 0600)).To(Succeed())
+ link = filepath.Join(out, "gallery/base.yaml")
+ }
+ Expect(os.Symlink(outsideFile, link)).To(Succeed())
+ Expect(packageGallery(root, "gallery", out)).ToNot(Succeed(), location)
+ data, err := os.ReadFile(outsideFile)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(string(data)).To(Equal("outside"))
+ }
+ })
+})
diff --git a/swagger/docs.go b/swagger/docs.go
index 6ae74b94a..c447043ae 100644
--- a/swagger/docs.go
+++ b/swagger/docs.go
@@ -7313,6 +7313,9 @@ const docTemplate = `{
"id": {
"type": "string"
},
+ "metadata": {
+ "type": "object"
+ },
"model": {
"type": "string"
},
@@ -7807,6 +7810,10 @@ const docTemplate = `{
},
"id": {
"type": "string"
+ },
+ "size_vram": {
+ "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
+ "type": "integer"
}
}
},
diff --git a/swagger/swagger.json b/swagger/swagger.json
index c19487f6c..b4e49b347 100644
--- a/swagger/swagger.json
+++ b/swagger/swagger.json
@@ -7310,6 +7310,9 @@
"id": {
"type": "string"
},
+ "metadata": {
+ "type": "object"
+ },
"model": {
"type": "string"
},
@@ -7804,6 +7807,10 @@
},
"id": {
"type": "string"
+ },
+ "size_vram": {
+ "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
+ "type": "integer"
}
}
},
diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml
index 6f9f6b1b7..34fcfc8d7 100644
--- a/swagger/swagger.yaml
+++ b/swagger/swagger.yaml
@@ -2266,6 +2266,8 @@ definitions:
type: array
id:
type: string
+ metadata:
+ type: object
model:
type: string
object:
@@ -2652,6 +2654,11 @@ definitions:
type: string
id:
type: string
+ size_vram:
+ description: |-
+ SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
+ the backend process tree has no complete supported reading.
+ type: integer
type: object
schema.SystemInformationResponse:
properties:
diff --git a/website/data/stats.yaml b/website/data/stats.yaml
index a54e240e8..253b5444d 100644
--- a/website/data/stats.yaml
+++ b/website/data/stats.yaml
@@ -3,10 +3,10 @@
# The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh,
# which runs weekly from .github/workflows/refresh-site-counters.yml. Editing
# them by hand works but will be overwritten on the next run.
-stars: 48949
-forks: 4430
-contributors: 237
-releases: 135
+stars: 49204
+forks: 4459
+contributors: 245
+releases: 136
# The GitHub API cannot answer for this one, so it is maintained by hand and
# the refresh script carries it through untouched.