Merge remote-tracking branch 'origin/master' into feat/failover-chains

Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Assisted-by: Claude:claude-opus-5-5 [Claude Code]
This commit is contained in:
Ettore Di Giacinto committed 2026-09-27 19:57:52 +00:00
commit dae9a431e8
69 files changed
+2725 -227

No files matched your search

+1 -1
View File
@@ -355,7 +355,7 @@ jobs:
with:
backend: ${{ matrix.backend }}
build-type: ${{ matrix.build-type }}
go-version: "1.25.x"
go-version: "1.27.x"
tag-suffix: ${{ matrix.tag-suffix }}
lang: ${{ matrix.lang || 'python' }}
use-pip: ${{ matrix.backend == 'diffusers' }}
+2 -1
View File
@@ -252,7 +252,8 @@ jobs:
name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
# Release matrices and their retries can outlive a one-day artifact.
retention-days: 7
- name: Build (PR)
uses: docker/build-push-action@v7
+2 -1
View File
@@ -22,7 +22,8 @@ on:
type: string
go-version:
description: 'Go version to use'
default: '1.24.x'
# Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs.
default: '1.27.x'
type: string
tag-suffix:
description: 'Tag suffix for the built image'
+1 -1
View File
@@ -281,7 +281,7 @@ jobs:
with:
backend: ${{ matrix.backend }}
build-type: ${{ matrix.build-type }}
go-version: "1.25.x"
go-version: "1.27.x"
tag-suffix: ${{ matrix.tag-suffix }}
lang: ${{ matrix.lang || 'python' }}
use-pip: ${{ matrix.backend == 'diffusers' }}
+78
View File
@@ -0,0 +1,78 @@
name: Publish official OCI galleries
on:
push:
branches: [master]
paths:
- 'gallery/**'
- 'backend/index.yaml'
- 'scripts/build/gallery/**'
- '.github/workflows/gallery_publish.yml'
workflow_dispatch:
permissions:
contents: read
concurrency:
group: publish-official-galleries
cancel-in-progress: false
jobs:
publish:
if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master'
runs-on: ubuntu-latest
permissions:
contents: read
id-token: write
env:
COSIGN_EXPERIMENTAL: '1'
GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends
strategy:
matrix:
include:
- source: gallery
tag: gallery-models
- source: backend
tag: gallery-backends
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
- name: Test and package gallery
env:
GALLERY_SOURCE: ${{ matrix.source }}
run: |
go test ./scripts/build/gallery -count=1
go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery"
- uses: oras-project/setup-oras@v1
with:
version: '1.3.0'
- uses: sigstore/cosign-installer@v3
with:
cosign-release: 'v2.6.5'
- name: Login to Quay.io
uses: docker/login-action@v4
with:
registry: quay.io
username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
- name: Publish and sign gallery
shell: bash
env:
GALLERY_TAG: ${{ matrix.tag }}
run: |
set -euo pipefail
cd "$RUNNER_TEMP/gallery"
files=()
while IFS= read -r -d '' file; do
files+=("${file#./}:application/yaml")
done < <(find . -type f -print0 | sort -z)
# Publish an immutable revision, then expose latest only after signing.
ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA"
oras push --artifact-type application/vnd.localai.gallery.v1 \
--format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json"
digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json")
cosign sign --yes --new-bundle-format \
--registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest"
oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG"
+1 -1
View File
@@ -40,7 +40,7 @@ jobs:
# fetch their own toolchains, and no step uses sudo, apt, make or unzip.
runs-on: ${{ github.repository == 'mudler/LocalAI' && 'arc-runner-set' || 'ubuntu-latest' }}
env:
HUGO_VERSION: "0.146.3"
HUGO_VERSION: "0.166.0"
steps:
- name: Checkout
uses: actions/checkout@v7
+1 -1
View File
@@ -9,7 +9,7 @@
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
# rebuild and so the bump bot can see the pin.
AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22
AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
+1 -1
View File
@@ -1,5 +1,5 @@
IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc
IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
CMAKE_ARGS?=
+1 -1
View File
@@ -1,5 +1,5 @@
LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4
LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
CMAKE_ARGS?=
+1 -1
View File
@@ -1,7 +1,7 @@
# Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant.
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64
TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d
LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant
CMAKE_ARGS?=
@@ -1,52 +0,0 @@
diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
index 680fd12..ffd6604 100644
--- a/ggml/src/ggml-cuda/fattn-vec.cuh
+++ b/ggml/src/ggml-cuda/fattn-vec.cuh
@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0);
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16);
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0);
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16);
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
index 5c614a9..d765cfc 100644
--- a/ggml/src/ggml-cuda/fattn.cu
+++ b/ggml/src/ggml-cuda/fattn.cu
@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16)
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16)
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0)
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0)
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0)
#ifdef GGML_CUDA_FA_ALL_QUANTS
FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16)
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
index a93be56..3630d87 100644
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
@@ -5,4 +5,3 @@
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
index 3c806c2..c8a4d9f 100644
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
@@ -5,4 +5,3 @@
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
index 180902f..1646ef0 100644
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
@@ -5,4 +5,3 @@
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# CrispASR version (release tag)
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee
CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
SO_TARGET?=libgocrispasr.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+2 -2
View File
@@ -1,6 +1,6 @@
# parakeet-cpp backend Makefile.
#
# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
# (.github/bump_deps.sh) can find and update it - matches the
# whisper.cpp / ds4 / vibevoice-cpp convention.
#
@@ -15,7 +15,7 @@
# That's what the L0 smoke test uses. The default target below does the
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
GOCMD?=go
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# stablediffusion.cpp (ggml)
STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
STABLEDIFFUSION_GGML_VERSION?=b167b942f77ecb17e7f78e163a8c32ff7ac95c10
STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551
CMAKE_ARGS+=-DGGML_MAX_NAME=128
+6 -6
View File
@@ -710,14 +710,14 @@ void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled) {
params->enabled = enabled;
}
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y) {
params->tile_size_x = tile_size_x;
params->tile_size_y = tile_size_y;
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h) {
params->tile_size_w = tile_size_w;
params->tile_size_h = tile_size_h;
}
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y) {
params->rel_size_x = rel_size_x;
params->rel_size_y = rel_size_y;
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h) {
params->rel_size_w = rel_size_w;
params->rel_size_h = rel_size_h;
}
void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap) {
+2 -2
View File
@@ -6,8 +6,8 @@ extern "C" {
#endif
void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled);
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y);
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y);
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h);
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h);
void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap);
sd_tiling_params_t* sd_img_gen_params_get_vae_tiling_params(sd_img_gen_params_t *params);
+1 -1
View File
@@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e
# vllm.cpp version
VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp
VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb
VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c
# MLX GEMM provider (darwin/metal only; see the metal branch below for why).
# Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun
@@ -2,9 +2,9 @@ torch==2.7.1
llvmlite==0.49.0
numba==0.67.0
accelerate
transformers>=5.15.1
transformers>=5.17.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
@@ -2,9 +2,9 @@ torch==2.7.1
accelerate
llvmlite==0.49.0
numba==0.67.0
transformers>=5.15.1
transformers>=5.17.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
@@ -2,9 +2,9 @@
torch==2.9.0
llvmlite==0.49.0
numba==0.67.0
transformers>=5.15.1
transformers>=5.17.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
@@ -1,11 +1,11 @@
--extra-index-url https://download.pytorch.org/whl/rocm7.0
torch==2.10.0+rocm7.0
accelerate
transformers>=5.15.1
transformers>=5.17.0
llvmlite==0.49.0
numba==0.67.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
@@ -3,9 +3,9 @@ torch
optimum[openvino]
llvmlite==0.49.0
numba==0.67.0
transformers>=5.15.1
transformers>=5.17.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
@@ -2,9 +2,9 @@ torch==2.7.1
llvmlite==0.49.0
numba==0.67.0
accelerate
transformers>=5.15.1
transformers>=5.17.0
bitsandbytes
sentence-transformers==5.7.0
sentence-transformers==6.1.0
diffusers
soundfile
protobuf==7.36.1
+2 -2
View File
@@ -1,6 +1,6 @@
grpcio==1.83.0
grpcio==1.84.0
protobuf==7.36.1
certifi
setuptools
scipy==1.18.0
numpy>=2.5.2
numpy>=2.5.3
+12
View File
@@ -132,6 +132,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: true,
message: "Kokoros TTS model loaded".into(),
..Default::default()
}))
}
@@ -180,11 +181,13 @@ impl Backend for KokorosService {
return Ok(Response::new(backend::Result {
success: false,
message: format!("Failed to write WAV: {}", e),
..Default::default()
}));
}
Ok(Response::new(backend::Result {
success: true,
message: String::new(),
..Default::default()
}))
}
Err(e) => {
@@ -192,6 +195,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: false,
message: format!("TTS error: {}", e),
..Default::default()
}))
}
}
@@ -292,6 +296,7 @@ impl Backend for KokorosService {
Ok(Response::new(backend::Result {
success: true,
message: "Model freed".into(),
..Default::default()
}))
}
@@ -348,6 +353,13 @@ impl Backend for KokorosService {
Err(Status::unimplemented("Not supported"))
}
async fn animate3_d(
&self,
_: Request<backend::Animate3DRequest>,
) -> Result<Response<backend::Result>, Status> {
Err(Status::unimplemented("Not supported"))
}
async fn audio_transcription(
&self,
_: Request<backend::TranscriptRequest>,
+11 -1
View File
@@ -47,10 +47,13 @@ type Gallery struct {
// fallback for availability, not a load-balancing pool: the primary is
// always preferred, and a mirror is only consulted after the one before
// it fails. Any URI the gallery loader understands works here
// (https://, github:, file://).
// (https://, github:, file://, oci://).
Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"`
Name string `json:"name" yaml:"name"`
Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"`
// ArtifactVerification overrides Verification only for the gallery OCI artifact.
// Backend images keep their separate Verification policy.
ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"`
}
// Equal reports whether two gallery entries describe the same gallery.
@@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool {
if !slices.Equal(g.Mirrors, other.Mirrors) {
return false
}
if g.ArtifactVerification == nil || other.ArtifactVerification == nil {
if g.ArtifactVerification != other.ArtifactVerification {
return false
}
} else if *g.ArtifactVerification != *other.ArtifactVerification {
return false
}
if g.Verification == nil || other.Verification == nil {
return g.Verification == other.Verification
}
+21
View File
@@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() {
Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery"))
})
})
var _ = Describe("Gallery artifact verification", func() {
It("compares artifact policies by value and preserves them in JSON and YAML", func() {
a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
Expect(a.Equal(b)).To(BeTrue())
b.ArtifactVerification.Identity = "another-workflow"
Expect(a.Equal(b)).To(BeFalse())
b.ArtifactVerification = nil
Expect(a.Equal(b)).To(BeFalse())
raw, err := json.Marshal(a)
Expect(err).ToNot(HaveOccurred())
Expect(json.Unmarshal(raw, &b)).To(Succeed())
Expect(a.Equal(b)).To(BeTrue())
raw, err = yaml.Marshal(a)
Expect(err).ToNot(HaveOccurred())
b = config.Gallery{}
Expect(yaml.Unmarshal(raw, &b)).To(Succeed())
Expect(a.Equal(b)).To(BeTrue())
})
})
+2 -2
View File
@@ -17,8 +17,8 @@ import (
// a caching mirror of the files below. The GitHub URI stays as a mirror so an
// install still resolves its gallery unchanged whenever the primary is
// unreachable - see the fallback chain in core/gallery/gallery_mirrors.go.
const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]`
const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]`
const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
func mustGalleries(jsonList string) []Gallery {
var g []Gallery
+8 -4
View File
@@ -10,22 +10,22 @@ import (
)
var _ = Describe("default galleries", func() {
It("serves the model gallery from index.localai.io with GitHub as a mirror", func() {
It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() {
var galleries []config.Gallery
Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed())
Expect(galleries).To(HaveLen(1))
Expect(galleries[0].Name).To(Equal("localai"))
Expect(galleries[0].URL).To(Equal("https://index.localai.io/models"))
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"}))
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"}))
})
It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() {
It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() {
var galleries []config.Gallery
Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed())
Expect(galleries).To(HaveLen(1))
Expect(galleries[0].Name).To(Equal("localai"))
Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends"))
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"}))
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"}))
})
// The mirror is the whole reason this default is safe to ship: if
@@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() {
Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed())
for _, g := range galleries {
Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name)
Expect(g.ArtifactVerification).ToNot(BeNil())
Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"))
Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com"))
Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust")
}
}
})
+1 -1
View File
@@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string {
if !looksLikeOCIGallery(candidate) {
continue
}
dir := ociGalleryCacheDir(basePath, candidate, g.Verification)
dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g))
if dir == "" {
continue
}
+31 -19
View File
@@ -4,7 +4,6 @@ import (
"context"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"sync"
@@ -19,6 +18,7 @@ import (
"github.com/mudler/LocalAI/pkg/vram"
"github.com/mudler/LocalAI/pkg/xsync"
"github.com/mudler/xlog"
"golang.org/x/sync/singleflight"
"gopkg.in/yaml.v3"
)
@@ -276,13 +276,12 @@ func FindGalleryElement[T GalleryElement](models []T, name string) T {
func AvailableGalleryModels(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryModel], error) {
var models []*GalleryModel
isInstalled := installedConfigs(systemState.Model.ModelsPath)
// Get models from galleries
for _, gallery := range galleries {
galleryModels, err := getGalleryElements(gallery, systemState.Model.ModelsPath, systemState.RequireBackendIntegrity, func(model *GalleryModel) bool {
if _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", model.GetName()))); err == nil {
return true
}
return false
return isInstalled(model.GetName())
})
if err != nil {
return nil, err
@@ -351,6 +350,7 @@ var (
// same cache-defeating loop the refresh interval exists to stop.
availableModelsLoaded bool
refreshing atomic.Bool
coldLoad singleflight.Group
galleryGeneration atomic.Uint64
lastRefreshUnixNano atomic.Int64
)
@@ -429,12 +429,15 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste
availableModelsMu.RUnlock()
if loaded {
// The directory is read before taking the lock. Held across the
// filesystem work, the lock serialized every caller behind it, and a
// page view is dozens of concurrent callers.
isInstalled := installedConfigs(systemState.Model.ModelsPath)
// Refresh installed status under write lock to avoid races with
// concurrent readers and the background refresh goroutine.
availableModelsMu.Lock()
for _, m := range cached {
_, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", m.GetName())))
m.SetInstalled(err == nil)
m.SetInstalled(isInstalled(m.GetName()))
}
availableModelsMu.Unlock()
// Trigger a background refresh if one is not already running.
@@ -442,20 +445,29 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste
return cached, nil
}
// No cache yet — must do a blocking load.
models, err := AvailableGalleryModels(galleries, systemState)
// No cache yet, so the load blocks. Callers arriving while it runs wait
// for it instead of each starting their own: a page view on a fresh
// server is the listing plus one estimate per row at once, and each load
// fetches the gallery index and every config it references.
v, err, _ := coldLoad.Do("gallery", func() (any, error) {
models, err := AvailableGalleryModels(galleries, systemState)
if err != nil {
return nil, err
}
availableModelsMu.Lock()
availableModelsCache = models
availableModelsLoaded = true
galleryGeneration.Add(1)
availableModelsMu.Unlock()
lastRefreshUnixNano.Store(time.Now().UnixNano())
return models, nil
})
if err != nil {
return nil, err
}
availableModelsMu.Lock()
availableModelsCache = models
availableModelsLoaded = true
galleryGeneration.Add(1)
availableModelsMu.Unlock()
lastRefreshUnixNano.Store(time.Now().UnixNano())
return models, nil
return v.(GalleryElements[*GalleryModel]), nil
}
// triggerGalleryRefresh starts a background goroutine that refreshes the
@@ -634,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]()
// would also point relative entry urls at an unpacked tree the new policy has
// not produced yet, so they could not be installed.
func galleryIndexCacheKey(g config.Gallery) string {
return g.Name + "-" + galleryCacheName(g.URL, g.Verification)
return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g))
}
func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) {
+138
View File
@@ -0,0 +1,138 @@
package gallery_test
import (
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"sync"
"sync/atomic"
"time"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/gallery"
"github.com/mudler/LocalAI/pkg/system"
)
// The models directory is often network storage (SMB, NFS), where every
// filesystem call is a round trip. The cached listing is read by the gallery
// page and by one VRAM estimate per row, so whatever it costs is paid dozens
// of times per page view.
var _ = Describe("Gallery cache installed status", func() {
const index = `
- name: plain
backend: llama-cpp
- name: linked
backend: llama-cpp
- name: dangling
backend: llama-cpp
- name: later
backend: llama-cpp
- name: absent
backend: llama-cpp
`
var (
modelsDir string
state *system.SystemState
galleries []config.Gallery
hits atomic.Int32
delay time.Duration
)
BeforeEach(func() {
var err error
modelsDir, err = os.MkdirTemp("", "gallery-installed")
Expect(err).ToNot(HaveOccurred())
DeferCleanup(func() { _ = os.RemoveAll(modelsDir) })
state, err = system.GetSystemState(system.WithModelPath(modelsDir))
Expect(err).ToNot(HaveOccurred())
hits.Store(0)
delay = 0
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
hits.Add(1)
time.Sleep(delay)
_, _ = w.Write([]byte(index))
}))
DeferCleanup(server.Close)
galleries = []config.Gallery{{Name: "test", URL: server.URL + "/index.yaml"}}
gallery.ResetGalleryModelCache()
DeferCleanup(gallery.ResetGalleryModelCache)
})
installed := func(models gallery.GalleryElements[*gallery.GalleryModel]) map[string]bool {
out := map[string]bool{}
for _, m := range models {
out[m.Name] = m.Installed
}
return out
}
It("reports what os.Stat would, for files, symlinks and dangling symlinks", func() {
Expect(os.WriteFile(filepath.Join(modelsDir, "plain.yaml"), []byte("name: plain\n"), 0o644)).To(Succeed())
target := filepath.Join(modelsDir, "target.txt")
Expect(os.WriteFile(target, []byte("name: linked\n"), 0o644)).To(Succeed())
Expect(os.Symlink(target, filepath.Join(modelsDir, "linked.yaml"))).To(Succeed())
Expect(os.Symlink(filepath.Join(modelsDir, "missing"), filepath.Join(modelsDir, "dangling.yaml"))).To(Succeed())
// Both the blocking first load and the cached path set the flag, and
// they must agree.
for range 2 {
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(installed(models)).To(Equal(map[string]bool{
"plain": true,
"linked": true,
"dangling": false,
"later": false,
"absent": false,
}))
}
})
It("picks up a config written after the gallery was cached", func() {
_, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(os.WriteFile(filepath.Join(modelsDir, "later.yaml"), []byte("name: later\n"), 0o644)).To(Succeed())
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(installed(models)).To(HaveKeyWithValue("later", true))
})
It("reports nothing installed when the models directory is gone", func() {
_, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(os.RemoveAll(modelsDir)).To(Succeed())
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(installed(models)).To(HaveEach(BeFalse()))
})
It("shares one upstream load between concurrent callers on a cold cache", func() {
// Slow enough that every caller arrives while the first load is still
// in flight, which is what a page view does to a freshly started
// server: the listing and every row's estimate at once.
delay = 300 * time.Millisecond
var wg sync.WaitGroup
for range 8 {
wg.Go(func() {
defer GinkgoRecover()
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
Expect(err).ToNot(HaveOccurred())
Expect(models).To(HaveLen(5))
})
}
wg.Wait()
Expect(hits.Load()).To(Equal(int32(1)))
})
})
+2 -2
View File
@@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
if !looksLikeOCIGallery(g.URL) {
return nil
}
return g.Verification
return galleryArtifactPolicy(g)
}
// verifiableCandidates drops the candidates that cannot answer for a signed
@@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
// at, and after a refusal it would turn "this artifact is not trusted" into
// "use this other, unchecked copy instead".
func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string {
if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) {
if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) {
return candidates
}
out := make([]string, 0, len(candidates))
+13 -5
View File
@@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) {
// later fetch served would hand the user a truncated gallery with no sign that
// anything went wrong.
func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) {
policy := galleryArtifactPolicy(g)
// Checked before the cache: a copy unpacked while strict integrity was
// off was never verified, and turning strict integrity on must not keep
// serving it for the rest of its TTL.
if g.Verification == nil && requireIntegrity {
if policy == nil && requireIntegrity {
return nil, &galleryVerificationError{
strict: true,
err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
}
}
cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification)
cacheDir := ociGalleryCacheDir(basePath, candidate, policy)
if cacheDir == "" {
return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate)
}
@@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
pullRef := downloader.URI(candidate).OCIReference()
if g.Verification != nil {
if policy != nil {
// Resolve first, verify the digest, then pull that same digest.
// Nothing has been fetched at this point beyond the manifest, so a
// policy failure leaves no content anywhere.
@@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
if err != nil {
return nil, err
}
if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil {
if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil {
// Only a decision about the artifact is a refusal. The
// verifier also reaches the Sigstore TUF mirror and the
// registry, and a timeout or a 5xx there says nothing about
@@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
return body, nil
}
func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification {
if g.ArtifactVerification != nil {
return g.ArtifactVerification
}
return g.Verification
}
+15
View File
@@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() {
})
})
It("uses the artifact policy without replacing backend image verification", func() {
srv, _, _ := ociRegistry()
url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}})
backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"}
artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"}
var seen *config.GalleryVerification
stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil })
g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy}
_, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true)
Expect(source).To(Equal(url))
Expect(err).ToNot(HaveOccurred())
Expect(seen).To(Equal(artifactPolicy))
Expect(g.Verification).To(Equal(backendPolicy))
})
It("refuses an unsigned gallery in strict integrity mode", func() {
srv, _, blobs := ociRegistry()
url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{
+61
View File
@@ -0,0 +1,61 @@
package gallery
import (
"errors"
"io/fs"
"os"
"path/filepath"
"strings"
)
const modelConfigExt = ".yaml"
// installedConfigs answers "does <modelsPath>/<name>.yaml exist?" for every
// entry of a gallery from a single read of the models directory.
//
// The question used to be asked with one os.Stat per gallery entry. The gallery
// holds thousands of entries and the models directory is often network storage
// (SMB, NFS), where each Stat is a round trip, so one listing cost seconds. The
// listing is read by the gallery page and by one VRAM estimate per row, which
// turned a page view into minutes.
//
// Answers match os.Stat on the same path: a symlink counts only when its target
// exists, and anything else carrying the name counts, directories included.
// Names that are not a plain file name are checked with os.Stat directly, since
// they point outside the listed directory.
func installedConfigs(modelsPath string) func(name string) bool {
statInstalled := func(name string) bool {
_, err := os.Stat(filepath.Join(modelsPath, name+modelConfigExt))
return err == nil
}
entries, err := os.ReadDir(modelsPath)
if err != nil {
if errors.Is(err, fs.ErrNotExist) {
return func(string) bool { return false }
}
// A directory that exists but cannot be listed may still answer a
// Stat, so fall back rather than report everything as not installed.
return statInstalled
}
present := make(map[string]struct{}, len(entries))
for _, e := range entries {
base, ok := strings.CutSuffix(e.Name(), modelConfigExt)
if !ok {
continue
}
if e.Type()&fs.ModeSymlink != 0 && !statInstalled(base) {
continue
}
present[base] = struct{}{}
}
return func(name string) bool {
if strings.ContainsRune(name, '/') || strings.ContainsRune(name, filepath.Separator) {
return statInstalled(name)
}
_, ok := present[name]
return ok
}
}
+6
View File
@@ -8,6 +8,7 @@ import (
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/monitoring"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/xsysinfo"
)
// SystemInformations returns the system informations
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
entry.Process = proc
}
}
if pid, ok := localPID(m); ok {
if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
entry.SizeVRAM = &used
}
}
sysmodels = append(sysmodels, entry)
}
if sampler != nil {
@@ -0,0 +1,49 @@
// SPDX-License-Identifier: MIT
package localai_test
import (
"encoding/json"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/http/endpoints/localai"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/system"
process "github.com/mudler/go-processmanager"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("SystemInformations memory", func() {
It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
path, err := os.MkdirTemp("", "system-info-")
Expect(err).NotTo(HaveOccurred())
DeferCleanup(os.RemoveAll, path)
configFile := filepath.Join(path, "remote.yaml")
Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
cl := config.NewModelConfigLoader(path)
Expect(cl.ReadModelConfig(configFile)).To(Succeed())
ml := model.NewModelLoader(&system.SystemState{})
store := model.NewInMemoryModelStore()
store.Set("remote", model.NewModel("remote", "worker:50051", nil))
store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
ml.SetModelStore(store)
app := echo.New()
app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
rec := httptest.NewRecorder()
app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
Expect(rec.Code).To(Equal(http.StatusOK))
var response struct {
Models []map[string]any `json:"loaded_models"`
}
Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
Expect(response.Models).To(ConsistOf(
map[string]any{"id": "remote", "backend": "llama-cpp"},
map[string]any{"id": "stopped"},
))
})
})
+47 -81
View File
@@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
return true
}
// Try JSON parsing as fallback
jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true)
if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount {
// Only completed JSON calls can be emitted as completed SSE items.
jsonResults := parseStreamingJSONToolCalls(cleanedResult)
if len(jsonResults) > lastEmittedToolCallCount {
for i := lastEmittedToolCallCount; i < len(jsonResults); i++ {
jsonObj := jsonResults[i]
if name, ok := jsonObj["name"].(string); ok && name != "" {
args := "{}"
if argsVal, ok := jsonObj["arguments"]; ok {
if argsStr, ok := argsVal.(string); ok {
args = argsStr
} else {
argsBytes, _ := json.Marshal(argsVal)
args = string(argsBytes)
}
}
tc := jsonResults[i]
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
outputIndex++
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
outputIndex++
functionCallItem := &schema.ORItemField{
Type: "function_call",
ID: toolCallID,
Status: "completed",
CallID: toolCallID,
Name: name,
Arguments: args,
}
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
Item: functionCallItem,
})
sequenceNumber++
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
Item: functionCallItem,
})
sequenceNumber++
functionCallItem := &schema.ORItemField{
Type: "function_call",
ID: toolCallID,
Status: "completed",
CallID: toolCallID,
Name: tc.Name,
Arguments: tc.Arguments,
}
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
Item: functionCallItem,
})
sequenceNumber++
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
Item: functionCallItem,
})
sequenceNumber++
}
lastEmittedToolCallCount = len(jsonResults)
c.Response().Flush()
@@ -2424,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
}
// Non-tool-call streaming path
messageOutputIndex := outputIndex
var reasoningOutputIndex int
// Emit output_item.added for message
currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String())
messageItem := &schema.ORItemField{
@@ -2436,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
Item: messageItem,
})
sequenceNumber++
@@ -2448,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.added",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
ContentIndex: &currentContentIndex,
Part: &emptyTextPart,
})
@@ -2471,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
}
// Handle reasoning item
if extractor.Reasoning() != "" {
if extractor.Reasoning() != "" || reasoningDelta != "" {
// Check if we need to create reasoning item
if currentReasoningID == "" {
outputIndex++
reasoningOutputIndex = outputIndex
currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String())
reasoningItem := &schema.ORItemField{
Type: "reasoning",
@@ -2484,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.added",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
Item: reasoningItem,
})
sequenceNumber++
@@ -2496,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.added",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
ContentIndex: &currentReasoningContentIndex,
Part: &emptyPart,
})
@@ -2509,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.delta",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
ContentIndex: &currentReasoningContentIndex,
Delta: strPtr(reasoningDelta),
Logprobs: emptyLogprobs(),
@@ -2526,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.delta",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
ContentIndex: &currentContentIndex,
Delta: strPtr(contentDelta),
Logprobs: emptyLogprobs(),
@@ -2595,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.done",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
ContentIndex: &currentReasoningContentIndex,
Text: strPtr(finalReasoning),
Logprobs: emptyLogprobs(),
@@ -2608,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.done",
SequenceNumber: sequenceNumber,
ItemID: currentReasoningID,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
ContentIndex: &currentReasoningContentIndex,
Part: &reasoningPart,
})
@@ -2624,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
OutputIndex: &reasoningOutputIndex,
Item: reasoningItem,
})
sequenceNumber++
@@ -2658,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.output_text.done",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
ContentIndex: &currentContentIndex,
Text: strPtr(result),
Logprobs: logprobsPtr(mcpStreamLogprobs),
@@ -2671,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
Type: "response.content_part.done",
SequenceNumber: sequenceNumber,
ItemID: currentMessageID,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
ContentIndex: &currentContentIndex,
Part: &resultPart,
})
@@ -2683,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
sendSSEEvent(c, &schema.ORStreamEvent{
Type: "response.output_item.done",
SequenceNumber: sequenceNumber,
OutputIndex: &outputIndex,
OutputIndex: &messageOutputIndex,
Item: messageItem,
})
sequenceNumber++
@@ -2723,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
// Emit response.completed
now := time.Now().Unix()
// Collect final output items (reasoning first, then messages, then tool calls)
var finalOutputItems []schema.ORItemField
// Add reasoning item if it exists
if currentReasoningID != "" && finalReasoning != "" {
finalOutputItems = append(finalOutputItems, schema.ORItemField{
Type: "reasoning",
ID: currentReasoningID,
Status: "completed",
Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)},
})
}
// Add message item
if len(collectedOutputItems) > 0 {
// Use collected items (may include reasoning already)
for _, item := range collectedOutputItems {
if item.Type == "message" {
finalOutputItems = append(finalOutputItems, item)
}
}
} else {
finalOutputItems = append(finalOutputItems, *messageItem)
}
// Add function_call items from fallback
for _, item := range collectedOutputItems {
if item.Type == "function_call" {
finalOutputItems = append(finalOutputItems, item)
}
}
// The final output array must use the indices announced in the stream.
// The message is opened first, followed by reasoning and fallback calls.
finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...)
responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{
InputTokens: noToolTokenUsage.Prompt,
OutputTokens: noToolTokenUsage.Completion,
@@ -0,0 +1,148 @@
// SPDX-License-Identifier: MIT
package openresponses
import (
"context"
"encoding/json"
"net/http/httptest"
"strings"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/backend"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/schema"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/LocalAI/pkg/model"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("Responses stream item consistency", func() {
DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) {
originalInference := backend.ModelInferenceFunc
DeferCleanup(func() { backend.ModelInferenceFunc = originalInference })
backend.ModelInferenceFunc = func(
ctx context.Context, prompt string, messages schema.Messages,
images, videos, audios []string, loader *model.ModelLoader,
cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig,
tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string,
logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string,
) (func() (backend.LLMResponse, error), error) {
return func() (backend.LLMResponse, error) {
for i, token := range tokens {
usage := backend.TokenUsage{}
if len(chatDeltas) > 0 {
usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]}
}
if !tokenCallback(token, usage) {
break
}
}
return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil
}, nil
}
cfg := &config.ModelConfig{}
cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback
cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)<tool_call>(.*?)</tool_call>`}
recorder := httptest.NewRecorder()
request := httptest.NewRequest("POST", "/v1/responses", nil)
c := echo.New().NewContext(request, recorder)
input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true}
err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil)
Expect(err).NotTo(HaveOccurred())
Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n"))
var events []schema.ORStreamEvent
var completed *schema.ORResponseResource
for _, line := range strings.Split(recorder.Body.String(), "\n") {
if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" {
continue
}
var event schema.ORStreamEvent
Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed())
Expect(event.Type).NotTo(Equal("error"))
events = append(events, event)
if event.Type == "response.completed" {
completed = event.Response
}
}
Expect(completed).NotTo(BeNil())
wantCount := 1
if wantReasoning != "" {
wantCount++
}
if fallback {
wantCount++
}
Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls")
indices := map[string]int{}
done := map[string]int{}
deltas := map[string]string{}
for i, event := range events {
Expect(event.SequenceNumber).To(Equal(i))
if event.Type == "response.output_item.added" {
Expect(event.Item).NotTo(BeNil())
Expect(event.OutputIndex).NotTo(BeNil())
Expect(indices).NotTo(HaveKey(event.Item.ID))
Expect(*event.OutputIndex).To(Equal(len(indices)))
indices[event.Item.ID] = *event.OutputIndex
}
id := event.ItemID
if event.Item != nil {
id = event.Item.ID
}
if id == "" {
continue
}
Expect(indices).To(HaveKey(id))
Expect(event.OutputIndex).NotTo(BeNil())
Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id)
Expect(completed.Output[indices[id]].ID).To(Equal(id))
if event.Type == "response.output_item.done" {
done[id]++
Expect(event.Item.Status).To(Equal("completed"))
Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type))
if event.Item.Type == "function_call" {
Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name))
Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments))
} else {
Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content))
}
}
if event.Type == "response.output_text.delta" {
deltas[id] += *event.Delta
}
}
Expect(indices).To(HaveLen(wantCount))
for _, item := range completed.Output {
Expect(done[item.ID]).To(Equal(1))
switch item.Type {
case "message", "reasoning":
want := wantAnswer
if item.Type == "reasoning" {
want = wantReasoning
}
parts, ok := item.Content.([]any)
Expect(ok).To(BeTrue())
Expect(parts).To(HaveLen(1))
Expect(parts[0].(map[string]any)["text"]).To(Equal(want))
if !fallback {
Expect(deltas[item.ID]).To(Equal(want))
}
case "function_call":
Expect(item.Name).To(Equal("get_weather"))
Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`))
Expect(item.CallID).NotTo(BeEmpty())
default:
Fail("unexpected output item type: " + item.Type)
}
}
},
Entry("tagged reasoning and answer", []string{"<think>", "Let me think.", "</think>", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false),
Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false),
Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false),
Entry("automatic fallback tool call", []string{`<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "", "", true),
Entry("reasoning and automatic fallback tool call", []string{"<think>", "Let me think.", "</think>", `<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "Let me think.", "", true),
)
})
@@ -0,0 +1,35 @@
package openresponses
import (
"encoding/json"
"github.com/mudler/LocalAI/pkg/functions"
)
func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults {
// Partial parsing heals unfinished arguments. The caller emits terminal
// events and never revisits emitted calls, so only accept complete JSON.
// Keep completed objects returned before an unfinished trailing object.
objects, _ := functions.ParseJSONIterative(text, false)
var calls []functions.FuncCallResults
for _, object := range objects {
name, ok := object["name"].(string)
if !ok || name == "" {
continue
}
arguments := "{}"
if value, ok := object["arguments"]; ok {
if s, ok := value.(string); ok {
arguments = s
} else {
data, err := json.Marshal(value)
if err != nil {
continue
}
arguments = string(data)
}
}
calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments})
}
return calls
}
@@ -0,0 +1,44 @@
package openresponses
import (
"github.com/mudler/LocalAI/pkg/functions"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("Streaming JSON tool calls", func() {
It("waits for the arguments before completing a split call", func() {
Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty())
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty())
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
}))
})
It("does not complete a call at any intermediate token boundary", func() {
text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}`
for end := 1; end < len(text); end++ {
Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end])
}
Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1))
})
It("keeps completed calls while the next call is incomplete", func() {
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
}))
})
It("preserves string arguments and calls that take no arguments", func() {
Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
{Name: "status", Arguments: `{}`},
}))
})
It("does not count unrelated JSON objects as emitted calls", func() {
Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{
{Name: "status", Arguments: `{}`},
}))
})
})
+3
View File
@@ -208,6 +208,9 @@ type SysInfoModel struct {
// when the model has no local process (a distributed worker holds it) or
// the process could not be read.
Process *SysInfoProcess `json:"process,omitempty"`
// SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
// the backend process tree has no complete supported reading.
SizeVRAM *uint64 `json:"size_vram,omitempty"`
}
// SysInfoProcess is a point-in-time reading of one backend process.
+1 -1
View File
@@ -99,7 +99,7 @@ type OpenAIResponse struct {
// OpenAI-SDK consumers that filter on a truthy `result.usage`
// (continuedev/continue, Kilo Code, Roo Code, etc.).
Usage *OpenAIUsage `json:"usage,omitempty"`
Metadata json.RawMessage `json:"metadata,omitempty"`
Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"`
}
// StreamOptions mirrors OpenAI's `stream_options` request field. The only
+25
View File
@@ -0,0 +1,25 @@
// SPDX-License-Identifier: MIT
package schema_test
import (
"encoding/json"
"github.com/mudler/LocalAI/core/schema"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("SysInfoModel memory", func() {
It("omits unavailable VRAM while preserving a measured zero", func() {
entry := schema.SysInfoModel{ID: "model"}
encoded, err := json.Marshal(entry)
Expect(err).NotTo(HaveOccurred())
Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
zero := uint64(0)
entry.SizeVRAM = &zero
encoded, err = json.Marshal(entry)
Expect(err).NotTo(HaveOccurred())
Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
})
})
+2 -1
View File
@@ -59,6 +59,7 @@ services:
# capabilities: [gpu, utility]
#
# For legacy NVIDIA driver (for older NVIDIA Container Toolkit):
# Request compute for CUDA libraries (libcuda.so.1) and utility for NVML.
# environment:
# NVIDIA_DRIVER_CAPABILITIES: "compute,utility"
# init: true
@@ -68,7 +69,7 @@ services:
# devices:
# - driver: nvidia
# count: 1
# capabilities: [gpu, utility]
# capabilities: [gpu, compute, utility]
## Uncomment for PostgreSQL-backed knowledge base (see Agents docs)
# postgres:
+1 -1
View File
@@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model
local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master
```
See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files.
See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files.
### Prompt templates
+2
View File
@@ -82,6 +82,8 @@ tags:
### Verifying OCI Backends
The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details.
Backend galleries can require keyless Sigstore signatures for every OCI image
they provide. Add a `verification` policy to the gallery configuration, then
enable strict integrity mode:
+6 -2
View File
@@ -417,8 +417,12 @@ usage is reported back to the frontend:
NVML library (and therefore `nvidia-smi`) is not available inside the
container. CUDA compute still works, but the worker cannot query free VRAM
and the Nodes page will show the node as fully used. Set
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI
runtime, list `capabilities: [gpu, utility]` on the device reservation).
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime.
For Docker Compose with `driver: nvidia`, use
`capabilities: [gpu, compute, utility]` on the device reservation.
Docker derives driver capabilities from this reservation, so include `compute`
for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone
enables monitoring but does not provide CUDA libraries.
- **Run the container with `init: true` (or `docker run --init`).** The
worker process becomes PID 1 in the container and cannot reap zombies on
+107 -6
View File
@@ -39,6 +39,99 @@ Both views use the same model selection and store the view, search, filter, and
selection in the URL. Installing from Explore does not move you away from the
catalog; the entry updates in place when the operation finishes.
## Cyber-Tiel-Coder
Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp.
The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector.
To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`.
Both configurations use the embedded chat template and default to 32,768 context tokens.
The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license.
## Qwen3.8-27B Agention Precision
The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of
Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image
input and use a 32,768-token context by default.
Install with automatic variant selection:
```bash
local-ai models install qwen3.8-27b-agention-iq4-xs
```
To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs`
or `--variant qwen3.8-27b-agention-q4-k-m` to the same command.
The files use standard llama.cpp quantization types and the Apache-2.0 license.
See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF)
for quantization details. These entries do not enable MTP speculative decoding.
## Swift 1.5 Qwen3.8-27B GSQ-RCO
Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp.
The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model.
To select IQ3_S explicitly, run:
```bash
local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
```
The configurations use the embedded chat template and default to 32,768 context tokens.
These builds support text chat only: the publisher has no verified vision projector for this release.
They use standard GGUF files without MTP decoding.
See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms.
## Sharp-Spark-X2.5-4B
Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp.
The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants.
To select the publisher's recommended Q6 build, run:
```bash
local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6
```
All builds use a 32,768-token default context and the embedded Sharp-Spark chat template.
That template adds a terseness instruction to the system prompt.
See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details.
## MiMo-V2.6-Distill-Qwen-9B
Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp.
This MIT-licensed 9B Qwen3.5 fine-tune targets coding, agent tasks, and visual coding.
The gallery groups Q4_K_M and Q8_0 builds as variants; both include the F16 vision projector.
To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9b --variant mimo-v2.6-distill-qwen-9b-q8`.
The configurations default to 32,768 context tokens and use the model's embedded chat template.
See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details.
## Qwopus3.8 Flash V2
Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection:
```bash
local-ai models install qwopus3.8-27b-flash-v2
local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8
```
Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector.
Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256.
This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks.
See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations.
## ThinkingCap Qwen3.8-27B
Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input.
The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector.
LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly:
```bash
local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8
```
Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings.
MTP speculative decoding is not enabled by these entries.
The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE).
Review that license for permitted use.
## Hemmingway-1
Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants.
@@ -88,7 +181,7 @@ To use a gallery that needs authentication, such as a private GitHub repository
A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted.
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
```json
GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
@@ -142,10 +235,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out
### Signature verification
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
```json
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
```
The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends.
@@ -164,9 +257,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK
The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}).
{{% notice warning %}}
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities.
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
{{% /notice %}}
### Official gallery publishing
The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA.
To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title.
### Private registries
A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use:
@@ -198,10 +299,10 @@ GALLERIES=[{"name":"<GALLERY_NAME>", "url":"<GALLERY_URL"}]
For example, to spell out the default `localai` repository, you can start `local-ai` with:
```
GALLERIES=[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
GALLERIES=[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]
```
`https://index.localai.io/models` is a caching mirror of the same index file, and the `github:` entry is the fallback used whenever it cannot be reached. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
LocalAI tries `https://index.localai.io/models` first, GitHub second, and the signed OCI gallery last. The OCI artifact includes the repository-local base configurations, so reading those configurations does not require GitHub. Model weights and external URLs still require their original hosts. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
Note: the url are expanded automatically for `github` and `huggingface`, however `https://` and `http://` prefix works as well.
+14
View File
@@ -340,6 +340,15 @@ curl http://localhost:8080/v1/responses \
}'
```
#### Streaming responses
Set `"stream": true` to receive Server-Sent Events. Each `response.output_item.added` event assigns an `output_index` to an item.
Use that index and the item ID to associate later deltas and completion events with the same item.
If a request without explicit tools produces reasoning, the stream uses separate items for reasoning and answer text.
Each item keeps its original index throughout the stream.
The `response.completed` event includes both items in the same index order, followed by any automatically parsed tool calls.
#### Background Processing
Run requests in the background for long-running tasks:
@@ -434,6 +443,11 @@ curl http://localhost:8080/v1/responses \
}'
```
For streaming requests with JSON tool output, LocalAI waits for the complete JSON
object before emitting a completed `function_call` item. Arguments can span
multiple tokens. Read the arguments from the `response.output_item.done` event
before executing the tool.
#### Reasoning Configuration
Configure reasoning effort and summary style:
+4
View File
@@ -30,6 +30,10 @@ To install the dependencies follow the instructions below:
{{< tabs >}}
{{% tab title="Apple" %}}
To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later.
Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts.
Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables.
Install `xcode` from the App Store
```bash
+35 -4
View File
@@ -63,9 +63,9 @@ against - and two modes:
`proxy.provider` selects the auth scheme and (in translate mode) the wire
format. Supported values: `openai`, `anthropic`.
API keys are loaded from either an environment variable (`api_key_env`) or a
file (`api_key_file`). The key never appears in the config file or the admin
UI; pick whichever fits your secret-management setup.
If the upstream requires an API key, configure either an environment variable
(`api_key_env`) or a file (`api_key_file`). The key never appears in the config
file or the admin UI. If the upstream requires no API key, omit both fields.
### OpenAI passthrough
@@ -129,7 +129,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with
Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak
the OpenAI chat-completions wire format. Use `provider: openai` with the
provider's URL and API key:
provider's URL and, if required, its API key:
```yaml
name: llama-3-70b-via-together
@@ -143,6 +143,37 @@ proxy:
upstream_model: meta-llama/Llama-3-70b-chat-hf
```
### Upstreams without an API key
For an OpenAI-compatible upstream that accepts requests without authentication,
omit both `api_key_env` and `api_key_file`:
```yaml
name: internal-chat-proxy
backend: cloud-proxy
proxy:
mode: passthrough
provider: openai
upstream_url: http://inference.internal:8000/v1/chat/completions
upstream_model: my-model
```
Replace the example URL and model name with your upstream's values. LocalAI
loads this configuration without resolving a key and adds no upstream
`Authorization` header. This also applies to OpenAI-compatible upstreams in
translate mode.
Omitting both fields differs from setting `api_key_env` to an empty or unset
environment variable: the latter causes a backend load error.
LocalAI's client authentication is separate. Clients must still authenticate
to LocalAI when its authentication is enabled. LocalAI does not forward their
`Authorization` header to the upstream.
An upstream without API keys can still require another authentication or
payment protocol. Omitting these fields does not implement that protocol.
### Translate mode
In translate mode the cloud-proxy backend converts LocalAI's internal proto
+4 -2
View File
@@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things:
NVML work inside the container. With `--gpus all` alone (or
`--runtime nvidia` without extra flags) only `compute` is wired in on
some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility`
to your `docker run`, or `capabilities: [gpu, utility]` in compose /
Kubernetes device reservations.
to your `docker run`. For Docker Compose with `driver: nvidia`, use
`capabilities: [gpu, compute, utility]` on the device reservation.
Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility`
alone only provides monitoring libraries and tools.
2. Pass `--init` to `docker run` (or `init: true` in compose) so the
container has a proper PID 1 reaper - otherwise short-lived child
processes like `nvidia-smi` can intermittently fail with
+24
View File
@@ -28,6 +28,29 @@ Returns available backends and currently loaded models.
| `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM |
| `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against |
| `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) |
| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes |
### Per-model VRAM
On Linux, `size_vram` reports resident device memory for the local backend
process and its child processes. LocalAI reads `drm-resident-local*` and
`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU.
Host-memory regions are excluded. The reading includes buffers attributed
to the backend, without separating weights, KV cache, and other allocations.
See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html)
for these counters.
The field is omitted when accounting is unavailable or incomplete. This
includes external and distributed backends, macOS, proprietary NVIDIA
drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters,
and unreadable process information.
A present value of `0` means the supported counters report zero bytes.
Treat an absent field as unknown.
This is a snapshot of driver accounting, not a memory reservation. Shared
buffers can appear in different clients' counters, and allocations can change
during collection. Do not treat the sum across models as exclusive physical
GPU usage. These readings do not replace capacity checks when scheduling work.
### Usage
@@ -49,6 +72,7 @@ curl http://localhost:8080/system
{
"id": "my-llama-model",
"backend": "llama-cpp",
"size_vram": 5368709120,
"process": {
"pid": 48213,
"rss_bytes": 5368709120,
+881 -1
View File
@@ -1,4 +1,96 @@
---
- name: "ternary-bonsai-2-27b"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
- https://github.com/PrismML-Eng/llama.cpp
description: |
Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary
transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits
per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a
Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's
llama.cpp fork) instead of stock llama.cpp.
license: "apache-2.0"
tags:
- llm
- gguf
- reasoning
- vision
- multimodal
icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg
overrides:
backend: bonsai
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf
- filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
- name: "swift-qwen3.8-27b"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/ukisai/Swift-Qwen3.8-27b
- https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF
description: |
Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B.
The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss.
This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding.
The weights use the Swift Open License v1.0.
license: "swift-open-license-1.0"
tags:
- llm
- gguf
- reasoning
- vision
- multimodal
- mtp
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
- spec_n_max:6
- spec_p_min:0.75
parameters:
min_p: 0
model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
presence_penalty: 1.5
repeat_penalty: 1
temperature: 0.7
top_k: 20
top_p: 0.8
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf
- filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf
- name: "ornith-1.5-9b-uncensored"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
@@ -205,7 +297,101 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7
sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
- name: "qwopus3.8-27b-flash-v2"
variants:
- model: qwopus3.8-27b-flash-v2-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
description: |
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
workloads. This Q4_K_M GGUF includes the F32 vision projector and uses
llama.cpp's embedded chat template with MTP speculative decoding.
license: "apache-2.0"
tags:
- llm
- gguf
- qwen
- qwen3
- vision
- multimodal
- instruction-tuned
- reasoning
- mtp
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
- spec_n_max:6
- spec_p_min:0.75
parameters:
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
- name: "qwopus3.8-27b-flash-v2-q8"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
description: |
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
workloads. This Q8_0 GGUF includes the F32 vision projector and uses
llama.cpp's embedded chat template with MTP speculative decoding.
license: "apache-2.0"
tags:
- llm
- gguf
- qwen
- qwen3
- vision
- multimodal
- instruction-tuned
- reasoning
- mtp
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
- spec_n_max:6
- spec_p_min:0.75
parameters:
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
- name: "qwopus3.8-27b-flash"
variants:
- model: qwopus3.8-27b-flash-q8
@@ -302,6 +488,188 @@
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-MTP-Q4_K_M/mmproj-F32.gguf
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-GGUF/resolve/e146d61e88782677805b3b68ad3adf8674dde80d/mmproj-F32.gguf
sha256: 52e6818e4d18eea010c50e5245eaa10a8cc3dcc30efea4ff60cbad8abf5669e1
- name: mimo-v2.6-distill-qwen-9b
variants:
- model: mimo-v2.6-distill-qwen-9b-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B
- https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF
description: |
MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding.
This Q4_K_M GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector.
license: mit
tags:
- llm
- gguf
- cpu
- gpu
- coding
- vision
- multimodal
last_checked: "2026-09-26"
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
options:
- use_jinja:true
template:
use_tokenizer_template: true
parameters:
model: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
files:
- filename: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
sha256: 4bca6f18c73f72270c7a20c2ea2bea581de8246e318714277120369d34048c81
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
- name: mimo-v2.6-distill-qwen-9b-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B
- https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF
description: |
MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding.
This Q8_0 GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector.
license: mit
tags:
- llm
- gguf
- cpu
- gpu
- coding
- vision
- multimodal
last_checked: "2026-09-26"
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
options:
- use_jinja:true
template:
use_tokenizer_template: true
parameters:
model: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
files:
- filename: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
sha256: 2fad0aa11bb9e7aa491ff12f768954f9dd0a6e7d4ce4a897ca73ec420f3b90ae
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
- name: thinkingcap-qwen3.8-27b
variants:
- model: thinkingcap-qwen3.8-27b-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
description: |
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
license: polyform-small-business-1.0.0
tags:
- llm
- gguf
- cpu
- gpu
- vision
- multimodal
- reasoning
last_checked: "2026-09-27"
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
options:
- use_jinja:true
template:
use_tokenizer_template: true
parameters:
model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
files:
- filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
- name: thinkingcap-qwen3.8-27b-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
description: |
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
license: polyform-small-business-1.0.0
tags:
- llm
- gguf
- cpu
- gpu
- vision
- multimodal
- reasoning
last_checked: "2026-09-27"
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
options:
- use_jinja:true
template:
use_tokenizer_template: true
parameters:
model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
files:
- filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
- name: hemmingway-1
variants:
- model: hemmingway-1-q8
@@ -4681,6 +5049,110 @@
- filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf
uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf
sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41
- name: cyber-tiel-coder-35b-a3b-q4-mtp
variants:
- model: cyber-tiel-coder-35b-a3b-q8-mtp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
license: mit
urls:
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
description: |
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
tags:
- llm
- gguf
- cpu
- gpu
- qwen
- moe
- coding
- tools
- vision
- multimodal
- mtp
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
parameters:
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
temperature: 0.6
top_p: 0.95
top_k: 20
min_p: 0.0
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
- name: cyber-tiel-coder-35b-a3b-q8-mtp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
license: mit
urls:
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
description: |
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
tags:
- llm
- gguf
- cpu
- gpu
- qwen
- moe
- coding
- tools
- vision
- multimodal
- mtp
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
parameters:
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
temperature: 0.6
top_p: 0.95
top_k: 20
min_p: 0.0
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
- &tiel-coder-35b-a3b
name: "tiel-coder-35b-a3b-q4"
variants:
@@ -5161,6 +5633,104 @@
- filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf
uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf
sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545
- name: qwen3.8-27b-agention-iq4-xs
url: github:mudler/LocalAI/gallery/virtual.yaml@master
variants:
- model: qwen3.8-27b-agention-q4-k-m
urls:
- https://huggingface.co/Qwen/Qwen3.8-27B
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
license: apache-2.0
description: |
Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp.
This 27B reasoning model supports text and image input. The download
includes the BF16 vision projector and uses the embedded chat template.
tags:
- llm
- gguf
- cpu
- gpu
- qwen
- reasoning
- vision
- multimodal
overrides:
backend: llama-cpp
context_size: 32768
known_usecases:
- chat
- vision
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
temperature: 1
top_p: 0.95
top_k: 20
min_p: 0
repeat_penalty: 1
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf
sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
- name: qwen3.8-27b-agention-q4-k-m
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/Qwen/Qwen3.8-27B
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
license: apache-2.0
description: |
Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp.
This 27B reasoning model supports text and image input. The download
includes the BF16 vision projector and uses the embedded chat template.
tags:
- llm
- gguf
- cpu
- gpu
- qwen
- reasoning
- vision
- multimodal
overrides:
backend: llama-cpp
context_size: 32768
known_usecases:
- chat
- vision
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
temperature: 1
top_p: 0.95
top_k: 20
min_p: 0
repeat_penalty: 1
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf
sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
- &qwen3-8-27b
name: "qwen3.8-27b-q4"
variants:
@@ -5447,6 +6017,178 @@
- filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf
uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf
sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2
- name: "swift-1.5-qwen3.8-27b-gsq-rco"
variants:
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
license: "swift-open-license-1.0"
description: |
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
Text chat only; the publisher provides no verified vision projector for this release.
The weights use the Swift Open License v1.0.
tags:
- llm
- gguf
- cpu
- gpu
- reasoning
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
options:
- use_jinja:true
parameters:
min_p: 0
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
presence_penalty: 0
repeat_penalty: 1
temperature: 1
top_k: 20
top_p: 0.95
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
license: "swift-open-license-1.0"
description: |
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
Text chat only; the publisher provides no verified vision projector for this release.
The weights use the Swift Open License v1.0.
tags:
- llm
- gguf
- cpu
- gpu
- reasoning
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
options:
- use_jinja:true
parameters:
min_p: 0
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
presence_penalty: 0
repeat_penalty: 1
temperature: 1
top_k: 20
top_p: 0.95
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
license: "swift-open-license-1.0"
description: |
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
Text chat only; the publisher provides no verified vision projector for this release.
The weights use the Swift Open License v1.0.
tags:
- llm
- gguf
- cpu
- gpu
- reasoning
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
options:
- use_jinja:true
parameters:
min_p: 0
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
presence_penalty: 0
repeat_penalty: 1
temperature: 1
top_k: 20
top_p: 0.95
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
license: "swift-open-license-1.0"
description: |
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
Text chat only; the publisher provides no verified vision projector for this release.
The weights use the Swift Open License v1.0.
tags:
- llm
- gguf
- cpu
- gpu
- reasoning
overrides:
backend: llama-cpp
context_size: 32768
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
options:
- use_jinja:true
parameters:
min_p: 0
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
presence_penalty: 0
repeat_penalty: 1
temperature: 1
top_k: 20
top_p: 0.95
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
- !!merge <<: *qwen3-8-27b
name: "qwen3.8-27b-gsq-rco-iq2-xs"
variants: []
@@ -5639,6 +6381,120 @@
- filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf
uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf
sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442
- name: sharp-spark-x2.5-4b
url: github:mudler/LocalAI/gallery/virtual.yaml@master
variants:
- model: sharp-spark-x2.5-4b-q5
- model: sharp-spark-x2.5-4b-q6
urls:
- https://huggingface.co/XHToken/Spark-X2.5-4B
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
description: |
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
with an adjusted chat template for coding. This Q4_K_XL build uses the
embedded Sharp-Spark template and a 32K-token default context.
license: apache-2.0
tags:
- llm
- gguf
- cpu
- gpu
- coding
- reasoning
last_checked: "2026-09-26"
overrides:
backend: llama-cpp
context_size: 32768
known_usecases:
- chat
options:
- use_jinja:true
parameters:
model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
temperature: 0.6
top_p: 0.95
top_k: 20
template:
use_tokenizer_template: true
files:
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
- name: sharp-spark-x2.5-4b-q5
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/XHToken/Spark-X2.5-4B
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
description: |
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
with an adjusted chat template for coding. This Q5_K_XL build uses the
embedded Sharp-Spark template and a 32K-token default context.
license: apache-2.0
tags:
- llm
- gguf
- cpu
- gpu
- coding
- reasoning
last_checked: "2026-09-26"
overrides:
backend: llama-cpp
context_size: 32768
known_usecases:
- chat
options:
- use_jinja:true
parameters:
model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
temperature: 0.6
top_p: 0.95
top_k: 20
template:
use_tokenizer_template: true
files:
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
- name: sharp-spark-x2.5-4b-q6
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/XHToken/Spark-X2.5-4B
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
description: |
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
with an adjusted chat template for coding. This Q6_K_XL build uses the
embedded Sharp-Spark template and a 32K-token default context.
license: apache-2.0
tags:
- llm
- gguf
- cpu
- gpu
- coding
- reasoning
last_checked: "2026-09-26"
overrides:
backend: llama-cpp
context_size: 32768
known_usecases:
- chat
options:
- use_jinja:true
parameters:
model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
temperature: 0.6
top_p: 0.95
top_k: 20
template:
use_tokenizer_template: true
files:
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
- &spark-x2-5-4b
name: "spark-x2.5-4b-q4"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
@@ -45126,6 +45982,12 @@
sha256: ""
uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors
- name: kimodo-soma-rp
variants:
- model: kimodo-soma-rp-bf16
- model: kimodo-soma-rp-q4_k
- model: kimodo-soma-rp-q4_k_m
- model: kimodo-soma-rp-q5_k
- model: kimodo-soma-rp-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45313,6 +46175,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-soma-seed
variants:
- model: kimodo-soma-seed-bf16
- model: kimodo-soma-seed-q4_k
- model: kimodo-soma-seed-q4_k_m
- model: kimodo-soma-seed-q5_k
- model: kimodo-soma-seed-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45500,6 +46368,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-g1-rp
variants:
- model: kimodo-g1-rp-bf16
- model: kimodo-g1-rp-q4_k
- model: kimodo-g1-rp-q4_k_m
- model: kimodo-g1-rp-q5_k
- model: kimodo-g1-rp-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
@@ -45687,6 +46561,12 @@
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
- name: kimodo-g1-seed
variants:
- model: kimodo-g1-seed-bf16
- model: kimodo-g1-seed-q4_k
- model: kimodo-g1-seed-q4_k_m
- model: kimodo-g1-seed-q5_k
- model: kimodo-g1-seed-q6_k
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
backend: kimodocpp
urls:
+2 -2
View File
@@ -259,7 +259,7 @@ require (
github.com/kevinburke/ssh_config v1.2.0 // indirect
github.com/labstack/gommon v0.4.2 // indirect
github.com/mschoch/smat v0.2.0 // indirect
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e
github.com/mudler/localrecall v0.6.5 // indirect
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145
github.com/olekukonko/tablewriter v0.0.5 // indirect
@@ -535,7 +535,7 @@ require (
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect
golang.org/x/mod v0.36.0 // indirect
golang.org/x/sync v0.20.0
golang.org/x/sys v0.45.0 // indirect
golang.org/x/sys v0.45.0
golang.org/x/term v0.43.0
golang.org/x/text v0.37.0
golang.org/x/tools v0.45.0 // indirect
+2
View File
@@ -1032,6 +1032,8 @@ github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxF
github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c=
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc=
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o=
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q=
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4=
github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU=
+141
View File
@@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
skippedFiles := 0
skippedBytes := int64(0)
tasks := make([]downloader.FileTask, 0, len(snapshot.Files))
// Sibling manifests are read once, before the staging loop, so the
// per-file reuse lookups below never re-read or re-parse them.
siblings := loadSiblingCandidates(modelsPath, spec, layout)
for index, file := range snapshot.Files {
if err := ctx.Err(); err != nil {
return Result{}, err
@@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
skippedBytes += file.Size
continue
}
// Before reaching for the network, consult committed sibling trees for the
// same Source (type+endpoint+repo+revision). A narrower allow_patterns
// request gets a different CacheKey, so committedResult misses even though a
// broader sibling already holds this exact file; reusing it avoids a
// redundant re-download of tens of gigabytes. The match is re-verified
// through verifyDownloadedFile (full SHA-256), never size-only, and a broader
// request can never inherit a narrower sibling's gaps because each file is
// matched individually against the sibling's manifest.
if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok {
manifest.Files[taskIndex] = entry
completedBytes.Add(file.Size)
skippedFiles++
skippedBytes += file.Size
continue
}
nameSum := sha256.Sum256([]byte(file.Path))
blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:]))
blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel))
@@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest
return entry, true
}
// siblingCandidate is one committed sibling artifact tree that shares this
// request's Source (type+endpoint+repo+revision), with its manifest files
// indexed by path.
type siblingCandidate struct {
final string
filesByPath map[string][]ManifestFile
}
// loadSiblingCandidates reads the committed sibling manifest set once, before
// the staging loop. Doing it per file instead would re-read and re-parse every
// sibling manifest for every file — 20 committed siblings and a 300-file
// snapshot means 6000 manifest reads before the first byte is fetched.
//
// The current artifact's own committed tree is excluded: it is either absent
// (the reason materializeLocked is running) or already handled by
// committedResult's exact-key fast path.
func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate {
if spec.Resolved == nil || layout.Final == "" {
return nil
}
siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface")
entries, err := os.ReadDir(siblingsRoot)
if err != nil {
return nil
}
var candidates []siblingCandidate
for _, entry := range entries {
if !entry.IsDir() {
continue
}
siblingFinal := filepath.Join(siblingsRoot, entry.Name())
if siblingFinal == layout.Final {
continue
}
siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json"))
if err != nil {
continue
}
siblingArtifact := siblingManifest.Artifact
if siblingArtifact.Resolved == nil ||
siblingArtifact.Source.Type != spec.Source.Type ||
siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint ||
siblingArtifact.Source.Repo != spec.Source.Repo ||
siblingArtifact.Resolved.Revision != spec.Resolved.Revision {
continue
}
byPath := make(map[string][]ManifestFile, len(siblingManifest.Files))
for _, f := range siblingManifest.Files {
byPath[f.Path] = append(byPath[f.Path], f)
}
candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath})
}
return candidates
}
// reuseFromCommittedSibling looks for a file already committed under a sibling
// artifact tree — same Source (type+endpoint+repo+revision), different
// allow/ignore patterns — and stages it for this writer instead of fetching.
// A narrower allow_patterns request gets a different CacheKey (path.go:62), so
// committedResult misses and materializeLocked would otherwise re-download
// files an already-committed broader sibling already holds.
//
// The match is never size-only: the sibling file is re-hashed through the
// shared verifyDownloadedFile against the current request's SnapshotFile (its
// LFS or git blob OID), so the staged entry is byte-for-byte identical to a
// fresh download. A broader request can never stand in for files a narrower
// sibling lacks, because each requested file is matched individually against
// the sibling's manifest file set. Hard-link keeps the shared models volume
// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the
// kernel cannot hard-link.
func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) {
snapshotRel := path.Join("snapshot", file.Path)
snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel))
for _, sibling := range candidates {
for _, siblingFile := range sibling.filesByPath[file.Path] {
if siblingFile.Size != file.Size {
continue
}
siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path))
verified, err := verifyDownloadedFile(siblingPath, file)
if err != nil {
continue
}
if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil {
return ManifestFile{}, false
}
_ = root.Remove(snapshotRel)
if err := linkOrCopy(siblingPath, snapshotAbs); err != nil {
return ManifestFile{}, false
}
return verified, true
}
}
return ManifestFile{}, false
}
// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only
// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking
// keeps the shared models volume neutral — a narrowed request does not double
// the storage of a broad sibling's files.
func linkOrCopy(src, dst string) error {
if err := os.Link(src, dst); err == nil {
return nil
} else if !errors.Is(err, syscall.EXDEV) {
return err
}
in, err := os.Open(src)
if err != nil {
return err
}
defer func() { _ = in.Close() }()
out, err := os.Create(dst)
if err != nil {
return err
}
if _, err := io.Copy(out, in); err != nil {
_ = out.Close()
_ = os.Remove(dst)
return err
}
return out.Close()
}
func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) {
file, err := os.Open(fileName)
if err != nil {
@@ -0,0 +1,230 @@
package modelartifacts_test
import (
"context"
"crypto/sha256"
"encoding/hex"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"strconv"
"strings"
"sync"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
hfapi "github.com/mudler/LocalAI/pkg/huggingface-api"
"github.com/mudler/LocalAI/pkg/modelartifacts"
)
const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567"
// recordingResolver serves a fixed full file set filtered by each request's
// allow/ignore patterns, so a narrower request genuinely resolves to a strict
// subset of a broader sibling's files. The HTTP server behind it records every
// fetch, which is the signal the sibling-reuse fix is verified through. The
// function under fix is never mocked: a real Manager drives the real staging +
// commit path against this stub collaborator.
type recordingResolver struct {
endpoint string
repo string
files []hfapi.SnapshotFile
server *httptest.Server
mu sync.Mutex
fetched map[string]int
}
func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver {
r := &recordingResolver{
endpoint: "https://huggingface.co",
repo: "owner/repo",
files: files,
fetched: map[string]int{},
}
r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) {
name := strings.TrimPrefix(req.URL.Path, "/file/")
body, ok := contents[name]
if !ok {
w.WriteHeader(http.StatusNotFound)
return
}
r.mu.Lock()
r.fetched[name]++
r.mu.Unlock()
w.Header().Set("Content-Length", strconv.Itoa(len(body)))
_, _ = w.Write(body)
}))
return r
}
func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) {
files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns)
if err != nil {
return hfapi.Snapshot{}, err
}
out := make([]hfapi.SnapshotFile, len(files))
for i, f := range files {
f.URL = r.server.URL + "/file/" + f.Path
out[i] = f
}
return hfapi.Snapshot{
Endpoint: r.endpoint, Repo: r.repo,
RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out,
}, nil
}
func (r *recordingResolver) fetchCount(path string) int {
r.mu.Lock()
defer r.mu.Unlock()
return r.fetched[path]
}
func (r *recordingResolver) resetFetches() {
r.mu.Lock()
defer r.mu.Unlock()
r.fetched = map[string]int{}
}
func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile {
paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"}
files := make([]hfapi.SnapshotFile, 0, len(paths))
for _, p := range paths {
sum := sha256.Sum256(contents[p])
files = append(files, hfapi.SnapshotFile{
Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]),
})
}
return files
}
// The narrow-request case proves the fix for #11047:
// a request with narrower allow_patterns (a strict subset) reuses files an
// already-committed broader sibling holds, hard-linking instead of re-fetching.
//
// On master this is RED: a narrower allow_patterns set hashes to a different
// CacheKey (path.go:62), so committedResult misses and materializeLocked
// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On
// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling,
// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0,
// os.SameFile true).
var _ = Describe("committed sibling reuse", func() {
It("reuses files from a broader committed sibling", func() {
contents := map[string][]byte{
"a/first.bin": []byte("first-file-bytes"),
"b/second.bin": []byte("second-file-bytes-longer"),
"c/third.bin": []byte("third-file"),
}
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
defer resolver.server.Close()
modelsPath := GinkgoT().TempDir()
manager := modelartifacts.NewManager(resolver,
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
// Commit the broad sibling: all three files, fetched from the resolver.
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
}}
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
Expect(err).NotTo(HaveOccurred())
Expect(broad.CacheHit).To(BeFalse())
Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0),
"the broad sibling must have fetched a/first.bin to commit it")
resolver.resetFetches()
// Narrowed request: a strict subset of the broad sibling's file set.
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
AllowPatterns: []string{"a/first.bin"},
}}
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
Expect(err).NotTo(HaveOccurred())
Expect(narrow.CacheHit).To(BeFalse())
// (b) The sibling-present file must NOT be re-fetched: zero fetches. This is
// the assertion that is RED on master (one fetch) and GREEN on the branch.
Expect(resolver.fetchCount("a/first.bin")).To(Equal(0),
"a/first.bin must be reused from the committed broad sibling, not re-fetched")
// (a) The narrowed tree's staged file is the same inode as the broad
// sibling's file (hard-link), not a freshly downloaded second copy. RED on
// master (separate file), GREEN on the branch (hard-link).
broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin")
narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin")
broadInfo, err := os.Stat(broadFile)
Expect(err).NotTo(HaveOccurred())
narrowInfo, err := os.Stat(narrowFile)
Expect(err).NotTo(HaveOccurred())
Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(),
"the narrowed request must hard-link the broad sibling's file rather than store a second copy")
// The reused bytes are intact end to end.
Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"]))
})
// The broader-request case is the manifest file-set guard: a broader request
// against a narrower committed sibling must still fetch the files the sibling
// lacks and commit a complete tree. Sibling-reuse can never serve an incomplete
// model as complete, because each requested file is matched individually against
// the sibling's manifest.
It("fetches files missing from a narrower committed sibling", func() {
contents := map[string][]byte{
"a/first.bin": []byte("first-file-bytes"),
"b/second.bin": []byte("second-file-bytes-longer"),
"c/third.bin": []byte("third-file"),
}
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
defer resolver.server.Close()
modelsPath := GinkgoT().TempDir()
manager := modelartifacts.NewManager(resolver,
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
// Commit a NARROW sibling first: only a/first.bin and b/second.bin.
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
AllowPatterns: []string{"a/first.bin", "b/second.bin"},
}}
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
Expect(err).NotTo(HaveOccurred())
narrowPaths := make([]string, 0, len(narrow.Manifest.Files))
for _, f := range narrow.Manifest.Files {
narrowPaths = append(narrowPaths, f.Path)
}
Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"}))
resolver.resetFetches()
// A BROADER request asks for all three files, including c/third.bin which the
// narrow sibling does not hold.
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
}}
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
Expect(err).NotTo(HaveOccurred())
// The file the narrow sibling lacks MUST be fetched: sibling-reuse must not
// inherit a narrower tree's gaps as if the broad request were complete.
Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0),
"c/third.bin is absent from the narrow sibling and must be fetched, not served as complete")
// The broad tree's manifest file set is exactly the full set — never the
// narrow sibling's subset. This file-set comparison proves no incomplete model
// is ever served as complete via sibling-reuse.
broadPaths := make([]string, 0, len(broad.Manifest.Files))
for _, f := range broad.Manifest.Files {
broadPaths = append(broadPaths, f.Path)
}
Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"}))
// Every file is present on disk with the right bytes after commit.
for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} {
Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))).
To(Equal(contents[p]))
}
})
})
+165
View File
@@ -0,0 +1,165 @@
//go:build linux
// SPDX-License-Identifier: MIT
package xsysinfo
import (
"bufio"
"bytes"
"math"
"os"
"path/filepath"
"strconv"
"strings"
)
// ProcessVRAM reports device-local resident bytes accounted to a process tree
// by DRM. Unsupported or incomplete accounting returns false, not a measured zero.
func ProcessVRAM(pid int) (uint64, bool) {
return processVRAM("/proc", pid)
}
func processVRAM(procRoot string, pid int) (uint64, bool) {
if pid <= 0 {
return 0, false
}
clients := map[string]uint64{}
seen := map[int]bool{}
pending := []int{pid}
for len(pending) > 0 {
current := pending[len(pending)-1]
pending = pending[:len(pending)-1]
if seen[current] {
continue
}
seen[current] = true
base := filepath.Join(procRoot, strconv.Itoa(current))
fds, err := os.ReadDir(filepath.Join(base, "fd"))
if err != nil {
return 0, false
}
for _, fd := range fds {
target, err := os.Readlink(filepath.Join(base, "fd", fd.Name()))
if err != nil {
return 0, false
}
// A mixed DRM/NVIDIA tree cannot provide a complete DRM reading.
if strings.HasPrefix(target, "/dev/nvidia") {
return 0, false
}
if !strings.HasPrefix(target, "/dev/dri/render") {
// Primary nodes can also own allocations. Until their device
// identity is resolved, omitting them would undercount the tree.
if strings.HasPrefix(target, "/dev/dri/") {
return 0, false
}
continue
}
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
// base adds an integer PID, and fd.Name comes from os.ReadDir.
// The kernel supplies these path components, not request input.
data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name()))
if err != nil {
return 0, false
}
client, used, ok := drmResidentClient(data)
if !ok {
return 0, false
}
key := target + ":" + client
// dup() and fork() can expose the same client more than once. The
// snapshot is not atomic; retain its largest observed reading.
clients[key] = max(clients[key], used)
}
// A worker may be spawned by any thread, not just the thread leader.
tasks, err := os.ReadDir(filepath.Join(base, "task"))
if err != nil || len(tasks) == 0 {
return 0, false
}
for _, task := range tasks {
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
// base adds an integer PID, and task.Name comes from os.ReadDir.
// The kernel supplies these path components, not request input.
data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children"))
if err != nil {
return 0, false
}
for _, raw := range strings.Fields(string(data)) {
child, err := strconv.Atoi(raw)
if err != nil || child <= 0 {
return 0, false
}
pending = append(pending, child)
}
}
}
var total uint64
for _, used := range clients {
if used > math.MaxUint64-total {
return 0, false
}
total += used
}
return total, len(clients) > 0
}
func drmResidentClient(data []byte) (string, uint64, bool) {
var client string
var total uint64
found := false
scanner := bufio.NewScanner(bytes.NewReader(data))
for scanner.Scan() {
key, value, ok := strings.Cut(scanner.Text(), ":")
if !ok {
continue
}
if key == "drm-client-id" {
id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64)
if err != nil {
return "", 0, false
}
client = strconv.FormatUint(id, 10)
}
region, resident := strings.CutPrefix(key, "drm-resident-")
if !resident || !isVRAMRegion(region) {
continue
}
used, ok := drmResidentBytes(value)
if !ok || used > math.MaxUint64-total {
return "", 0, false
}
total += used
found = true
}
return client, total, scanner.Err() == nil && client != "" && found
}
func drmResidentBytes(value string) (uint64, bool) {
fields := strings.Fields(value)
if len(fields) == 0 || len(fields) > 2 {
return 0, false
}
n, err := strconv.ParseUint(fields[0], 10, 64)
if err != nil {
return 0, false
}
unit := uint64(1)
if len(fields) == 2 {
switch strings.ToLower(fields[1]) {
case "b":
case "kib":
unit = 1 << 10
case "mib":
unit = 1 << 20
case "gib":
unit = 1 << 30
default:
return 0, false
}
}
if n > math.MaxUint64/unit {
return 0, false
}
return n * unit, true
}
+105
View File
@@ -0,0 +1,105 @@
//go:build linux
// SPDX-License-Identifier: MIT
package xsysinfo
import (
"os"
"path/filepath"
"strconv"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("ProcessVRAM", func() {
var root string
write := func(path, contents string) {
Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed())
Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed())
}
addProcess := func(pid int, children string) {
base := filepath.Join(root, strconv.Itoa(pid))
Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed())
write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children)
}
addFD := func(pid, fd int, render, info string) {
base := filepath.Join(root, strconv.Itoa(pid))
name := strconv.Itoa(fd)
Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed())
write(filepath.Join(base, "fdinfo", name), info)
}
BeforeEach(func() {
var err error
root, err = os.MkdirTemp("", "process-vram-")
Expect(err).NotTo(HaveOccurred())
DeferCleanup(os.RemoveAll, root)
addProcess(100, "")
})
It("sums resident device memory across GPUs and child processes without duplicate clients", func() {
write(filepath.Join(root, "100/task/101/children"), "200")
addProcess(200, "")
info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n"
addFD(100, 3, "renderD128", info)
addFD(100, 4, "renderD128", info)
addFD(200, 3, "renderD128", info)
addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n")
used, ok := processVRAM(root, 100)
Expect(ok).To(BeTrue())
Expect(used).To(Equal(uint64(384 * 1024 * 1024)))
})
It("distinguishes a measured zero from unavailable accounting", func() {
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n")
used, ok := processVRAM(root, 100)
Expect(ok).To(BeTrue())
Expect(used).To(BeZero())
})
DescribeTable("does not invent readings from unsupported or invalid accounting",
func(info string) {
addFD(100, 3, "renderD128", info)
_, ok := processVRAM(root, 100)
Expect(ok).To(BeFalse())
},
Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"),
Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"),
Entry("no client identity", "drm-resident-vram0: 128 MiB\n"),
Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"),
Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"),
Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"),
)
It("omits a partial reading if a child cannot be inspected", func() {
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
write(filepath.Join(root, "100/task/100/children"), "200")
_, ok := processVRAM(root, 100)
Expect(ok).To(BeFalse())
})
It("omits a partial reading if another DRM client lacks accounting", func() {
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
addFD(100, 4, "renderD129", "drm-client-id: 8\n")
_, ok := processVRAM(root, 100)
Expect(ok).To(BeFalse())
})
DescribeTable("omits mixed readings with unsupported GPU descriptors",
func(target string) {
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed())
_, ok := processVRAM(root, 100)
Expect(ok).To(BeFalse())
},
Entry("primary DRM node", "/dev/dri/card0"),
Entry("NVIDIA device", "/dev/nvidia0"),
)
It("returns unavailable for missing processes or no DRM descriptors", func() {
for _, pid := range []int{-1, 0, 100, 999} {
_, ok := processVRAM(root, pid)
Expect(ok).To(BeFalse())
}
})
})
+9
View File
@@ -0,0 +1,9 @@
//go:build !linux
// SPDX-License-Identifier: MIT
package xsysinfo
// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting.
func ProcessVRAM(pid int) (uint64, bool) {
return 0, false
}
+94
View File
@@ -0,0 +1,94 @@
// SPDX-License-Identifier: MIT
// Package the official index and its repository-local base configurations.
package main
import (
"fmt"
"os"
"path/filepath"
"strings"
"gopkg.in/yaml.v3"
)
func main() {
if len(os.Args) != 4 {
fmt.Fprintln(os.Stderr, "usage: gallery REPOSITORY {gallery|backend} OUTPUT")
os.Exit(1)
}
if err := packageGallery(os.Args[1], os.Args[2], os.Args[3]); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
}
func packageGallery(root, source, output string) error {
if source != "gallery" && source != "backend" {
return fmt.Errorf("unsupported gallery directory %q", source)
}
repository, err := os.OpenRoot(root)
if err != nil {
return err
}
defer func() { _ = repository.Close() }()
body, err := repository.ReadFile(filepath.Join(source, "index.yaml"))
if err != nil {
return err
}
var doc yaml.Node
if err := yaml.Unmarshal(body, &doc); err != nil {
return err
}
// The build operator explicitly selects the output directory via the CLI.
if err := os.MkdirAll(output, 0700); err != nil { // #nosec G703 -- caller-selected output root
return err
}
destination, err := os.OpenRoot(output)
if err != nil {
return err
}
defer func() { _ = destination.Close() }()
// Keep the tree relative to the repository root so repeated base configs
// share a layer, even when an index refers outside its own directory.
const prefix = "github:mudler/LocalAI/"
var walk func(*yaml.Node) error
walk = func(n *yaml.Node) error {
if n.Kind == yaml.MappingNode {
for i := 0; i < len(n.Content); i += 2 {
value := n.Content[i+1]
if n.Content[i].Value != "url" || value.Kind != yaml.ScalarNode || !strings.HasPrefix(value.Value, prefix) || !strings.HasSuffix(value.Value, "@master") {
continue
}
path := strings.TrimSuffix(strings.TrimPrefix(value.Value, prefix), "@master")
if !filepath.IsLocal(path) {
return fmt.Errorf("base config escapes repository: %q", path)
}
config, err := repository.ReadFile(path)
if err != nil {
return err
}
if err := destination.MkdirAll(filepath.Dir(path), 0700); err != nil {
return err
}
if err := destination.WriteFile(path, config, 0600); err != nil {
return err
}
value.Value = filepath.ToSlash(path)
}
}
for _, child := range n.Content {
if err := walk(child); err != nil {
return err
}
}
return nil
}
if err := walk(&doc); err != nil {
return err
}
body, err = yaml.Marshal(&doc)
if err != nil {
return err
}
return destination.WriteFile("index.yaml", body, 0600)
}
+83
View File
@@ -0,0 +1,83 @@
// SPDX-License-Identifier: MIT
package main
import (
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"gopkg.in/yaml.v3"
"os"
"path/filepath"
"strings"
"testing"
)
func TestGalleryPackage(t *testing.T) { RegisterFailHandler(Fail); RunSpecs(t, "Gallery packaging") }
var _ = Describe("Gallery packaging", func() {
It("packages both official indexes with every repository-local base available offline", func() {
for _, source := range []string{"gallery", "backend"} {
out := GinkgoT().TempDir()
Expect(packageGallery("../../..", source, out)).To(Succeed())
body, err := os.ReadFile(filepath.Join(out, "index.yaml"))
Expect(err).ToNot(HaveOccurred())
var entries []map[string]any
Expect(yaml.Unmarshal(body, &entries)).To(Succeed())
Expect(entries).ToNot(BeEmpty())
for _, entry := range entries {
url, _ := entry["url"].(string)
Expect(url).ToNot(HavePrefix("github:mudler/LocalAI/"))
if strings.HasPrefix(url, "gallery/") {
Expect(filepath.Join(out, url)).To(BeAnExistingFile())
}
}
}
})
It("bundles local base configs and preserves external URLs and YAML aliases", func() {
root := GinkgoT().TempDir()
Expect(os.MkdirAll(filepath.Join(root, "gallery"), 0755)).To(Succeed())
Expect(os.WriteFile(filepath.Join(root, "gallery/base.yaml"), []byte("backend: llama-cpp\n"), 0644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- &base\n name: first\n url: github:mudler/LocalAI/gallery/base.yaml@master\n- <<: *base\n name: second\n- name: external\n url: https://example.com/config.yaml\n"), 0644)).To(Succeed())
out := filepath.Join(root, "out")
Expect(packageGallery(root, "gallery", out)).To(Succeed())
data, err := os.ReadFile(filepath.Join(out, "index.yaml"))
Expect(err).ToNot(HaveOccurred())
var entries []map[string]any
Expect(yaml.Unmarshal(data, &entries)).To(Succeed())
Expect(entries[0]["url"]).To(Equal("gallery/base.yaml"))
Expect(entries[1]["url"]).To(Equal("gallery/base.yaml"))
Expect(entries[2]["url"]).To(Equal("https://example.com/config.yaml"))
body, err := os.ReadFile(filepath.Join(out, "gallery/base.yaml"))
Expect(err).ToNot(HaveOccurred())
Expect(string(body)).To(Equal("backend: llama-cpp\n"))
})
It("fails if a referenced config is missing or escapes the repository", func() {
for _, ref := range []string{"missing.yaml", "../../outside.yaml"} {
root := GinkgoT().TempDir()
Expect(os.Mkdir(filepath.Join(root, "gallery"), 0755)).To(Succeed())
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- name: broken\n url: github:mudler/LocalAI/gallery/"+ref+"@master\n"), 0644)).To(Succeed())
Expect(packageGallery(root, "gallery", filepath.Join(root, "out"))).ToNot(Succeed())
}
})
It("rejects symlink escapes when reading configs or writing the bundle", func() {
for _, location := range []string{"source", "output"} {
root, out, outside := GinkgoT().TempDir(), GinkgoT().TempDir(), GinkgoT().TempDir()
for _, dir := range []string{filepath.Join(root, "gallery"), filepath.Join(out, "gallery")} {
Expect(os.Mkdir(dir, 0700)).To(Succeed())
}
index := []byte("- name: test\n url: github:mudler/LocalAI/gallery/base.yaml@master\n")
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), index, 0600)).To(Succeed())
outsideFile := filepath.Join(outside, "base.yaml")
Expect(os.WriteFile(outsideFile, []byte("outside"), 0600)).To(Succeed())
link := filepath.Join(root, "gallery/base.yaml")
if location == "output" {
Expect(os.WriteFile(link, []byte("inside"), 0600)).To(Succeed())
link = filepath.Join(out, "gallery/base.yaml")
}
Expect(os.Symlink(outsideFile, link)).To(Succeed())
Expect(packageGallery(root, "gallery", out)).ToNot(Succeed(), location)
data, err := os.ReadFile(outsideFile)
Expect(err).ToNot(HaveOccurred())
Expect(string(data)).To(Equal("outside"))
}
})
})
+7
View File
@@ -7313,6 +7313,9 @@ const docTemplate = `{
"id": {
"type": "string"
},
"metadata": {
"type": "object"
},
"model": {
"type": "string"
},
@@ -7807,6 +7810,10 @@ const docTemplate = `{
},
"id": {
"type": "string"
},
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
+7
View File
@@ -7310,6 +7310,9 @@
"id": {
"type": "string"
},
"metadata": {
"type": "object"
},
"model": {
"type": "string"
},
@@ -7804,6 +7807,10 @@
},
"id": {
"type": "string"
},
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
+7
View File
@@ -2266,6 +2266,8 @@ definitions:
type: array
id:
type: string
metadata:
type: object
model:
type: string
object:
@@ -2652,6 +2654,11 @@ definitions:
type: string
id:
type: string
size_vram:
description: |-
SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
the backend process tree has no complete supported reading.
type: integer
type: object
schema.SystemInformationResponse:
properties:
+4 -4
View File
@@ -3,10 +3,10 @@
# The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh,
# which runs weekly from .github/workflows/refresh-site-counters.yml. Editing
# them by hand works but will be overwritten on the next run.
stars: 48949
forks: 4430
contributors: 237
releases: 135
stars: 49204
forks: 4459
contributors: 245
releases: 136
# The GitHub API cannot answer for this one, so it is maintained by hand and
# the refresh script carries it through untouched.