mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 01:25:03 -04:00
chore: merge master into distributed transport PR
Keep the newer SQLite dependency from master to resolve the conflict. Assisted-by: Codex:gpt-6
This commit is contained in:
commit
733eeda123
102 files changed
+3773
-295
No files matched your search
@@ -355,7 +355,7 @@ jobs:
|
||||
with:
|
||||
backend: ${{ matrix.backend }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
go-version: "1.25.x"
|
||||
go-version: "1.27.x"
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
lang: ${{ matrix.lang || 'python' }}
|
||||
use-pip: ${{ matrix.backend == 'diffusers' }}
|
||||
|
||||
@@ -252,7 +252,8 @@ jobs:
|
||||
name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
# Release matrices and their retries can outlive a one-day artifact.
|
||||
retention-days: 7
|
||||
|
||||
- name: Build (PR)
|
||||
uses: docker/build-push-action@v7
|
||||
|
||||
@@ -22,7 +22,8 @@ on:
|
||||
type: string
|
||||
go-version:
|
||||
description: 'Go version to use'
|
||||
default: '1.24.x'
|
||||
# Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs.
|
||||
default: '1.27.x'
|
||||
type: string
|
||||
tag-suffix:
|
||||
description: 'Tag suffix for the built image'
|
||||
|
||||
@@ -281,7 +281,7 @@ jobs:
|
||||
with:
|
||||
backend: ${{ matrix.backend }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
go-version: "1.25.x"
|
||||
go-version: "1.27.x"
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
lang: ${{ matrix.lang || 'python' }}
|
||||
use-pip: ${{ matrix.backend == 'diffusers' }}
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: Publish official OCI galleries
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [master]
|
||||
paths:
|
||||
- 'gallery/**'
|
||||
- 'backend/index.yaml'
|
||||
- 'scripts/build/gallery/**'
|
||||
- '.github/workflows/gallery_publish.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: publish-official-galleries
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
env:
|
||||
COSIGN_EXPERIMENTAL: '1'
|
||||
GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- source: gallery
|
||||
tag: gallery-models
|
||||
- source: backend
|
||||
tag: gallery-backends
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Test and package gallery
|
||||
env:
|
||||
GALLERY_SOURCE: ${{ matrix.source }}
|
||||
run: |
|
||||
go test ./scripts/build/gallery -count=1
|
||||
go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery"
|
||||
- uses: oras-project/setup-oras@v1
|
||||
with:
|
||||
version: '1.3.0'
|
||||
- uses: sigstore/cosign-installer@v3
|
||||
with:
|
||||
cosign-release: 'v2.6.5'
|
||||
- name: Login to Quay.io
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
registry: quay.io
|
||||
username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
- name: Publish and sign gallery
|
||||
shell: bash
|
||||
env:
|
||||
GALLERY_TAG: ${{ matrix.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
cd "$RUNNER_TEMP/gallery"
|
||||
files=()
|
||||
while IFS= read -r -d '' file; do
|
||||
files+=("${file#./}:application/yaml")
|
||||
done < <(find . -type f -print0 | sort -z)
|
||||
# Publish an immutable revision, then expose latest only after signing.
|
||||
ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA"
|
||||
oras push --artifact-type application/vnd.localai.gallery.v1 \
|
||||
--format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json"
|
||||
digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json")
|
||||
cosign sign --yes --new-bundle-format \
|
||||
--registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest"
|
||||
oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG"
|
||||
@@ -9,7 +9,7 @@
|
||||
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
|
||||
# rebuild and so the bump bot can see the pin.
|
||||
|
||||
AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22
|
||||
AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
|
||||
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
|
||||
|
||||
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc
|
||||
IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
|
||||
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4
|
||||
LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234
|
||||
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
|
||||
# Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant.
|
||||
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
|
||||
TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64
|
||||
TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d
|
||||
LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,52 +0,0 @@
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
index 680fd12..ffd6604 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
|
||||
index 5c614a9..d765cfc 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn.cu
|
||||
@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16)
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0)
|
||||
|
||||
#ifdef GGML_CUDA_FA_ALL_QUANTS
|
||||
FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16)
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
index a93be56..3630d87 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
index 3c806c2..c8a4d9f 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
index 180902f..1646ef0 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# CrispASR version (release tag)
|
||||
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
|
||||
CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee
|
||||
CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
|
||||
SO_TARGET?=libgocrispasr.so
|
||||
|
||||
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# parakeet-cpp backend Makefile.
|
||||
#
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
|
||||
# (.github/bump_deps.sh) can find and update it - matches the
|
||||
# whisper.cpp / ds4 / vibevoice-cpp convention.
|
||||
#
|
||||
@@ -15,7 +15,7 @@
|
||||
# That's what the L0 smoke test uses. The default target below does the
|
||||
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
|
||||
|
||||
PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
|
||||
PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
|
||||
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
|
||||
|
||||
GOCMD?=go
|
||||
|
||||
@@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e
|
||||
|
||||
# vllm.cpp version
|
||||
VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp
|
||||
VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb
|
||||
VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c
|
||||
|
||||
# MLX GEMM provider (darwin/metal only; see the metal branch below for why).
|
||||
# Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
grpcio==1.83.1
|
||||
grpcio==1.84.0
|
||||
protobuf
|
||||
certifi
|
||||
packaging==26.3
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
accelerate
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@
|
||||
torch==2.9.0
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -1,11 +1,11 @@
|
||||
--extra-index-url https://download.pytorch.org/whl/rocm7.0
|
||||
torch==2.10.0+rocm7.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -3,9 +3,9 @@ torch
|
||||
optimum[openvino]
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -1,6 +1,6 @@
|
||||
grpcio==1.83.0
|
||||
grpcio==1.84.0
|
||||
protobuf==7.36.1
|
||||
certifi
|
||||
setuptools
|
||||
scipy==1.18.0
|
||||
numpy>=2.5.2
|
||||
numpy>=2.5.3
|
||||
@@ -132,6 +132,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: "Kokoros TTS model loaded".into(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -180,11 +181,13 @@ impl Backend for KokorosService {
|
||||
return Ok(Response::new(backend::Result {
|
||||
success: false,
|
||||
message: format!("Failed to write WAV: {}", e),
|
||||
..Default::default()
|
||||
}));
|
||||
}
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: String::new(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
Err(e) => {
|
||||
@@ -192,6 +195,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: false,
|
||||
message: format!("TTS error: {}", e),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
}
|
||||
@@ -292,6 +296,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: "Model freed".into(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
|
||||
|
||||
@@ -11,8 +11,9 @@ import (
|
||||
)
|
||||
|
||||
// BackendAdmissionError reports that the process-wide backend execution
|
||||
// ceiling is full. HTTP callers map it to 503; internal callers receive the
|
||||
// same typed error instead of silently queueing and growing in-flight state.
|
||||
// ceiling is full. HTTP callers map it to 429 (Too Many Requests) with a
|
||||
// Retry-After header; internal callers receive the same typed error instead
|
||||
// of silently queueing and growing in-flight state.
|
||||
type BackendAdmissionError struct {
|
||||
Limit int
|
||||
RetryAfter time.Duration
|
||||
|
||||
+11
-1
@@ -47,10 +47,13 @@ type Gallery struct {
|
||||
// fallback for availability, not a load-balancing pool: the primary is
|
||||
// always preferred, and a mirror is only consulted after the one before
|
||||
// it fails. Any URI the gallery loader understands works here
|
||||
// (https://, github:, file://).
|
||||
// (https://, github:, file://, oci://).
|
||||
Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"`
|
||||
Name string `json:"name" yaml:"name"`
|
||||
Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"`
|
||||
// ArtifactVerification overrides Verification only for the gallery OCI artifact.
|
||||
// Backend images keep their separate Verification policy.
|
||||
ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"`
|
||||
}
|
||||
|
||||
// Equal reports whether two gallery entries describe the same gallery.
|
||||
@@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool {
|
||||
if !slices.Equal(g.Mirrors, other.Mirrors) {
|
||||
return false
|
||||
}
|
||||
if g.ArtifactVerification == nil || other.ArtifactVerification == nil {
|
||||
if g.ArtifactVerification != other.ArtifactVerification {
|
||||
return false
|
||||
}
|
||||
} else if *g.ArtifactVerification != *other.ArtifactVerification {
|
||||
return false
|
||||
}
|
||||
if g.Verification == nil || other.Verification == nil {
|
||||
return g.Verification == other.Verification
|
||||
}
|
||||
|
||||
@@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() {
|
||||
Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery"))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("Gallery artifact verification", func() {
|
||||
It("compares artifact policies by value and preserves them in JSON and YAML", func() {
|
||||
a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
|
||||
b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
b.ArtifactVerification.Identity = "another-workflow"
|
||||
Expect(a.Equal(b)).To(BeFalse())
|
||||
b.ArtifactVerification = nil
|
||||
Expect(a.Equal(b)).To(BeFalse())
|
||||
raw, err := json.Marshal(a)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(json.Unmarshal(raw, &b)).To(Succeed())
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
raw, err = yaml.Marshal(a)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
b = config.Gallery{}
|
||||
Expect(yaml.Unmarshal(raw, &b)).To(Succeed())
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
@@ -17,8 +17,8 @@ import (
|
||||
// a caching mirror of the files below. The GitHub URI stays as a mirror so an
|
||||
// install still resolves its gallery unchanged whenever the primary is
|
||||
// unreachable - see the fallback chain in core/gallery/gallery_mirrors.go.
|
||||
const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]`
|
||||
const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]`
|
||||
const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
|
||||
const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
|
||||
|
||||
func mustGalleries(jsonList string) []Gallery {
|
||||
var g []Gallery
|
||||
|
||||
@@ -10,22 +10,22 @@ import (
|
||||
)
|
||||
|
||||
var _ = Describe("default galleries", func() {
|
||||
It("serves the model gallery from index.localai.io with GitHub as a mirror", func() {
|
||||
It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() {
|
||||
var galleries []config.Gallery
|
||||
Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed())
|
||||
Expect(galleries).To(HaveLen(1))
|
||||
Expect(galleries[0].Name).To(Equal("localai"))
|
||||
Expect(galleries[0].URL).To(Equal("https://index.localai.io/models"))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"}))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"}))
|
||||
})
|
||||
|
||||
It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() {
|
||||
It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() {
|
||||
var galleries []config.Gallery
|
||||
Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed())
|
||||
Expect(galleries).To(HaveLen(1))
|
||||
Expect(galleries[0].Name).To(Equal("localai"))
|
||||
Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends"))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"}))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"}))
|
||||
})
|
||||
|
||||
// The mirror is the whole reason this default is safe to ship: if
|
||||
@@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() {
|
||||
Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed())
|
||||
for _, g := range galleries {
|
||||
Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name)
|
||||
Expect(g.ArtifactVerification).ToNot(BeNil())
|
||||
Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"))
|
||||
Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com"))
|
||||
Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust")
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
@@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string {
|
||||
if !looksLikeOCIGallery(candidate) {
|
||||
continue
|
||||
}
|
||||
dir := ociGalleryCacheDir(basePath, candidate, g.Verification)
|
||||
dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g))
|
||||
if dir == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -646,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]()
|
||||
// would also point relative entry urls at an unpacked tree the new policy has
|
||||
// not produced yet, so they could not be installed.
|
||||
func galleryIndexCacheKey(g config.Gallery) string {
|
||||
return g.Name + "-" + galleryCacheName(g.URL, g.Verification)
|
||||
return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g))
|
||||
}
|
||||
|
||||
func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) {
|
||||
|
||||
@@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
|
||||
if !looksLikeOCIGallery(g.URL) {
|
||||
return nil
|
||||
}
|
||||
return g.Verification
|
||||
return galleryArtifactPolicy(g)
|
||||
}
|
||||
|
||||
// verifiableCandidates drops the candidates that cannot answer for a signed
|
||||
@@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
|
||||
// at, and after a refusal it would turn "this artifact is not trusted" into
|
||||
// "use this other, unchecked copy instead".
|
||||
func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string {
|
||||
if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) {
|
||||
if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) {
|
||||
return candidates
|
||||
}
|
||||
out := make([]string, 0, len(candidates))
|
||||
|
||||
@@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) {
|
||||
// later fetch served would hand the user a truncated gallery with no sign that
|
||||
// anything went wrong.
|
||||
func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) {
|
||||
policy := galleryArtifactPolicy(g)
|
||||
// Checked before the cache: a copy unpacked while strict integrity was
|
||||
// off was never verified, and turning strict integrity on must not keep
|
||||
// serving it for the rest of its TTL.
|
||||
if g.Verification == nil && requireIntegrity {
|
||||
if policy == nil && requireIntegrity {
|
||||
return nil, &galleryVerificationError{
|
||||
strict: true,
|
||||
err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
|
||||
err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
|
||||
}
|
||||
}
|
||||
|
||||
cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification)
|
||||
cacheDir := ociGalleryCacheDir(basePath, candidate, policy)
|
||||
if cacheDir == "" {
|
||||
return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate)
|
||||
}
|
||||
@@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
|
||||
pullRef := downloader.URI(candidate).OCIReference()
|
||||
|
||||
if g.Verification != nil {
|
||||
if policy != nil {
|
||||
// Resolve first, verify the digest, then pull that same digest.
|
||||
// Nothing has been fetched at this point beyond the manifest, so a
|
||||
// policy failure leaves no content anywhere.
|
||||
@@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil {
|
||||
if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil {
|
||||
// Only a decision about the artifact is a refusal. The
|
||||
// verifier also reaches the Sigstore TUF mirror and the
|
||||
// registry, and a timeout or a 5xx there says nothing about
|
||||
@@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
|
||||
return body, nil
|
||||
}
|
||||
|
||||
func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification {
|
||||
if g.ArtifactVerification != nil {
|
||||
return g.ArtifactVerification
|
||||
}
|
||||
return g.Verification
|
||||
}
|
||||
@@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() {
|
||||
})
|
||||
})
|
||||
|
||||
It("uses the artifact policy without replacing backend image verification", func() {
|
||||
srv, _, _ := ociRegistry()
|
||||
url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}})
|
||||
backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"}
|
||||
artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"}
|
||||
var seen *config.GalleryVerification
|
||||
stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil })
|
||||
g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy}
|
||||
_, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true)
|
||||
Expect(source).To(Equal(url))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(seen).To(Equal(artifactPolicy))
|
||||
Expect(g.Verification).To(Equal(backendPolicy))
|
||||
})
|
||||
|
||||
It("refuses an unsigned gallery in strict integrity mode", func() {
|
||||
srv, _, blobs := ociRegistry()
|
||||
url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
package http
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"time"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
corebackend "github.com/mudler/LocalAI/core/backend"
|
||||
"github.com/mudler/LocalAI/core/services/nodes"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Backend admission", func() {
|
||||
It("maps BackendAdmissionError to 429 with Retry-After", func() {
|
||||
e := echo.New()
|
||||
req := httptest.NewRequest(http.MethodPost, "/", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
err := &corebackend.BackendAdmissionError{Limit: 4, RetryAfter: 3 * time.Second}
|
||||
code := applyBackendAdmission(err, http.StatusInternalServerError, c)
|
||||
|
||||
Expect(code).To(Equal(http.StatusTooManyRequests))
|
||||
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
|
||||
})
|
||||
|
||||
It("passes through non-admission errors unchanged", func() {
|
||||
e := echo.New()
|
||||
req := httptest.NewRequest(http.MethodPost, "/", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
code := applyBackendAdmission(errors.New("some other error"), http.StatusInternalServerError, c)
|
||||
Expect(code).To(Equal(http.StatusInternalServerError))
|
||||
Expect(rec.Header().Get("Retry-After")).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("No available nodes", func() {
|
||||
It("maps ErrNoAvailableNodes to 503", func() {
|
||||
// The scheduler wraps the sentinel in fmt.Errorf chains and via
|
||||
// errors.Join — errors.Is must still find it.
|
||||
wrapped := fmt.Errorf("routing model foo: %w",
|
||||
fmt.Errorf("no available nodes: %w",
|
||||
fmt.Errorf("no healthy nodes available: %w",
|
||||
errors.Join(nodes.ErrEvictionBusy, nodes.ErrNoAvailableNodes))))
|
||||
|
||||
code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusServiceUnavailable))
|
||||
})
|
||||
|
||||
It("maps selector-mismatch chain to 503", func() {
|
||||
wrapped := fmt.Errorf("routing model bar: %w",
|
||||
fmt.Errorf("no available nodes: %w",
|
||||
fmt.Errorf("no healthy nodes match selector for model bar: {\"gpu.vendor\":\"tpu\"}: %w",
|
||||
nodes.ErrNoAvailableNodes)))
|
||||
|
||||
code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusServiceUnavailable))
|
||||
})
|
||||
|
||||
It("passes through unrelated errors unchanged", func() {
|
||||
code := applyNoAvailableNodes(errors.New("database timeout"), http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusInternalServerError))
|
||||
})
|
||||
})
|
||||
+22
-2
@@ -88,7 +88,20 @@ func applyBackendAdmission(err error, code int, c echo.Context) int {
|
||||
return code
|
||||
}
|
||||
c.Response().Header().Set("Retry-After", strconv.Itoa(int(capacityErr.RetryAfter.Seconds())))
|
||||
return http.StatusServiceUnavailable
|
||||
return http.StatusTooManyRequests
|
||||
}
|
||||
|
||||
// applyNoAvailableNodes maps scheduler "no available nodes" errors to 503.
|
||||
// When the cluster has no healthy node to serve a model — all are full, a
|
||||
// node selector excludes every candidate, or eviction could not free a slot —
|
||||
// the request is retryable, not a server bug. Without this the error fell
|
||||
// through to 500, which tells clients something is broken when they just
|
||||
// need to wait for a node.
|
||||
func applyNoAvailableNodes(err error, code int) int {
|
||||
if errors.Is(err, nodes.ErrNoAvailableNodes) {
|
||||
return http.StatusServiceUnavailable
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
// respondModelLoading answers a request whose model is still cold-loading with
|
||||
@@ -211,6 +224,7 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
}
|
||||
code = applyModelLoadCooldown(err, code, c)
|
||||
code = applyBackendAdmission(err, code, c)
|
||||
code = applyNoAvailableNodes(err, code)
|
||||
|
||||
// Handle 404 errors: serve React SPA for HTML requests, JSON otherwise
|
||||
if code == http.StatusNotFound {
|
||||
@@ -227,8 +241,13 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
}
|
||||
|
||||
// Send custom error page
|
||||
errType := ""
|
||||
var capErr *corebackend.BackendAdmissionError
|
||||
if errors.As(err, &capErr) {
|
||||
errType = "rate_limit_error"
|
||||
}
|
||||
c.JSON(code, schema.ErrorResponse{
|
||||
Error: &schema.APIError{Message: err.Error(), Code: code},
|
||||
Error: &schema.APIError{Message: err.Error(), Code: code, Type: errType},
|
||||
})
|
||||
}
|
||||
} else {
|
||||
@@ -243,6 +262,7 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
// Opaque errors deliberately withhold the body, so a still-loading
|
||||
// model gets the status and Retry-After but no progress detail.
|
||||
code = applyModelLoading(err, code, c)
|
||||
code = applyNoAvailableNodes(err, code)
|
||||
c.NoContent(code)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
"github.com/mudler/LocalAI/core/services/monitoring"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/xsysinfo"
|
||||
)
|
||||
|
||||
// SystemInformations returns the system informations
|
||||
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
|
||||
entry.Process = proc
|
||||
}
|
||||
}
|
||||
if pid, ok := localPID(m); ok {
|
||||
if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
|
||||
entry.SizeVRAM = &used
|
||||
}
|
||||
}
|
||||
sysmodels = append(sysmodels, entry)
|
||||
}
|
||||
if sampler != nil {
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package localai_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/http/endpoints/localai"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
process "github.com/mudler/go-processmanager"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SystemInformations memory", func() {
|
||||
It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
|
||||
path, err := os.MkdirTemp("", "system-info-")
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
DeferCleanup(os.RemoveAll, path)
|
||||
configFile := filepath.Join(path, "remote.yaml")
|
||||
Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
|
||||
cl := config.NewModelConfigLoader(path)
|
||||
Expect(cl.ReadModelConfig(configFile)).To(Succeed())
|
||||
ml := model.NewModelLoader(&system.SystemState{})
|
||||
store := model.NewInMemoryModelStore()
|
||||
store.Set("remote", model.NewModel("remote", "worker:50051", nil))
|
||||
store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
|
||||
ml.SetModelStore(store)
|
||||
app := echo.New()
|
||||
app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
|
||||
rec := httptest.NewRecorder()
|
||||
app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
var response struct {
|
||||
Models []map[string]any `json:"loaded_models"`
|
||||
}
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
|
||||
Expect(response.Models).To(ConsistOf(
|
||||
map[string]any{"id": "remote", "backend": "llama-cpp"},
|
||||
map[string]any{"id": "stopped"},
|
||||
))
|
||||
})
|
||||
})
|
||||
@@ -3,6 +3,8 @@ package ollama
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -37,7 +39,7 @@ func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) ec
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
ModifiedAt: time.Now().UTC(),
|
||||
Size: 0,
|
||||
Size: modelOnDiskSize(bcl, ml, name),
|
||||
Digest: digest,
|
||||
Details: details,
|
||||
Capabilities: caps,
|
||||
@@ -101,13 +103,15 @@ func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) e
|
||||
|
||||
details, caps := modelMetaFromConfig(bcl, name)
|
||||
entry := schema.OllamaPsEntry{
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
Size: 0,
|
||||
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
|
||||
Details: details,
|
||||
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
|
||||
SizeVRAM: 0,
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
Size: modelOnDiskSize(bcl, ml, name),
|
||||
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
|
||||
Details: details,
|
||||
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
|
||||
// SizeVRAM is left unset: LocalAI has no authoritative per-model
|
||||
// VRAM figure to report, and a literal 0 is worse than omitting
|
||||
// the field (clients treat 0 as "costs nothing").
|
||||
Capabilities: caps,
|
||||
}
|
||||
models = append(models, entry)
|
||||
@@ -143,6 +147,37 @@ func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.Oll
|
||||
return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg)
|
||||
}
|
||||
|
||||
// modelOnDiskSize returns the on-disk byte size of a model's primary weight
|
||||
// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath.
|
||||
// Returns nil when the size is unknown so callers omit the JSON field instead
|
||||
// of emitting an authoritative 0 (issue #11969).
|
||||
func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 {
|
||||
if ml == nil || ml.ModelPath == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
// List endpoints pass the stored model ID, including any configured tag.
|
||||
configName := name
|
||||
rel := configName
|
||||
if bcl != nil {
|
||||
if cfg, exists := bcl.GetModelConfig(configName); exists {
|
||||
if fileName := cfg.ModelFileName(); fileName != "" {
|
||||
rel = fileName
|
||||
}
|
||||
}
|
||||
}
|
||||
if rel == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
info, err := os.Stat(filepath.Join(ml.ModelPath, rel))
|
||||
if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 {
|
||||
return nil
|
||||
}
|
||||
size := info.Size()
|
||||
return &size
|
||||
}
|
||||
|
||||
func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails {
|
||||
family := cfg.Backend
|
||||
details := schema.OllamaModelDetails{
|
||||
|
||||
@@ -13,6 +13,8 @@ import (
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/http/endpoints/ollama"
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
@@ -163,8 +165,165 @@ parameters:
|
||||
})
|
||||
|
||||
Describe("ListModelsEndpoint", func() {
|
||||
It("includes capabilities and details for each listed model in /api/tags", func() {
|
||||
Skip("covered by per-entry tests; integration smoke test")
|
||||
var (
|
||||
tmpDir string
|
||||
bcl *config.ModelConfigLoader
|
||||
ml *model.ModelLoader
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
tmpDir, err = os.MkdirTemp("", "ollama-tags-test-*")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
ml = model.NewModelLoader(systemState)
|
||||
bcl = config.NewModelConfigLoader(tmpDir)
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
_ = os.RemoveAll(tmpDir)
|
||||
})
|
||||
|
||||
writeConfig := func(name, yaml string) {
|
||||
path := filepath.Join(tmpDir, name+".yaml")
|
||||
Expect(os.WriteFile(path, []byte(yaml), 0o644)).To(Succeed())
|
||||
Expect(bcl.ReadModelConfig(path)).To(Succeed())
|
||||
}
|
||||
|
||||
callTags := func() (schema.OllamaListResponse, []byte) {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/tags", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
handler := ollama.ListModelsEndpoint(bcl, ml)
|
||||
Expect(handler(c)).To(Succeed())
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
|
||||
var resp schema.OllamaListResponse
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
|
||||
return resp, rec.Body.Bytes()
|
||||
}
|
||||
|
||||
It("uses the exact configured name when a model has a tag", func() {
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "base.gguf"), []byte("base"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "tagged.gguf"), []byte("tagged-weights"), 0o644)).To(Succeed())
|
||||
writeConfig("chat", "name: chat\nparameters:\n model: base.gguf\n")
|
||||
writeConfig("tagged", "name: chat:q8\nparameters:\n model: tagged.gguf\n")
|
||||
|
||||
resp, _ := callTags()
|
||||
var tagged *int64
|
||||
for _, entry := range resp.Models {
|
||||
if entry.Name == "chat:q8" {
|
||||
tagged = entry.Size
|
||||
}
|
||||
}
|
||||
Expect(tagged).ToNot(BeNil())
|
||||
Expect(*tagged).To(Equal(int64(len("tagged-weights"))))
|
||||
})
|
||||
|
||||
It("reports on-disk size from ModelFileName+ModelPath and omits size when unknown", func() {
|
||||
weight := []byte("fake-gguf-weights-0123456789")
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "Llama-3-8B-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
|
||||
writeConfig("chat", `
|
||||
name: chat
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: Llama-3-8B-Q4_K_M.gguf
|
||||
`)
|
||||
writeConfig("missing-weights", `
|
||||
name: missing-weights
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: does-not-exist.gguf
|
||||
`)
|
||||
|
||||
resp, raw := callTags()
|
||||
Expect(resp.Models).To(HaveLen(2))
|
||||
|
||||
byName := map[string]schema.OllamaModelEntry{}
|
||||
for _, m := range resp.Models {
|
||||
byName[m.Name] = m
|
||||
}
|
||||
|
||||
chat := byName["chat:latest"]
|
||||
Expect(chat.Size).ToNot(BeNil())
|
||||
Expect(*chat.Size).To(Equal(int64(len(weight))))
|
||||
Expect(chat.Capabilities).To(ContainElement("completion"))
|
||||
Expect(chat.Details.QuantizationLevel).To(Equal("Q4_K_M"))
|
||||
|
||||
missing := byName["missing-weights:latest"]
|
||||
Expect(missing.Size).To(BeNil())
|
||||
Expect(string(raw)).ToNot(ContainSubstring(`"size":0`))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("ListRunningEndpoint", func() {
|
||||
var (
|
||||
tmpDir string
|
||||
bcl *config.ModelConfigLoader
|
||||
ml *model.ModelLoader
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
tmpDir, err = os.MkdirTemp("", "ollama-ps-test-*")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
ml = model.NewModelLoader(systemState)
|
||||
bcl = config.NewModelConfigLoader(tmpDir)
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
_ = os.RemoveAll(tmpDir)
|
||||
})
|
||||
|
||||
It("reports on-disk size for loaded models and omits size_vram when unknown", func() {
|
||||
weight := []byte("loaded-model-weights-abcdef")
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "granite-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
|
||||
|
||||
cfgPath := filepath.Join(tmpDir, "granite.yaml")
|
||||
Expect(os.WriteFile(cfgPath, []byte(`
|
||||
name: granite
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: granite-Q4_K_M.gguf
|
||||
`), 0o644)).To(Succeed())
|
||||
Expect(bcl.ReadModelConfig(cfgPath)).To(Succeed())
|
||||
|
||||
store := model.NewInMemoryModelStore()
|
||||
store.Set("granite", model.NewModel("granite", "addr", nil))
|
||||
ml.SetModelStore(store)
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/ps", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
handler := ollama.ListRunningEndpoint(bcl, ml)
|
||||
Expect(handler(c)).To(Succeed())
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
|
||||
raw := rec.Body.String()
|
||||
Expect(raw).ToNot(ContainSubstring(`"size":0`))
|
||||
Expect(raw).ToNot(ContainSubstring(`"size_vram"`))
|
||||
|
||||
var resp schema.OllamaPsResponse
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
|
||||
Expect(resp.Models).To(HaveLen(1))
|
||||
Expect(resp.Models[0].Name).To(Equal("granite:latest"))
|
||||
Expect(resp.Models[0].Size).ToNot(BeNil())
|
||||
Expect(*resp.Models[0].Size).To(Equal(int64(len(weight))))
|
||||
Expect(resp.Models[0].SizeVRAM).To(BeNil())
|
||||
Expect(resp.Models[0].Details.QuantizationLevel).To(Equal("Q4_K_M"))
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -451,7 +451,7 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator
|
||||
}
|
||||
|
||||
// Update input grammar or json_schema based on use_llama_grammar option
|
||||
jsStruct := funcs.ToJSONStructure(config.FunctionsConfig.FunctionNameKey, config.FunctionsConfig.FunctionNameKey)
|
||||
jsStruct := config.FunctionsConfig.ToJSONStructure(funcs)
|
||||
g, err := jsStruct.Grammar(config.FunctionsConfig.GrammarOptions()...)
|
||||
if err == nil {
|
||||
config.Grammar = g
|
||||
|
||||
@@ -36,6 +36,12 @@ func ListModelCapabilitiesEndpoint(bcl *config.ModelConfigLoader, ml *model.Mode
|
||||
for _, m := range modelNames {
|
||||
entry := schema.ModelCapabilities{ID: m, Object: "model"}
|
||||
if cfg, ok := modelConfigFor(bcl, m); ok {
|
||||
// Mirror the request path: SetDefaults applies the application
|
||||
// default only when the model leaves context_size unset. An
|
||||
// explicit 0 or -1 falls through to the backend fallback there.
|
||||
if cfg.ContextSize == nil && appConfig != nil && appConfig.ContextSize > 0 {
|
||||
cfg.ContextSize = &appConfig.ContextSize
|
||||
}
|
||||
entry.Capabilities = cfg.Capabilities()
|
||||
entry.ThreeDOperations = cfg.ThreeDOperations()
|
||||
entry.InputModalities = cfg.InputModalities()
|
||||
|
||||
@@ -160,6 +160,33 @@ parameters:
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
|
||||
})
|
||||
|
||||
It("uses application config context size when model context_size is unset", func() {
|
||||
writeConfig("llm-app-default", `
|
||||
name: llm-app-default
|
||||
backend: llama-cpp
|
||||
parameters:
|
||||
model: model.gguf
|
||||
`)
|
||||
appConf.ContextSize = 8192
|
||||
entry := entryFor(call(), "llm-app-default")
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(8192))
|
||||
})
|
||||
|
||||
It("keeps the backend fallback when the model sets a non-positive context_size", func() {
|
||||
writeConfig("llm-explicit-zero", `
|
||||
name: llm-explicit-zero
|
||||
backend: llama-cpp
|
||||
context_size: 0
|
||||
parameters:
|
||||
model: model.gguf
|
||||
`)
|
||||
appConf.ContextSize = 8192
|
||||
entry := entryFor(call(), "llm-explicit-zero")
|
||||
Expect(entry).NotTo(BeNil())
|
||||
Expect(entry.ContextSize).To(Equal(backend.DefaultContextSize))
|
||||
})
|
||||
It("reports an alias with its target's capabilities and context_size", func() {
|
||||
writeConfig("real-llm", `
|
||||
name: real-llm
|
||||
|
||||
@@ -293,7 +293,7 @@ func (m *wrappedModel) Predict(ctx context.Context, messages schema.Messages, im
|
||||
}
|
||||
|
||||
// Generate grammar from function definitions
|
||||
jsStruct := functions.Functions(funcs).ToJSONStructure(turnCfg.FunctionsConfig.FunctionNameKey, turnCfg.FunctionsConfig.FunctionNameKey)
|
||||
jsStruct := turnCfg.FunctionsConfig.ToJSONStructure(functions.Functions(funcs))
|
||||
g, err := jsStruct.Grammar(turnCfg.FunctionsConfig.GrammarOptions()...)
|
||||
if err == nil {
|
||||
turnCfg.Grammar = g
|
||||
|
||||
@@ -204,7 +204,7 @@ func ResponsesEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, eval
|
||||
}
|
||||
|
||||
// Generate grammar to constrain model output to valid function calls
|
||||
jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey)
|
||||
jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction)
|
||||
g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...)
|
||||
if err == nil {
|
||||
cfg.Grammar = g
|
||||
@@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
return true
|
||||
}
|
||||
|
||||
// Try JSON parsing as fallback
|
||||
jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true)
|
||||
if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount {
|
||||
// Only completed JSON calls can be emitted as completed SSE items.
|
||||
jsonResults := parseStreamingJSONToolCalls(cleanedResult)
|
||||
if len(jsonResults) > lastEmittedToolCallCount {
|
||||
for i := lastEmittedToolCallCount; i < len(jsonResults); i++ {
|
||||
jsonObj := jsonResults[i]
|
||||
if name, ok := jsonObj["name"].(string); ok && name != "" {
|
||||
args := "{}"
|
||||
if argsVal, ok := jsonObj["arguments"]; ok {
|
||||
if argsStr, ok := argsVal.(string); ok {
|
||||
args = argsStr
|
||||
} else {
|
||||
argsBytes, _ := json.Marshal(argsVal)
|
||||
args = string(argsBytes)
|
||||
}
|
||||
}
|
||||
tc := jsonResults[i]
|
||||
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
|
||||
outputIndex++
|
||||
|
||||
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
|
||||
outputIndex++
|
||||
|
||||
functionCallItem := &schema.ORItemField{
|
||||
Type: "function_call",
|
||||
ID: toolCallID,
|
||||
Status: "completed",
|
||||
CallID: toolCallID,
|
||||
Name: name,
|
||||
Arguments: args,
|
||||
}
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
functionCallItem := &schema.ORItemField{
|
||||
Type: "function_call",
|
||||
ID: toolCallID,
|
||||
Status: "completed",
|
||||
CallID: toolCallID,
|
||||
Name: tc.Name,
|
||||
Arguments: tc.Arguments,
|
||||
}
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
}
|
||||
lastEmittedToolCallCount = len(jsonResults)
|
||||
c.Response().Flush()
|
||||
@@ -2424,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
}
|
||||
|
||||
// Non-tool-call streaming path
|
||||
messageOutputIndex := outputIndex
|
||||
var reasoningOutputIndex int
|
||||
// Emit output_item.added for message
|
||||
currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String())
|
||||
messageItem := &schema.ORItemField{
|
||||
@@ -2436,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
Item: messageItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2448,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Part: &emptyTextPart,
|
||||
})
|
||||
@@ -2471,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
}
|
||||
|
||||
// Handle reasoning item
|
||||
if extractor.Reasoning() != "" {
|
||||
if extractor.Reasoning() != "" || reasoningDelta != "" {
|
||||
// Check if we need to create reasoning item
|
||||
if currentReasoningID == "" {
|
||||
outputIndex++
|
||||
reasoningOutputIndex = outputIndex
|
||||
currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String())
|
||||
reasoningItem := &schema.ORItemField{
|
||||
Type: "reasoning",
|
||||
@@ -2484,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
Item: reasoningItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2496,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Part: &emptyPart,
|
||||
})
|
||||
@@ -2509,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.delta",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Delta: strPtr(reasoningDelta),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2526,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.delta",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Delta: strPtr(contentDelta),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2595,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Text: strPtr(finalReasoning),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2608,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Part: &reasoningPart,
|
||||
})
|
||||
@@ -2624,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
Item: reasoningItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2658,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Text: strPtr(result),
|
||||
Logprobs: logprobsPtr(mcpStreamLogprobs),
|
||||
@@ -2671,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Part: &resultPart,
|
||||
})
|
||||
@@ -2683,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
Item: messageItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2723,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
// Emit response.completed
|
||||
now := time.Now().Unix()
|
||||
|
||||
// Collect final output items (reasoning first, then messages, then tool calls)
|
||||
var finalOutputItems []schema.ORItemField
|
||||
// Add reasoning item if it exists
|
||||
if currentReasoningID != "" && finalReasoning != "" {
|
||||
finalOutputItems = append(finalOutputItems, schema.ORItemField{
|
||||
Type: "reasoning",
|
||||
ID: currentReasoningID,
|
||||
Status: "completed",
|
||||
Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)},
|
||||
})
|
||||
}
|
||||
// Add message item
|
||||
if len(collectedOutputItems) > 0 {
|
||||
// Use collected items (may include reasoning already)
|
||||
for _, item := range collectedOutputItems {
|
||||
if item.Type == "message" {
|
||||
finalOutputItems = append(finalOutputItems, item)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
finalOutputItems = append(finalOutputItems, *messageItem)
|
||||
}
|
||||
// Add function_call items from fallback
|
||||
for _, item := range collectedOutputItems {
|
||||
if item.Type == "function_call" {
|
||||
finalOutputItems = append(finalOutputItems, item)
|
||||
}
|
||||
}
|
||||
// The final output array must use the indices announced in the stream.
|
||||
// The message is opened first, followed by reasoning and fallback calls.
|
||||
finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...)
|
||||
responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{
|
||||
InputTokens: noToolTokenUsage.Prompt,
|
||||
OutputTokens: noToolTokenUsage.Completion,
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/backend"
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Responses stream item consistency", func() {
|
||||
DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) {
|
||||
originalInference := backend.ModelInferenceFunc
|
||||
DeferCleanup(func() { backend.ModelInferenceFunc = originalInference })
|
||||
backend.ModelInferenceFunc = func(
|
||||
ctx context.Context, prompt string, messages schema.Messages,
|
||||
images, videos, audios []string, loader *model.ModelLoader,
|
||||
cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig,
|
||||
tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string,
|
||||
logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string,
|
||||
) (func() (backend.LLMResponse, error), error) {
|
||||
return func() (backend.LLMResponse, error) {
|
||||
for i, token := range tokens {
|
||||
usage := backend.TokenUsage{}
|
||||
if len(chatDeltas) > 0 {
|
||||
usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]}
|
||||
}
|
||||
if !tokenCallback(token, usage) {
|
||||
break
|
||||
}
|
||||
}
|
||||
return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil
|
||||
}, nil
|
||||
}
|
||||
cfg := &config.ModelConfig{}
|
||||
cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback
|
||||
cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)<tool_call>(.*?)</tool_call>`}
|
||||
recorder := httptest.NewRecorder()
|
||||
request := httptest.NewRequest("POST", "/v1/responses", nil)
|
||||
c := echo.New().NewContext(request, recorder)
|
||||
input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true}
|
||||
err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n"))
|
||||
|
||||
var events []schema.ORStreamEvent
|
||||
var completed *schema.ORResponseResource
|
||||
for _, line := range strings.Split(recorder.Body.String(), "\n") {
|
||||
if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" {
|
||||
continue
|
||||
}
|
||||
var event schema.ORStreamEvent
|
||||
Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed())
|
||||
Expect(event.Type).NotTo(Equal("error"))
|
||||
events = append(events, event)
|
||||
if event.Type == "response.completed" {
|
||||
completed = event.Response
|
||||
}
|
||||
}
|
||||
Expect(completed).NotTo(BeNil())
|
||||
wantCount := 1
|
||||
if wantReasoning != "" {
|
||||
wantCount++
|
||||
}
|
||||
if fallback {
|
||||
wantCount++
|
||||
}
|
||||
Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls")
|
||||
|
||||
indices := map[string]int{}
|
||||
done := map[string]int{}
|
||||
deltas := map[string]string{}
|
||||
for i, event := range events {
|
||||
Expect(event.SequenceNumber).To(Equal(i))
|
||||
if event.Type == "response.output_item.added" {
|
||||
Expect(event.Item).NotTo(BeNil())
|
||||
Expect(event.OutputIndex).NotTo(BeNil())
|
||||
Expect(indices).NotTo(HaveKey(event.Item.ID))
|
||||
Expect(*event.OutputIndex).To(Equal(len(indices)))
|
||||
indices[event.Item.ID] = *event.OutputIndex
|
||||
}
|
||||
id := event.ItemID
|
||||
if event.Item != nil {
|
||||
id = event.Item.ID
|
||||
}
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
Expect(indices).To(HaveKey(id))
|
||||
Expect(event.OutputIndex).NotTo(BeNil())
|
||||
Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id)
|
||||
Expect(completed.Output[indices[id]].ID).To(Equal(id))
|
||||
if event.Type == "response.output_item.done" {
|
||||
done[id]++
|
||||
Expect(event.Item.Status).To(Equal("completed"))
|
||||
Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type))
|
||||
if event.Item.Type == "function_call" {
|
||||
Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name))
|
||||
Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments))
|
||||
} else {
|
||||
Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content))
|
||||
}
|
||||
}
|
||||
if event.Type == "response.output_text.delta" {
|
||||
deltas[id] += *event.Delta
|
||||
}
|
||||
}
|
||||
Expect(indices).To(HaveLen(wantCount))
|
||||
for _, item := range completed.Output {
|
||||
Expect(done[item.ID]).To(Equal(1))
|
||||
switch item.Type {
|
||||
case "message", "reasoning":
|
||||
want := wantAnswer
|
||||
if item.Type == "reasoning" {
|
||||
want = wantReasoning
|
||||
}
|
||||
parts, ok := item.Content.([]any)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(parts).To(HaveLen(1))
|
||||
Expect(parts[0].(map[string]any)["text"]).To(Equal(want))
|
||||
if !fallback {
|
||||
Expect(deltas[item.ID]).To(Equal(want))
|
||||
}
|
||||
case "function_call":
|
||||
Expect(item.Name).To(Equal("get_weather"))
|
||||
Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`))
|
||||
Expect(item.CallID).NotTo(BeEmpty())
|
||||
default:
|
||||
Fail("unexpected output item type: " + item.Type)
|
||||
}
|
||||
}
|
||||
},
|
||||
Entry("tagged reasoning and answer", []string{"<think>", "Let me think.", "</think>", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false),
|
||||
Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false),
|
||||
Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false),
|
||||
Entry("automatic fallback tool call", []string{`<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "", "", true),
|
||||
Entry("reasoning and automatic fallback tool call", []string{"<think>", "Let me think.", "</think>", `<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "Let me think.", "", true),
|
||||
)
|
||||
})
|
||||
@@ -0,0 +1,35 @@
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/mudler/LocalAI/pkg/functions"
|
||||
)
|
||||
|
||||
func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults {
|
||||
// Partial parsing heals unfinished arguments. The caller emits terminal
|
||||
// events and never revisits emitted calls, so only accept complete JSON.
|
||||
// Keep completed objects returned before an unfinished trailing object.
|
||||
objects, _ := functions.ParseJSONIterative(text, false)
|
||||
var calls []functions.FuncCallResults
|
||||
for _, object := range objects {
|
||||
name, ok := object["name"].(string)
|
||||
if !ok || name == "" {
|
||||
continue
|
||||
}
|
||||
arguments := "{}"
|
||||
if value, ok := object["arguments"]; ok {
|
||||
if s, ok := value.(string); ok {
|
||||
arguments = s
|
||||
} else {
|
||||
data, err := json.Marshal(value)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
arguments = string(data)
|
||||
}
|
||||
}
|
||||
calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments})
|
||||
}
|
||||
return calls
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"github.com/mudler/LocalAI/pkg/functions"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Streaming JSON tool calls", func() {
|
||||
It("waits for the arguments before completing a split call", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty())
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty())
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("does not complete a call at any intermediate token boundary", func() {
|
||||
text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}`
|
||||
for end := 1; end < len(text); end++ {
|
||||
Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end])
|
||||
}
|
||||
Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1))
|
||||
})
|
||||
|
||||
It("keeps completed calls while the next call is incomplete", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("preserves string arguments and calls that take no arguments", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
{Name: "status", Arguments: `{}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("does not count unrelated JSON objects as emitted calls", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "status", Arguments: `{}`},
|
||||
}))
|
||||
})
|
||||
})
|
||||
@@ -381,7 +381,7 @@ func handleWSResponseCreate(connCtx context.Context, conn *lockedConn, connectio
|
||||
funcsWithNoAction = funcsWithNoAction.Select(cfg.FunctionToCall())
|
||||
}
|
||||
|
||||
jsStruct := funcsWithNoAction.ToJSONStructure(cfg.FunctionsConfig.FunctionNameKey, cfg.FunctionsConfig.FunctionNameKey)
|
||||
jsStruct := cfg.FunctionsConfig.ToJSONStructure(funcsWithNoAction)
|
||||
g, err := jsStruct.Grammar(cfg.FunctionsConfig.GrammarOptions()...)
|
||||
if err == nil {
|
||||
cfg.Grammar = g
|
||||
|
||||
@@ -20,7 +20,7 @@ import (
|
||||
// SERVED model — a router fanout that lands on a saturated downstream
|
||||
// model gets rejected even though the requested router-model has slack.
|
||||
//
|
||||
// On reject: HTTP 503, Retry-After header, error JSON. An audit row
|
||||
// On reject: HTTP 429, Retry-After header, error JSON. An audit row
|
||||
// goes into the shared event store under KindAdmission so admins see
|
||||
// rejection rates alongside PII and proxy events.
|
||||
//
|
||||
@@ -39,9 +39,10 @@ func AdmissionControl(limiter *admission.Limiter, events pii.EventStore) echo.Mi
|
||||
retryAfter := admission.RetryAfter(cfg.Limits.RetryAfterSeconds)
|
||||
recordAdmissionRejection(events, cfg.Name, retryAfter)
|
||||
c.Response().Header().Set("Retry-After", strconv.Itoa(int(retryAfter.Seconds())))
|
||||
return c.JSON(http.StatusServiceUnavailable, map[string]any{
|
||||
return c.JSON(http.StatusTooManyRequests, map[string]any{
|
||||
"error": map[string]any{
|
||||
"type": "admission_rejected",
|
||||
"type": "rate_limit_error",
|
||||
"code": "admission_rejected",
|
||||
"message": fmt.Sprintf("model %q is at capacity (max_concurrent=%d); retry after %s", cfg.Name, max, retryAfter),
|
||||
},
|
||||
})
|
||||
@@ -61,7 +62,7 @@ func recordAdmissionRejection(events pii.EventStore, modelName string, retryAfte
|
||||
if events == nil {
|
||||
return
|
||||
}
|
||||
statusCode := http.StatusServiceUnavailable
|
||||
statusCode := http.StatusTooManyRequests
|
||||
durMS := retryAfter.Milliseconds()
|
||||
id := fmt.Sprintf("adm_%d_%s", admissionEventSeq.Add(1), randHex(4))
|
||||
_ = events.Record(context.Background(), pii.PIIEvent{
|
||||
|
||||
@@ -60,7 +60,7 @@ var _ = Describe("Admission", func() {
|
||||
|
||||
It("rejects when full", func() {
|
||||
// Saturate the limiter outside the middleware, then a request
|
||||
// at the same model gets 503 with a Retry-After header.
|
||||
// at the same model gets 429 with a Retry-After header.
|
||||
lim := admission.New()
|
||||
release, ok := lim.Acquire("busy", 1)
|
||||
Expect(ok).To(BeTrue(), "setup acquire should succeed")
|
||||
@@ -75,7 +75,7 @@ var _ = Describe("Admission", func() {
|
||||
return c.String(http.StatusOK, "ok")
|
||||
})
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(rec.Code).To(Equal(http.StatusServiceUnavailable))
|
||||
Expect(rec.Code).To(Equal(http.StatusTooManyRequests))
|
||||
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
|
||||
Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects")
|
||||
Expect(rec.Body.String()).To(ContainSubstring("admission_rejected"))
|
||||
|
||||
@@ -65,6 +65,31 @@ type CorpusLoader interface {
|
||||
EnsureLoaded(ctx context.Context, storeName, embeddingModel, embeddingFingerprint string, embedder backend.Embedder, store backend.VectorStore) (int, error)
|
||||
}
|
||||
|
||||
// reseedingVectorStore runs the corpus sync before every lookup. It
|
||||
// wraps the RAW store and hands that raw store to the loader, so the
|
||||
// loader's own probe lookup never re-enters this wrapper. A sync error
|
||||
// fails the lookup closed, like the build-time load does: a decision
|
||||
// taken on an index that could not be synced is exactly the blind-
|
||||
// router bug this guards against.
|
||||
type reseedingVectorStore struct {
|
||||
backend.VectorStore
|
||||
ensure func(ctx context.Context) error
|
||||
}
|
||||
|
||||
func (s *reseedingVectorStore) SearchK(ctx context.Context, vec []float32, k int) ([]backend.Neighbor, error) {
|
||||
if err := s.ensure(ctx); err != nil {
|
||||
return nil, fmt.Errorf("router: knn corpus sync before lookup: %w", err)
|
||||
}
|
||||
return s.VectorStore.SearchK(ctx, vec, k)
|
||||
}
|
||||
|
||||
func (s *reseedingVectorStore) Search(ctx context.Context, vec []float32) (float64, []byte, bool, error) {
|
||||
if err := s.ensure(ctx); err != nil {
|
||||
return 0, nil, false, fmt.Errorf("router: knn corpus sync before lookup: %w", err)
|
||||
}
|
||||
return s.VectorStore.Search(ctx, vec)
|
||||
}
|
||||
|
||||
// ClassifierDeps bundles the backend factories the router middleware
|
||||
// needs to build a classifier and its optional L2 cache. Bundled into
|
||||
// one struct because RouteModel already takes many positional
|
||||
@@ -478,12 +503,23 @@ func buildClassifier(cfg *config.ModelConfig, deps ClassifierDeps) (router.Class
|
||||
// Loading fails closed: a live index from a different embedding
|
||||
// space may have the same vector width and return plausible but
|
||||
// incorrect routes.
|
||||
if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, vstore); err != nil {
|
||||
raw := vstore
|
||||
if n, err := deps.Corpus.EnsureLoaded(context.Background(), storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw); err != nil {
|
||||
return nil, fmt.Errorf("router classifier knn: load corpus %q: %w", storeName, err)
|
||||
} else if n > 0 {
|
||||
xlog.Info("router: knn corpus loaded",
|
||||
"router_model", cfg.Name, "store", storeName, "entries", n)
|
||||
}
|
||||
// The classifier built below is cached for the process lifetime
|
||||
// (GetOrBuildClassifier), so this sync would otherwise be the
|
||||
// only one — while the local-store process behind the index can
|
||||
// be evicted or idle-killed and relaunched EMPTY at any later
|
||||
// request. Re-check on every lookup; the corpus loader probes
|
||||
// the live index and re-seeds it from the file on a miss.
|
||||
vstore = &reseedingVectorStore{VectorStore: raw, ensure: func(ctx context.Context) error {
|
||||
_, err := deps.Corpus.EnsureLoaded(ctx, storeName, rc.KNN.EmbeddingModel, embeddingFingerprint, embedder, raw)
|
||||
return err
|
||||
}}
|
||||
}
|
||||
knnClassifier := router.NewKNNClassifier(embedder, vstore, router.KNNClassifierOptions{
|
||||
K: rc.KNN.K,
|
||||
|
||||
@@ -581,6 +581,29 @@ var (
|
||||
errTestKNNInsert = errors.New("knn classifier must never insert into the corpus")
|
||||
)
|
||||
|
||||
// reseedingCorpusLoader mirrors corpus.Manager's contract: EnsureLoaded
|
||||
// is a no-op while the index still answers, and re-seeds it when the
|
||||
// store came back empty. It insists on the RAW scripted store, so a
|
||||
// wrapper leaking into the loader (and recursing) fails the spec.
|
||||
type reseedingCorpusLoader struct {
|
||||
seed []backend.Neighbor
|
||||
calls, reseeds int
|
||||
}
|
||||
|
||||
func (r *reseedingCorpusLoader) EnsureLoaded(_ context.Context, _, _, _ string, _ backend.Embedder, store backend.VectorStore) (int, error) {
|
||||
r.calls++
|
||||
s, ok := store.(*scriptedVectorStore)
|
||||
if !ok {
|
||||
return 0, errors.New("corpus loader must receive the raw store, not a wrapper")
|
||||
}
|
||||
if len(s.neighbors) == 0 {
|
||||
s.neighbors = r.seed
|
||||
r.reseeds++
|
||||
return len(r.seed), nil
|
||||
}
|
||||
return 0, nil
|
||||
}
|
||||
|
||||
type failingCorpusLoader struct{ err error }
|
||||
|
||||
func (f failingCorpusLoader) EnsureLoaded(context.Context, string, string, string, backend.Embedder, backend.VectorStore) (int, error) {
|
||||
@@ -710,6 +733,43 @@ var _ = Describe("RouteModel middleware (knn classifier)", func() {
|
||||
Expect(err.Error()).To(ContainSubstring("knn"))
|
||||
})
|
||||
|
||||
It("re-seeds a relaunched corpus index behind the cached classifier", func() {
|
||||
// The classifier is built once and cached; the local-store
|
||||
// process behind its index may be evicted or idle-killed and
|
||||
// relaunched empty afterwards. Measured in production: every probe
|
||||
// then fell back with similarity 0 while corpus/stats still
|
||||
// reported the full count. The lookup path must re-seed.
|
||||
routerCfg := newKNNRouterModel(modelDir, "smart-router")
|
||||
writeCandidate(modelDir, "small-model")
|
||||
writeCandidate(modelDir, "big-model")
|
||||
seeded := []backend.Neighbor{
|
||||
{Similarity: 0.92, Payload: corpusPayload("code-generation")},
|
||||
{Similarity: 0.88, Payload: corpusPayload("code-generation")},
|
||||
}
|
||||
vstore.neighbors = seeded
|
||||
corpus := &reseedingCorpusLoader{seed: seeded}
|
||||
deps := knnDeps()
|
||||
deps.EmbedderFingerprint = func(string) (string, error) { return "fp", nil }
|
||||
deps.Corpus = corpus
|
||||
registry := router.NewRegistry()
|
||||
|
||||
first, err := GetOrBuildClassifier(registry, routerCfg, deps)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
d, err := first.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"})
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(d.Labels).To(ContainElement("code-generation"))
|
||||
Expect(corpus.reseeds).To(Equal(0), "a healthy index is not re-seeded")
|
||||
|
||||
vstore.neighbors = nil // the store process was relaunched empty
|
||||
again, err := GetOrBuildClassifier(registry, routerCfg, deps)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(again).To(BeIdenticalTo(first), "the classifier stays cached — the sync must live on the lookup path")
|
||||
d, err = again.Classify(context.Background(), router.Probe{Prompt: "debug my Go null pointer"})
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(d.Labels).To(ContainElement("code-generation"), "lookup on a relaunched index re-seeds instead of falling back")
|
||||
Expect(corpus.reseeds).To(Equal(1))
|
||||
})
|
||||
|
||||
It("fails closed when the persisted corpus cannot sync into the live index", func() {
|
||||
routerCfg := newKNNRouterModel(modelDir, "smart-router")
|
||||
writeCandidate(modelDir, "small-model")
|
||||
|
||||
@@ -931,7 +931,8 @@ function eventDetails(e) {
|
||||
}
|
||||
case 'admission': {
|
||||
const retry = e.duration_ms != null ? `retry-after ${Math.round(e.duration_ms / 1000)}s` : ''
|
||||
return `HTTP 503 rejected · ${retry}`
|
||||
// Older audit rows were recorded as 503; newer ones as 429.
|
||||
return `HTTP ${e.status_code || 429} rejected · ${retry}`
|
||||
}
|
||||
default: {
|
||||
const len = e.length != null ? `len ${e.length}` : ''
|
||||
|
||||
@@ -208,6 +208,9 @@ type SysInfoModel struct {
|
||||
// when the model has no local process (a distributed worker holds it) or
|
||||
// the process could not be read.
|
||||
Process *SysInfoProcess `json:"process,omitempty"`
|
||||
// SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
|
||||
// the backend process tree has no complete supported reading.
|
||||
SizeVRAM *uint64 `json:"size_vram,omitempty"`
|
||||
}
|
||||
|
||||
// SysInfoProcess is a point-in-time reading of one backend process.
|
||||
|
||||
+10
-5
@@ -293,12 +293,14 @@ type OllamaModelDetails struct {
|
||||
QuantizationLevel string `json:"quantization_level,omitempty"`
|
||||
}
|
||||
|
||||
// OllamaModelEntry represents a model in the list response
|
||||
// OllamaModelEntry represents a model in the list response.
|
||||
// Size is a pointer so an unknown on-disk size can be omitted instead of
|
||||
// serializing as the misleading literal 0 (see issue #11969).
|
||||
type OllamaModelEntry struct {
|
||||
Name string `json:"name"`
|
||||
Model string `json:"model"`
|
||||
ModifiedAt time.Time `json:"modified_at"`
|
||||
Size int64 `json:"size"`
|
||||
Size *int64 `json:"size,omitempty"`
|
||||
Digest string `json:"digest"`
|
||||
Details OllamaModelDetails `json:"details"`
|
||||
Capabilities []string `json:"capabilities,omitempty"`
|
||||
@@ -309,15 +311,18 @@ type OllamaListResponse struct {
|
||||
Models []OllamaModelEntry `json:"models"`
|
||||
}
|
||||
|
||||
// OllamaPsEntry represents a running model in the ps response
|
||||
// OllamaPsEntry represents a running model in the ps response.
|
||||
// Size and SizeVRAM are pointers so unknown values are omitted rather than
|
||||
// reported as authoritative zeros (see issue #11969). SizeVRAM is only set
|
||||
// when the runtime can provide a real VRAM figure.
|
||||
type OllamaPsEntry struct {
|
||||
Name string `json:"name"`
|
||||
Model string `json:"model"`
|
||||
Size int64 `json:"size"`
|
||||
Size *int64 `json:"size,omitempty"`
|
||||
Digest string `json:"digest"`
|
||||
Details OllamaModelDetails `json:"details"`
|
||||
ExpiresAt time.Time `json:"expires_at"`
|
||||
SizeVRAM int64 `json:"size_vram"`
|
||||
SizeVRAM *int64 `json:"size_vram,omitempty"`
|
||||
Capabilities []string `json:"capabilities,omitempty"`
|
||||
}
|
||||
|
||||
|
||||
@@ -99,7 +99,7 @@ type OpenAIResponse struct {
|
||||
// OpenAI-SDK consumers that filter on a truthy `result.usage`
|
||||
// (continuedev/continue, Kilo Code, Roo Code, etc.).
|
||||
Usage *OpenAIUsage `json:"usage,omitempty"`
|
||||
Metadata json.RawMessage `json:"metadata,omitempty"`
|
||||
Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"`
|
||||
}
|
||||
|
||||
// StreamOptions mirrors OpenAI's `stream_options` request field. The only
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package schema_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SysInfoModel memory", func() {
|
||||
It("omits unavailable VRAM while preserving a measured zero", func() {
|
||||
entry := schema.SysInfoModel{ID: "model"}
|
||||
encoded, err := json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
|
||||
|
||||
zero := uint64(0)
|
||||
entry.SizeVRAM = &zero
|
||||
encoded, err = json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
|
||||
})
|
||||
})
|
||||
@@ -107,6 +107,13 @@ type AgentConfig struct {
|
||||
LoopDetection int `json:"loop_detection"`
|
||||
EnableAutoCompaction bool `json:"enable_auto_compaction"`
|
||||
AutoCompactionThreshold int `json:"auto_compaction_threshold"`
|
||||
|
||||
// Tool policy (see toolpolicy.go)
|
||||
RequiredToolBeforeFinish string `json:"required_tool_before_finish"`
|
||||
RequiredToolBeforeFinishPrompt string `json:"required_tool_before_finish_prompt"`
|
||||
RequiredToolBeforeFinishAttempts int `json:"required_tool_before_finish_attempts"`
|
||||
AllowedTools ToolNames `json:"allowed_tools"`
|
||||
ExcludedTools ToolNames `json:"excluded_tools"`
|
||||
}
|
||||
|
||||
// ConnectorConfig defines a connector integration (Slack, Discord, etc.).
|
||||
|
||||
@@ -134,6 +134,14 @@ func defaultFields() []ConfigField {
|
||||
{Name: "enable_reasoning_tool", Label: "Enable Reasoning for Tools", Type: FieldCheckbox, DefaultValue: true, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "enable_reasoning_for_instruct", Label: "Enable Reasoning for Instruct Models", Type: FieldCheckbox, DefaultValue: false, HelpText: "Force structured reasoning before tool selection (recommended for instruct-tuned models)", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "enable_guided_tools", Label: "Enable Guided Tools", Type: FieldCheckbox, DefaultValue: false, HelpText: "Filter tools through guidance using descriptions", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "allowed_tools", Label: "Allowed Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "get_document_content, search",
|
||||
HelpText: "Comma or newline separated tool names. When set, the agent is offered only these tools (actions, knowledge base tools and MCP tools). send_message, stop and update_state are always kept. Leave empty to offer every tool.",
|
||||
Tags: ConfigFieldTags{Section: "AdvancedSettings"},
|
||||
},
|
||||
{Name: "excluded_tools", Label: "Excluded Tools", Type: FieldTextarea, DefaultValue: "", Placeholder: "search_memory",
|
||||
HelpText: "Comma or newline separated tool names that are never offered to the agent, even if they are in Allowed Tools. send_message, stop and update_state cannot be excluded; use their own settings instead.",
|
||||
Tags: ConfigFieldTags{Section: "AdvancedSettings"},
|
||||
},
|
||||
{Name: "enable_skills", Label: "Enable Skills", Type: FieldCheckbox, DefaultValue: false, HelpText: "Inject skills into the agent", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "skills_mode", Label: "Skills Injection Mode", Type: FieldSelect, DefaultValue: "prompt",
|
||||
Options: []ConfigFieldOption{
|
||||
@@ -146,6 +154,18 @@ func defaultFields() []ConfigField {
|
||||
},
|
||||
{Name: "parallel_jobs", Label: "Parallel Jobs", Type: FieldNumber, DefaultValue: 5, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "max_attempts", Label: "Max Attempts", Type: FieldNumber, DefaultValue: 2, Min: 1, Step: 1, Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
{Name: "required_tool_before_finish", Label: "Required Tool Before Finish", Type: FieldText, DefaultValue: "", Placeholder: "check_policy",
|
||||
HelpText: "Name of a tool the agent must call successfully (a JSON result with \"ok\": true) before it may send its final answer. Has no effect if the agent does not have this tool. Leave empty to disable.",
|
||||
Tags: ConfigFieldTags{Section: "AdvancedSettings"},
|
||||
},
|
||||
{Name: "required_tool_before_finish_prompt", Label: "Required Tool Prompt", Type: FieldTextarea, DefaultValue: "",
|
||||
HelpText: "Instruction sent to the model when it tries to finish before the required tool has passed. Leave empty to use a default that names the tool.",
|
||||
Tags: ConfigFieldTags{Section: "AdvancedSettings"},
|
||||
},
|
||||
{Name: "required_tool_before_finish_attempts", Label: "Required Tool Attempts", Type: FieldNumber, DefaultValue: 3, Min: 1, Step: 1,
|
||||
HelpText: "How many times the model is told to run the required tool before its answer is sent anyway",
|
||||
Tags: ConfigFieldTags{Section: "AdvancedSettings"},
|
||||
},
|
||||
{Name: "max_iterations", Label: "Max Iterations", Type: FieldNumber, DefaultValue: 1, Min: 1, Step: 1, HelpText: "Maximum tool loop iterations per execution", Tags: ConfigFieldTags{Section: "AdvancedSettings"}},
|
||||
|
||||
// MCP
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"cmp"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -181,6 +182,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
|
||||
// Build cogito options
|
||||
var cogitoOpts []cogito.Option
|
||||
// Local tools are collected first so the tool filter applies to all of
|
||||
// them at once; cogito only runs tools it offered, so filtering what is
|
||||
// offered also filters the lookup of the model's tool calls.
|
||||
var localTools []cogito.ToolDefinitionInterface
|
||||
filter := newToolFilter(cfg.AllowedTools, cfg.ExcludedTools)
|
||||
|
||||
// MCP sessions
|
||||
sessions, cleanup := setupMCPSessions(ctx, cfg)
|
||||
@@ -188,7 +194,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
defer cleanup()
|
||||
}
|
||||
if len(sessions) > 0 {
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...))
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithMCPs(sessions...), cogito.WithMCPToolFilter(filter.mcpToolFilter()))
|
||||
}
|
||||
|
||||
// KB tools (search_memory / add_memory) — when kb mode is "tools" or "both"
|
||||
@@ -197,7 +203,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
if kbResults <= 0 {
|
||||
kbResults = 5
|
||||
}
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithTools(
|
||||
localTools = append(localTools,
|
||||
cogito.NewToolDefinition(
|
||||
KBSearchMemoryTool{APIURL: effectiveURL, APIKey: effectiveKey, Collection: cfg.Name, MaxResults: kbResults, UserID: userID, CitationCollector: kbCitations},
|
||||
KBSearchMemoryArgs{},
|
||||
@@ -210,7 +216,7 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
"add_memory",
|
||||
"Store content in memory for later retrieval",
|
||||
),
|
||||
))
|
||||
)
|
||||
}
|
||||
|
||||
// Skill tools — when skills_mode is "tools" or "both"
|
||||
@@ -220,18 +226,36 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
allSkills, _ := skillProvider.ListSkills()
|
||||
filtered := FilterSkills(allSkills, cfg.SelectedSkills)
|
||||
if len(filtered) > 0 {
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithTools(
|
||||
localTools = append(localTools,
|
||||
cogito.NewToolDefinition(
|
||||
RequestSkillTool{Skills: filtered},
|
||||
RequestSkillArgs{},
|
||||
"request_skill",
|
||||
"Request a skill by name. Available skills: "+skillNames(filtered),
|
||||
),
|
||||
))
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
localTools = filter.filterTools(localTools)
|
||||
if len(localTools) > 0 {
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithTools(localTools...))
|
||||
}
|
||||
|
||||
// Required-tool gate: the agent must run the configured tool to success
|
||||
// before its answer is final. It is enforced on the output because a
|
||||
// model follows "always call X first" unreliably.
|
||||
requiredTool := cfg.RequiredToolBeforeFinish
|
||||
requiredPassed := false
|
||||
requiredAttempts := 0
|
||||
maxRequiredAttempts := cfg.RequiredToolBeforeFinishAttempts
|
||||
if maxRequiredAttempts <= 0 {
|
||||
maxRequiredAttempts = defaultRequiredFinishAttempts
|
||||
}
|
||||
requiredPrompt := requiredFinishPromptFor(requiredTool, cfg.RequiredToolBeforeFinishPrompt)
|
||||
requiredAvailable := requiredTool != "" && requiredToolAvailable(ctx, requiredTool, localTools, sessions, filter)
|
||||
|
||||
// Sink state is always disabled — the agent responds directly when no tools match.
|
||||
cogitoOpts = append(cogitoOpts, cogito.DisableSinkState)
|
||||
|
||||
@@ -250,8 +274,11 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
}
|
||||
|
||||
// Tool call result callback
|
||||
if cb.OnToolResult != nil || cb.OnToolCall != nil {
|
||||
if cb.OnToolResult != nil || cb.OnToolCall != nil || requiredAvailable {
|
||||
cogitoOpts = append(cogitoOpts, cogito.WithToolCallResultCallback(func(t cogito.ToolStatus) {
|
||||
if requiredAvailable && t.Name == requiredTool && requiredToolResultOK(t.Result) {
|
||||
requiredPassed = true
|
||||
}
|
||||
if isInternalCogitoTool(t.Name) {
|
||||
return
|
||||
}
|
||||
@@ -327,6 +354,32 @@ func ExecuteChatWithLLM(ctx context.Context, llm cogito.LLM, cfg *AgentConfig, m
|
||||
return "", fmt.Errorf("agent execution failed: %w", err)
|
||||
}
|
||||
|
||||
for len(result.Messages) > 0 && textFinalizationNeedsRequiredTool(requiredAvailable, requiredPassed,
|
||||
requiredAttempts, maxRequiredAttempts, result.LastMessage().Role, result.LastMessage().Content) {
|
||||
requiredAttempts++
|
||||
xlog.Info("required-tool gate: answer without the required tool, nudging",
|
||||
"agent", cfg.Name, "tool", requiredTool, "attempt", requiredAttempts)
|
||||
answered := result
|
||||
next, err := cogito.ExecuteTools(llm, result.AddMessage(cogito.UserMessageRole, requiredPrompt), cogitoOpts...)
|
||||
if err != nil && ctx.Err() != nil {
|
||||
if cb.OnStatus != nil {
|
||||
cb.OnStatus("error: " + err.Error())
|
||||
}
|
||||
return "", fmt.Errorf("agent execution failed: %w", err)
|
||||
}
|
||||
// A failed retry must not throw away the answer the model already gave.
|
||||
if err != nil && !errors.Is(err, cogito.ErrNoToolSelected) {
|
||||
xlog.Error("required-tool gate: retry failed, keeping the previous answer", "agent", cfg.Name, "error", err)
|
||||
result = answered
|
||||
break
|
||||
}
|
||||
result = next
|
||||
}
|
||||
if requiredAvailable && !requiredPassed && requiredAttempts >= maxRequiredAttempts {
|
||||
xlog.Warn("required-tool gate: bypass after max attempts, answer finalized ungated",
|
||||
"agent", cfg.Name, "tool", requiredTool)
|
||||
}
|
||||
|
||||
// Extract response
|
||||
response := ""
|
||||
if len(result.Messages) > 0 {
|
||||
|
||||
@@ -0,0 +1,220 @@
|
||||
package agents
|
||||
|
||||
// Tool policy for the distributed executor: the allowed/excluded tool lists and
|
||||
// the required-tool-before-finish gate. The semantics mirror LocalAGI's
|
||||
// core/agent/toolfilter.go and the gate in core/agent/agent.go. Those helpers
|
||||
// are unexported there, so the small pieces below are kept in step by hand; the
|
||||
// meta parity spec in toolpolicy_test.go catches drift in the form fields.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
gomcp "github.com/modelcontextprotocol/go-sdk/mcp"
|
||||
"github.com/mudler/cogito"
|
||||
"github.com/mudler/xlog"
|
||||
)
|
||||
|
||||
// ToolNames is a list of tool names. The agent form submits it as a comma or
|
||||
// newline separated string and the API as a JSON array, so both are accepted.
|
||||
type ToolNames []string
|
||||
|
||||
// UnmarshalJSON accepts a JSON array of strings, a comma or newline separated
|
||||
// string, or null. Names are trimmed and empty entries dropped.
|
||||
func (t *ToolNames) UnmarshalJSON(data []byte) error {
|
||||
var value any
|
||||
if err := json.Unmarshal(data, &value); err != nil {
|
||||
return err
|
||||
}
|
||||
var raw []string
|
||||
switch v := value.(type) {
|
||||
case nil:
|
||||
*t = nil
|
||||
return nil
|
||||
case string:
|
||||
raw = strings.FieldsFunc(v, func(r rune) bool { return r == ',' || r == '\n' || r == '\r' })
|
||||
case []any:
|
||||
for _, item := range v {
|
||||
name, ok := item.(string)
|
||||
if !ok {
|
||||
return fmt.Errorf("expected a list of tool names, got %T", item)
|
||||
}
|
||||
raw = append(raw, name)
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("expected a list of tool names or a comma separated string, got %T", value)
|
||||
}
|
||||
var names ToolNames
|
||||
for _, n := range raw {
|
||||
if n = strings.TrimSpace(n); n != "" {
|
||||
names = append(names, n)
|
||||
}
|
||||
}
|
||||
*t = names
|
||||
return nil
|
||||
}
|
||||
|
||||
// controlActionNames are LocalAGI's loop-driving actions. The distributed
|
||||
// executor does not offer them today, but an agent config is shared between
|
||||
// both modes, so the filter must treat them the same way in both.
|
||||
var controlActionNames = map[string]struct{}{
|
||||
"send_message": {},
|
||||
"stop": {},
|
||||
"update_state": {},
|
||||
}
|
||||
|
||||
// toolFilter is an allow/deny list over tool names. A nil *toolFilter allows
|
||||
// everything.
|
||||
type toolFilter struct {
|
||||
allow map[string]struct{}
|
||||
deny map[string]struct{}
|
||||
}
|
||||
|
||||
func newToolFilter(allow, deny []string) *toolFilter {
|
||||
f := &toolFilter{allow: toNameSet(allow), deny: toNameSet(deny)}
|
||||
if len(f.allow) == 0 && len(f.deny) == 0 {
|
||||
return nil
|
||||
}
|
||||
return f
|
||||
}
|
||||
|
||||
func toNameSet(names []string) map[string]struct{} {
|
||||
set := make(map[string]struct{}, len(names))
|
||||
for _, n := range names {
|
||||
if n = strings.TrimSpace(n); n != "" {
|
||||
set[n] = struct{}{}
|
||||
}
|
||||
}
|
||||
return set
|
||||
}
|
||||
|
||||
func (f *toolFilter) allows(name string) bool {
|
||||
if f == nil {
|
||||
return true
|
||||
}
|
||||
if _, ok := controlActionNames[name]; ok {
|
||||
return true
|
||||
}
|
||||
if _, denied := f.deny[name]; denied {
|
||||
return false
|
||||
}
|
||||
if len(f.allow) == 0 {
|
||||
return true
|
||||
}
|
||||
_, allowed := f.allow[name]
|
||||
return allowed
|
||||
}
|
||||
|
||||
func (f *toolFilter) filterTools(tools []cogito.ToolDefinitionInterface) []cogito.ToolDefinitionInterface {
|
||||
if f == nil {
|
||||
return tools
|
||||
}
|
||||
out := make([]cogito.ToolDefinitionInterface, 0, len(tools))
|
||||
for _, t := range tools {
|
||||
if f.allows(t.Tool().Function.Name) {
|
||||
out = append(out, t)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// mcpToolFilter is needed on top of filterTools because cogito discovers MCP
|
||||
// tools straight from the live sessions.
|
||||
func (f *toolFilter) mcpToolFilter() cogito.MCPToolFilter {
|
||||
if f == nil {
|
||||
return nil
|
||||
}
|
||||
return func(_ *gomcp.ClientSession, toolName string) bool {
|
||||
return f.allows(toolName)
|
||||
}
|
||||
}
|
||||
|
||||
// defaultRequiredFinishAttempts bounds the reminders: a gate that can loop
|
||||
// forever is worse than one that gives up loudly.
|
||||
const defaultRequiredFinishAttempts = 3
|
||||
|
||||
func requiredFinishPromptFor(tool, override string) string {
|
||||
if override != "" {
|
||||
return override
|
||||
}
|
||||
return "Before you send your final answer you MUST first call the tool " + tool +
|
||||
" and it must succeed (ok:true). Call " + tool + " now; only send the final " +
|
||||
"message after it passes."
|
||||
}
|
||||
|
||||
// requiredToolResultOK reports whether a tool result is a JSON object with a
|
||||
// top-level "ok": true. When the result is not JSON as a whole (MCP content
|
||||
// may wrap it in text), each top-level object embedded in it is checked.
|
||||
func requiredToolResultOK(result string) bool {
|
||||
trimmed := strings.TrimSpace(result)
|
||||
if json.Valid([]byte(trimmed)) {
|
||||
return jsonObjectOK([]byte(trimmed))
|
||||
}
|
||||
for i := 0; i < len(result); {
|
||||
j := strings.IndexByte(result[i:], '{')
|
||||
if j < 0 {
|
||||
return false
|
||||
}
|
||||
start := i + j
|
||||
dec := json.NewDecoder(strings.NewReader(result[start:]))
|
||||
var raw json.RawMessage
|
||||
if err := dec.Decode(&raw); err != nil {
|
||||
i = start + 1
|
||||
continue
|
||||
}
|
||||
if jsonObjectOK(raw) {
|
||||
return true
|
||||
}
|
||||
// Skip the whole object so its nested objects are not checked on their own.
|
||||
i = start + int(dec.InputOffset())
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func jsonObjectOK(data []byte) bool {
|
||||
var obj map[string]json.RawMessage
|
||||
if err := json.Unmarshal(data, &obj); err != nil {
|
||||
return false
|
||||
}
|
||||
var ok bool
|
||||
if err := json.Unmarshal(obj["ok"], &ok); err != nil {
|
||||
return false
|
||||
}
|
||||
return ok
|
||||
}
|
||||
|
||||
// requiredToolAvailable reports whether the model is offered the required
|
||||
// tool. The gate stays inert otherwise, so a pool-wide setting is harmless for
|
||||
// agents that lack the tool. MCP sessions are only listed when the tool is not
|
||||
// a local one.
|
||||
func requiredToolAvailable(ctx context.Context, name string, local []cogito.ToolDefinitionInterface, sessions []*gomcp.ClientSession, filter *toolFilter) bool {
|
||||
if name == "" || !filter.allows(name) {
|
||||
return false
|
||||
}
|
||||
if cogito.Tools(local).Find(name) != nil {
|
||||
return true
|
||||
}
|
||||
for _, s := range sessions {
|
||||
res, err := s.ListTools(ctx, nil)
|
||||
if err != nil {
|
||||
xlog.Warn("required-tool gate: failed to list MCP tools", "error", err)
|
||||
continue
|
||||
}
|
||||
for _, t := range res.Tools {
|
||||
if t.Name == name {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// textFinalizationNeedsRequiredTool reports whether the run ended with a
|
||||
// non-empty assistant answer although the required tool has not passed and
|
||||
// reminders are left.
|
||||
func textFinalizationNeedsRequiredTool(toolAvailable, toolPassed bool, attempts, max int, lastRole, lastContent string) bool {
|
||||
return toolAvailable && !toolPassed && attempts < max &&
|
||||
lastRole == "assistant" && strings.TrimSpace(lastContent) != ""
|
||||
}
|
||||
@@ -0,0 +1,387 @@
|
||||
package agents
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sort"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/modelcontextprotocol/go-sdk/mcp"
|
||||
"github.com/mudler/LocalAGI/core/state"
|
||||
"github.com/mudler/cogito"
|
||||
openai "github.com/sashabaranov/go-openai"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
// mcpFixture serves an MCP server over SSE whose tools return fixed results,
|
||||
// so the executor reaches it through the same transport as a real agent.
|
||||
type mcpFixture struct {
|
||||
server *httptest.Server
|
||||
calls map[string]*atomic.Int32
|
||||
}
|
||||
|
||||
func newMCPFixture(results map[string]string) *mcpFixture {
|
||||
srv := mcp.NewServer(&mcp.Implementation{Name: "fixture", Version: "v0.0.1"}, nil)
|
||||
fx := &mcpFixture{calls: map[string]*atomic.Int32{}}
|
||||
for name, result := range results {
|
||||
counter := &atomic.Int32{}
|
||||
fx.calls[name] = counter
|
||||
srv.AddTool(&mcp.Tool{
|
||||
Name: name,
|
||||
Description: "fixture tool " + name,
|
||||
InputSchema: json.RawMessage(`{"type":"object","properties":{}}`),
|
||||
}, func(context.Context, *mcp.CallToolRequest) (*mcp.CallToolResult, error) {
|
||||
counter.Add(1)
|
||||
return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: result}}}, nil
|
||||
})
|
||||
}
|
||||
fx.server = httptest.NewServer(mcp.NewSSEHandler(func(*http.Request) *mcp.Server { return srv }, nil))
|
||||
return fx
|
||||
}
|
||||
|
||||
func (fx *mcpFixture) close() { fx.server.Close() }
|
||||
func (fx *mcpFixture) callCount(n string) int32 { return fx.calls[n].Load() }
|
||||
|
||||
// policyLLM answers each chat completion through respond (plain text answer
|
||||
// when respond is nil) and records every
|
||||
// request, so specs can see which tools were offered and which messages the
|
||||
// executor added.
|
||||
type policyLLM struct {
|
||||
mu sync.Mutex
|
||||
requests []openai.ChatCompletionRequest
|
||||
asked [][]openai.ChatCompletionMessage
|
||||
respond func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage
|
||||
answer string
|
||||
}
|
||||
|
||||
func (m *policyLLM) Ask(_ context.Context, f cogito.Fragment) (cogito.Fragment, error) {
|
||||
m.mu.Lock()
|
||||
m.asked = append(m.asked, append([]openai.ChatCompletionMessage(nil), f.Messages...))
|
||||
m.mu.Unlock()
|
||||
return f.AddMessage(cogito.AssistantMessageRole, m.answer), nil
|
||||
}
|
||||
|
||||
func (m *policyLLM) CreateChatCompletion(_ context.Context, req openai.ChatCompletionRequest) (cogito.LLMReply, cogito.LLMUsage, error) {
|
||||
m.mu.Lock()
|
||||
m.requests = append(m.requests, req)
|
||||
m.mu.Unlock()
|
||||
msg := openai.ChatCompletionMessage{Role: "assistant", Content: m.answer}
|
||||
if m.respond != nil {
|
||||
msg = m.respond(req)
|
||||
}
|
||||
return cogito.LLMReply{
|
||||
ChatCompletionResponse: openai.ChatCompletionResponse{
|
||||
Choices: []openai.ChatCompletionChoice{{Message: msg}},
|
||||
},
|
||||
}, cogito.LLMUsage{}, nil
|
||||
}
|
||||
|
||||
// offeredTools returns the sorted tool names of the first request that
|
||||
// offered tools to the model.
|
||||
func (m *policyLLM) offeredTools() []string {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
for _, req := range m.requests {
|
||||
if len(req.Tools) == 0 {
|
||||
continue
|
||||
}
|
||||
names := []string{}
|
||||
for _, t := range req.Tools {
|
||||
if t.Function != nil {
|
||||
names = append(names, t.Function.Name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
return names
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// nudges counts the user messages carrying prompt in the longest conversation
|
||||
// the model saw, which is the number of times the gate sent it.
|
||||
func (m *policyLLM) nudges(prompt string) int {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
best := 0
|
||||
count := func(msgs []openai.ChatCompletionMessage) {
|
||||
n := 0
|
||||
for _, msg := range msgs {
|
||||
if msg.Role == "user" && msg.Content == prompt {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n > best {
|
||||
best = n
|
||||
}
|
||||
}
|
||||
for _, req := range m.requests {
|
||||
count(req.Messages)
|
||||
}
|
||||
for _, msgs := range m.asked {
|
||||
count(msgs)
|
||||
}
|
||||
return best
|
||||
}
|
||||
|
||||
func toolCallMessage(name string) openai.ChatCompletionMessage {
|
||||
return openai.ChatCompletionMessage{
|
||||
Role: "assistant",
|
||||
ToolCalls: []openai.ToolCall{{
|
||||
ID: "call-" + name,
|
||||
Type: openai.ToolTypeFunction,
|
||||
Function: openai.FunctionCall{Name: name, Arguments: `{}`},
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
func lastMessage(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
|
||||
if len(req.Messages) == 0 {
|
||||
return openai.ChatCompletionMessage{}
|
||||
}
|
||||
return req.Messages[len(req.Messages)-1]
|
||||
}
|
||||
|
||||
var _ = Describe("tool policy settings", func() {
|
||||
Describe("config parsing", func() {
|
||||
It("accepts the tool lists as a comma or newline separated string", func() {
|
||||
var cfg AgentConfig
|
||||
Expect(ParseConfigJSON(`{"allowed_tools":"a, b\nc,,","excluded_tools":" d \r\n"}`, &cfg)).To(Succeed())
|
||||
Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b", "c"}))
|
||||
Expect([]string(cfg.ExcludedTools)).To(Equal([]string{"d"}))
|
||||
})
|
||||
|
||||
It("accepts the tool lists as a JSON array", func() {
|
||||
var cfg AgentConfig
|
||||
Expect(ParseConfigJSON(`{"allowed_tools":["a"," b ",""],"excluded_tools":null}`, &cfg)).To(Succeed())
|
||||
Expect([]string(cfg.AllowedTools)).To(Equal([]string{"a", "b"}))
|
||||
Expect(cfg.ExcludedTools).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("rejects a list with non-string entries", func() {
|
||||
var cfg AgentConfig
|
||||
Expect(ParseConfigJSON(`{"allowed_tools":[1]}`, &cfg)).ToNot(Succeed())
|
||||
})
|
||||
|
||||
It("keeps every setting when the config is stored through LocalAGI's config", func() {
|
||||
// The REST handlers decode into state.AgentConfig and store its JSON;
|
||||
// the distributed dispatcher decodes that JSON into AgentConfig.
|
||||
var in state.AgentConfig
|
||||
Expect(json.Unmarshal([]byte(`{
|
||||
"name": "a",
|
||||
"allowed_tools": "search, check_policy",
|
||||
"excluded_tools": ["add_memory"],
|
||||
"required_tool_before_finish": "check_policy",
|
||||
"required_tool_before_finish_prompt": "run it",
|
||||
"required_tool_before_finish_attempts": 4
|
||||
}`), &in)).To(Succeed())
|
||||
stored, err := json.Marshal(in)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
var out AgentConfig
|
||||
Expect(ParseConfigJSON(string(stored), &out)).To(Succeed())
|
||||
Expect([]string(out.AllowedTools)).To(Equal([]string{"search", "check_policy"}))
|
||||
Expect([]string(out.ExcludedTools)).To(Equal([]string{"add_memory"}))
|
||||
Expect(out.RequiredToolBeforeFinish).To(Equal("check_policy"))
|
||||
Expect(out.RequiredToolBeforeFinishPrompt).To(Equal("run it"))
|
||||
Expect(out.RequiredToolBeforeFinishAttempts).To(Equal(4))
|
||||
|
||||
again, err := json.Marshal(out)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
var back state.AgentConfig
|
||||
Expect(json.Unmarshal(again, &back)).To(Succeed())
|
||||
Expect(back.AllowedTools).To(Equal([]string{"search", "check_policy"}))
|
||||
Expect(back.RequiredToolBeforeFinishAttempts).To(Equal(4))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("config meta", func() {
|
||||
It("describes the settings exactly like LocalAGI does", func() {
|
||||
upstream := map[string]ConfigField{}
|
||||
for _, f := range state.NewAgentConfigMeta(nil, nil, nil, nil).Fields {
|
||||
upstream[f.Name] = ConfigField{
|
||||
Name: f.Name, Type: string(f.Type), Label: f.Label, DefaultValue: f.DefaultValue,
|
||||
Placeholder: f.Placeholder, HelpText: f.HelpText, Min: f.Min, Max: f.Max, Step: f.Step,
|
||||
Tags: ConfigFieldTags{Section: f.Tags.Section},
|
||||
}
|
||||
}
|
||||
local := map[string]ConfigField{}
|
||||
for _, f := range DefaultConfigMeta().Fields {
|
||||
local[f.Name] = f
|
||||
}
|
||||
for _, name := range []string{
|
||||
"allowed_tools", "excluded_tools",
|
||||
"required_tool_before_finish", "required_tool_before_finish_prompt", "required_tool_before_finish_attempts",
|
||||
} {
|
||||
Expect(upstream).To(HaveKey(name))
|
||||
Expect(local).To(HaveKeyWithValue(name, upstream[name]), name)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
Describe("tool filter", func() {
|
||||
It("keeps the control actions even when they are excluded or not allowed", func() {
|
||||
f := newToolFilter([]string{"search"}, []string{"send_message", "stop", "update_state", "search"})
|
||||
for _, name := range []string{"send_message", "stop", "update_state"} {
|
||||
Expect(f.allows(name)).To(BeTrue(), name)
|
||||
}
|
||||
Expect(f.allows("search")).To(BeFalse())
|
||||
Expect(f.allows("other")).To(BeFalse())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("ExecuteChatWithLLM", func() {
|
||||
var fx *mcpFixture
|
||||
|
||||
BeforeEach(func() {
|
||||
fx = newMCPFixture(map[string]string{
|
||||
"check_policy": `{"ok":true}`,
|
||||
"mcp_allowed": "allowed result",
|
||||
"mcp_blocked": "blocked result",
|
||||
})
|
||||
})
|
||||
|
||||
AfterEach(func() { fx.close() })
|
||||
|
||||
baseConfig := func() *AgentConfig {
|
||||
return &AgentConfig{
|
||||
Name: "policy-agent",
|
||||
Model: "test-model",
|
||||
MCPServers: []MCPServer{{URL: fx.server.URL}},
|
||||
EnableKnowledgeBase: true,
|
||||
KBMode: KBModeTools,
|
||||
}
|
||||
}
|
||||
|
||||
Context("with allowed and excluded tools", func() {
|
||||
It("offers the model only the allowed tools that are not excluded, MCP tools included", func() {
|
||||
llm := &policyLLM{answer: "final"}
|
||||
cfg := baseConfig()
|
||||
cfg.AllowedTools = ToolNames{"mcp_allowed", "search_memory", "add_memory"}
|
||||
cfg.ExcludedTools = ToolNames{"add_memory"}
|
||||
|
||||
_, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(llm.offeredTools()).To(Equal([]string{"mcp_allowed", "search_memory"}))
|
||||
})
|
||||
|
||||
It("offers every tool when no list is set", func() {
|
||||
llm := &policyLLM{answer: "final"}
|
||||
_, err := ExecuteChatWithLLM(context.Background(), llm, baseConfig(), "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(llm.offeredTools()).To(Equal([]string{"add_memory", "check_policy", "mcp_allowed", "mcp_blocked", "search_memory"}))
|
||||
})
|
||||
|
||||
It("does not run a filtered MCP tool the model calls anyway", func() {
|
||||
var calls atomic.Int32
|
||||
llm := &policyLLM{answer: "final", respond: func(openai.ChatCompletionRequest) openai.ChatCompletionMessage {
|
||||
if calls.Add(1) == 1 {
|
||||
return toolCallMessage("mcp_blocked")
|
||||
}
|
||||
return openai.ChatCompletionMessage{Role: "assistant", Content: "done"}
|
||||
}}
|
||||
cfg := baseConfig()
|
||||
cfg.ExcludedTools = ToolNames{"mcp_blocked"}
|
||||
|
||||
_, _ = ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(fx.callCount("mcp_blocked")).To(BeZero())
|
||||
})
|
||||
})
|
||||
|
||||
Context("with a required tool before finish", func() {
|
||||
const prompt = "RUN check_policy NOW"
|
||||
|
||||
It("nudges the model until the required tool passes, then returns its answer", func() {
|
||||
llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
|
||||
if last := lastMessage(req); last.Role == "user" && last.Content == prompt {
|
||||
return toolCallMessage("check_policy")
|
||||
}
|
||||
return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"}
|
||||
}}
|
||||
cfg := baseConfig()
|
||||
cfg.RequiredToolBeforeFinish = "check_policy"
|
||||
cfg.RequiredToolBeforeFinishPrompt = prompt
|
||||
|
||||
result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(result).To(Equal("final answer"))
|
||||
Expect(fx.callCount("check_policy")).To(Equal(int32(1)))
|
||||
Expect(llm.nudges(prompt)).To(Equal(1))
|
||||
})
|
||||
|
||||
It("lets the answer through after the configured number of reminders", func() {
|
||||
llm := &policyLLM{answer: "stubborn answer"}
|
||||
cfg := baseConfig()
|
||||
cfg.RequiredToolBeforeFinish = "check_policy"
|
||||
cfg.RequiredToolBeforeFinishPrompt = prompt
|
||||
cfg.RequiredToolBeforeFinishAttempts = 2
|
||||
|
||||
result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(result).To(Equal("stubborn answer"))
|
||||
Expect(llm.nudges(prompt)).To(Equal(2))
|
||||
Expect(fx.callCount("check_policy")).To(BeZero())
|
||||
})
|
||||
|
||||
It("uses three reminders and a prompt naming the tool by default", func() {
|
||||
llm := &policyLLM{answer: "stubborn answer"}
|
||||
cfg := baseConfig()
|
||||
cfg.RequiredToolBeforeFinish = "check_policy"
|
||||
|
||||
_, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(llm.nudges(requiredFinishPromptFor("check_policy", ""))).To(Equal(3))
|
||||
Expect(requiredFinishPromptFor("check_policy", "")).To(ContainSubstring("check_policy"))
|
||||
})
|
||||
|
||||
It("keeps nudging when the required tool fails", func() {
|
||||
fx.close()
|
||||
fx = newMCPFixture(map[string]string{"check_policy": `{"ok":false}`})
|
||||
llm := &policyLLM{answer: "final answer", respond: func(req openai.ChatCompletionRequest) openai.ChatCompletionMessage {
|
||||
if last := lastMessage(req); last.Role == "user" && last.Content == prompt {
|
||||
return toolCallMessage("check_policy")
|
||||
}
|
||||
return openai.ChatCompletionMessage{Role: "assistant", Content: "final answer"}
|
||||
}}
|
||||
cfg := baseConfig()
|
||||
cfg.RequiredToolBeforeFinish = "check_policy"
|
||||
cfg.RequiredToolBeforeFinishPrompt = prompt
|
||||
cfg.RequiredToolBeforeFinishAttempts = 2
|
||||
|
||||
_, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(fx.callCount("check_policy")).To(Equal(int32(2)))
|
||||
Expect(llm.nudges(prompt)).To(Equal(2))
|
||||
})
|
||||
|
||||
It("does nothing when the agent does not have the required tool", func() {
|
||||
llm := &policyLLM{answer: "final"}
|
||||
cfg := baseConfig()
|
||||
cfg.RequiredToolBeforeFinish = "check_policy"
|
||||
cfg.RequiredToolBeforeFinishPrompt = prompt
|
||||
cfg.ExcludedTools = ToolNames{"check_policy"}
|
||||
|
||||
result, err := ExecuteChatWithLLM(context.Background(), llm, cfg, "hi", Callbacks{})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(result).To(Equal("final"))
|
||||
Expect(llm.nudges(prompt)).To(BeZero())
|
||||
})
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
var _ = DescribeTable("requiredToolResultOK",
|
||||
func(result string, want bool) {
|
||||
Expect(requiredToolResultOK(result)).To(Equal(want))
|
||||
},
|
||||
Entry("top-level ok true", `{"ok":true}`, true),
|
||||
Entry("top-level ok false", `{"ok":false}`, false),
|
||||
Entry("ok as a string", `{"ok":"true"}`, false),
|
||||
Entry("ok nested in another object", `{"data":{"ok":true}}`, false),
|
||||
Entry("object embedded in text", `result: {"ok": true, "n": 1} done`, true),
|
||||
Entry("plain text", `"ok": true`, false),
|
||||
)
|
||||
@@ -1060,7 +1060,7 @@ func (r *SmartRouter) resolveSelectorCandidates(ctx context.Context, modelID str
|
||||
return nil, fmt.Errorf("looking up nodes for selector %s: %w", sched.NodeSelector, err)
|
||||
}
|
||||
if len(candidates) == 0 {
|
||||
return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s", modelID, sched.NodeSelector)
|
||||
return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s: %w", modelID, sched.NodeSelector, ErrNoAvailableNodes)
|
||||
}
|
||||
return extractNodeIDs(candidates), nil
|
||||
}
|
||||
@@ -1273,9 +1273,9 @@ func (r *SmartRouter) scheduleNewModel(ctx context.Context, backendType, modelID
|
||||
evictedNode, evictErr := r.evictLRUAndFreeNodeFrom(ctx, candidateNodeIDs)
|
||||
if evictErr != nil {
|
||||
if errors.Is(evictErr, ErrEvictionBusy) {
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", evictErr)
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", errors.Join(evictErr, ErrNoAvailableNodes))
|
||||
}
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", evictErr)
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", errors.Join(evictErr, ErrNoAvailableNodes))
|
||||
}
|
||||
node = evictedNode
|
||||
}
|
||||
@@ -2247,6 +2247,13 @@ func (r *SmartRouter) EvictLRU(ctx context.Context, nodeID string) (string, erro
|
||||
// and none can be evicted to make room.
|
||||
var ErrEvictionBusy = errors.New("all models busy, cannot evict")
|
||||
|
||||
// ErrNoAvailableNodes is returned when the scheduler cannot find any healthy
|
||||
// node to serve a model — all nodes are full and eviction cannot free a slot,
|
||||
// or a node selector excludes every candidate. The HTTP layer maps this to
|
||||
// 503 so clients treat it as a transient condition rather than a server bug
|
||||
// (which is what 500 would imply).
|
||||
var ErrNoAvailableNodes = errors.New("no available nodes")
|
||||
|
||||
// evictLRUAndFreeNode finds the globally least-recently-used model with zero in-flight,
|
||||
// unloads it, and returns its node for reuse. If all models are busy, retries briefly.
|
||||
//
|
||||
|
||||
@@ -828,6 +828,26 @@ var _ = Describe("SmartRouter", func() {
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no available nodes"))
|
||||
})
|
||||
|
||||
It("wraps ErrNoAvailableNodes when all nodes are full and eviction cannot help", func() {
|
||||
// gorm.ErrRecordNotFound is the registry's verdict that no node
|
||||
// matches — the scheduler then falls through to eviction. With
|
||||
// DB nil, eviction returns ErrEvictionBusy, and the scheduler
|
||||
// wraps the error with ErrNoAvailableNodes so the HTTP layer can
|
||||
// map it to 503 instead of 500.
|
||||
reg.findIdleErr = errors.New("no idle")
|
||||
reg.findLeastLoadedErr = gorm.ErrRecordNotFound
|
||||
|
||||
router := NewSmartRouter(reg, SmartRouterOptions{
|
||||
Unloader: unloader,
|
||||
ClientFactory: factory,
|
||||
})
|
||||
|
||||
_, err := router.Route(context.Background(), "m5", "models/m5.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
Expect(errors.Is(err, ErrEvictionBusy)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("UnloadModel (mock-based)", func() {
|
||||
@@ -955,6 +975,7 @@ var _ = Describe("SmartRouter", func() {
|
||||
_, err := router.Route(context.Background(), "aliased-model", "models/aliased.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
})
|
||||
|
||||
It("returns error when no nodes match selector", func() {
|
||||
@@ -973,6 +994,7 @@ var _ = Describe("SmartRouter", func() {
|
||||
_, err := router.Route(context.Background(), "no-match-model", "models/nomatch.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
})
|
||||
|
||||
It("uses regular methods when model has no scheduling config", func() {
|
||||
|
||||
@@ -627,6 +627,21 @@ func sanitizeQuantModelName(s string) string {
|
||||
return strings.ToLower(s)
|
||||
}
|
||||
|
||||
// inferenceBackendFor returns the backend that can load what a quantization
|
||||
// backend produced.
|
||||
//
|
||||
// The gallery publishes a quantizer as a release channel of the engine that
|
||||
// runs its output: "llama-cpp-quantization" is llama.cpp's quantizer, and the
|
||||
// GGUF it writes is served by "llama-cpp". The suffix is a channel marker and
|
||||
// carries no engine information, so stripping it yields the backend to pin in
|
||||
// the imported model's config. Names that carry no channel suffix (a backend
|
||||
// that both quantizes and serves, such as "rocmfp4") are already the engine
|
||||
// name and pass through unchanged, as do pinned hardware variants
|
||||
// ("rocm-rocmfp4"), which are valid values for a config's `backend:`.
|
||||
func inferenceBackendFor(quantBackend string) string {
|
||||
return strings.TrimSuffix(config.NormalizeBackendName(quantBackend), "-quantization")
|
||||
}
|
||||
|
||||
// ImportModel imports a quantized model into LocalAI asynchronously.
|
||||
func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID string, req schema.QuantizationImportRequest) (string, error) {
|
||||
s.mu.Lock()
|
||||
@@ -725,6 +740,17 @@ func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID str
|
||||
|
||||
cfg.Name = modelName
|
||||
|
||||
// The importer detects the file format and defaults to llama-cpp for any
|
||||
// GGUF. That is wrong for a model this service just quantized with a
|
||||
// backend stock llama.cpp cannot read: the job knows which backend
|
||||
// produced the file, so pin that one instead of the detected default.
|
||||
if backend := inferenceBackendFor(job.Backend); backend != "" {
|
||||
cfg.Backend = backend
|
||||
}
|
||||
if job.QuantizationType != "" {
|
||||
cfg.Description = "Quantized model (" + job.QuantizationType + ", GGUF)"
|
||||
}
|
||||
|
||||
// Write YAML config
|
||||
yamlData, err := yaml.Marshal(cfg)
|
||||
if err != nil {
|
||||
|
||||
@@ -353,6 +353,28 @@ var _ = Describe("QuantizationService", func() {
|
||||
})
|
||||
})
|
||||
|
||||
Describe("imported model backend", func() {
|
||||
It("strips the quantization channel suffix so the config pins the serving engine", func() {
|
||||
Expect(inferenceBackendFor("llama-cpp-quantization")).To(Equal("llama-cpp"))
|
||||
})
|
||||
|
||||
It("leaves a backend that both quantizes and serves unchanged", func() {
|
||||
Expect(inferenceBackendFor("rocmfp4")).To(Equal("rocmfp4"))
|
||||
})
|
||||
|
||||
It("keeps a pinned hardware variant, which is a valid backend value", func() {
|
||||
Expect(inferenceBackendFor("rocm-rocmfp4-quantization")).To(Equal("rocm-rocmfp4"))
|
||||
})
|
||||
|
||||
It("normalizes dots the way gallery names are written", func() {
|
||||
Expect(inferenceBackendFor("llama.cpp-quantization")).To(Equal("llama-cpp"))
|
||||
})
|
||||
|
||||
It("returns empty for an unset backend so the detected default is kept", func() {
|
||||
Expect(inferenceBackendFor("")).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("compile-time adapter contract", func() {
|
||||
It("satisfies syncstate.Store for *distributed.QuantStore", func() {
|
||||
// Guards against drift between the adapter and the component interface;
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// Package admission is routing-module subsystem 5: per-model
|
||||
// concurrency control + audit. The middleware acquires a slot
|
||||
// before the handler runs; on full, the request gets 503 with
|
||||
// before the handler runs; on full, the request gets 429 with
|
||||
// Retry-After so clients back off rather than pile on. The audit
|
||||
// row goes into the shared event store alongside PII and proxy
|
||||
// rows so admins see a single timeline of routing pressure.
|
||||
|
||||
@@ -30,9 +30,10 @@ import (
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/mudler/xlog"
|
||||
|
||||
"github.com/mudler/LocalAI/core/backend"
|
||||
"github.com/mudler/LocalAI/core/services/routing/router"
|
||||
"github.com/mudler/xlog"
|
||||
)
|
||||
|
||||
// Entry is one labelled exemplar. Vector, EmbeddingModel, and
|
||||
@@ -104,6 +105,9 @@ type storeState struct {
|
||||
syncedFile fileFingerprint
|
||||
needsSync bool
|
||||
indexedEntries int
|
||||
// probe is a vector we inserted ourselves; storeHolds uses it to
|
||||
// tell a live index apart from a relaunched, empty one.
|
||||
probe []float32
|
||||
}
|
||||
|
||||
type cachedStats struct {
|
||||
@@ -172,7 +176,17 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e
|
||||
}
|
||||
delete(m.states, storeName)
|
||||
} else if !state.needsSync && state.syncedFile.equal(fileKey) {
|
||||
return 0, nil
|
||||
if state.indexedEntries == 0 || store == nil || storeHolds(ctx, store, state.probe) {
|
||||
return 0, nil
|
||||
}
|
||||
// The file is unchanged but the live index no longer answers
|
||||
// for a vector we inserted: the store backend was relaunched
|
||||
// (evicted by the active-backend cap or memory pressure, then
|
||||
// started fresh and empty on this request). Fall through and
|
||||
// re-seed it from the file — no re-embedding, the vectors are
|
||||
// persisted.
|
||||
xlog.Warn("router: knn corpus index came back empty, re-seeding from file",
|
||||
"store", storeName, "entries", state.indexedEntries)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,10 +238,40 @@ func (m *Manager) EnsureLoaded(ctx context.Context, storeName, embeddingModel, e
|
||||
embeddingFingerprint: embeddingFingerprint,
|
||||
syncedFile: fileKey,
|
||||
indexedEntries: len(entries),
|
||||
probe: entries[0].Vector,
|
||||
}
|
||||
return len(entries), nil
|
||||
}
|
||||
|
||||
// storeHolds reports whether the live vector index still contains the
|
||||
// corpus. The local-store backend is an in-memory gRPC process: when
|
||||
// the model loader evicts it (active-backend cap, memory pressure) and
|
||||
// relaunches it on the next request, it comes back EMPTY while the
|
||||
// manager still records the file as synced — from then on every probe
|
||||
// routes to the fallback with similarity 0, and corpus/stats keeps
|
||||
// reporting the full count because it reads the file. One nearest-
|
||||
// neighbour lookup with a vector we inserted ourselves tells the two
|
||||
// states apart. An index that cannot answer is treated as empty; the
|
||||
// re-seed that follows surfaces the real error.
|
||||
func storeHolds(ctx context.Context, store backend.VectorStore, probe []float32) bool {
|
||||
if len(probe) == 0 {
|
||||
return true
|
||||
}
|
||||
sim, _, ok, err := store.Search(ctx, probe)
|
||||
return err == nil && ok && sim > 0.999
|
||||
}
|
||||
|
||||
// firstVector returns the vector of the first entry across lists —
|
||||
// the probe storeHolds checks the live index with.
|
||||
func firstVector(lists ...[]Entry) []float32 {
|
||||
for _, l := range lists {
|
||||
if len(l) > 0 {
|
||||
return l[0].Vector
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Add validates, embeds, persists, and indexes new exemplars. Entries
|
||||
// whose text is already in the corpus are skipped (an exemplar's
|
||||
// labels are corrected via Clear + reseed, not silent overwrite).
|
||||
@@ -288,6 +332,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF
|
||||
embeddingFingerprint: embeddingFingerprint,
|
||||
syncedFile: m.fingerprint(storeName),
|
||||
indexedEntries: len(existing),
|
||||
probe: firstVector(existing),
|
||||
}
|
||||
}
|
||||
seen := make(map[string]struct{}, len(existing))
|
||||
@@ -377,6 +422,7 @@ func (m *Manager) Add(ctx context.Context, storeName, embeddingModel, embeddingF
|
||||
embeddingFingerprint: embeddingFingerprint,
|
||||
syncedFile: m.fingerprint(storeName),
|
||||
indexedEntries: entryCount,
|
||||
probe: firstVector(current, added),
|
||||
}
|
||||
} else {
|
||||
m.states[storeName] = storeState{embeddingFingerprint: embeddingFingerprint, needsSync: true, indexedEntries: entryCount}
|
||||
|
||||
@@ -31,17 +31,34 @@ func (e *countingEmbedder) Embed(_ context.Context, text string) ([]float32, err
|
||||
return []float32{float32(len(text)), e.model}, nil
|
||||
}
|
||||
|
||||
// capturingStore records index mutations. Search/SearchK are
|
||||
// irrelevant to the manager and return clean misses.
|
||||
// capturingStore records index mutations. Search answers like a live
|
||||
// index for vectors that were inserted (the manager probes with one of
|
||||
// its own vectors to detect a relaunched, empty store); SearchK is
|
||||
// irrelevant to the manager and returns a clean miss.
|
||||
type capturingStore struct {
|
||||
mu sync.Mutex
|
||||
vecs [][]float32
|
||||
payloads [][]byte
|
||||
batches int
|
||||
deleted [][]float32
|
||||
failBatches int
|
||||
}
|
||||
|
||||
func (s *capturingStore) Search(_ context.Context, _ []float32) (float64, []byte, bool, error) {
|
||||
func (s *capturingStore) Search(_ context.Context, vec []float32) (float64, []byte, bool, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for i, v := range s.vecs {
|
||||
if len(v) == len(vec) && func() bool {
|
||||
for j := range v {
|
||||
if v[j] != vec[j] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}() {
|
||||
return 1, s.payloads[i], true, nil
|
||||
}
|
||||
}
|
||||
return 0, nil, false, nil
|
||||
}
|
||||
|
||||
@@ -49,9 +66,10 @@ func (s *capturingStore) SearchK(_ context.Context, _ []float32, _ int) ([]backe
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func (s *capturingStore) Insert(_ context.Context, _ []float32, payload []byte) error {
|
||||
func (s *capturingStore) Insert(_ context.Context, vec []float32, payload []byte) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.vecs = append(s.vecs, vec)
|
||||
s.payloads = append(s.payloads, payload)
|
||||
return nil
|
||||
}
|
||||
@@ -64,8 +82,8 @@ func (s *capturingStore) InsertBatch(_ context.Context, vecs [][]float32, payloa
|
||||
s.failBatches--
|
||||
return errors.New("transient batch failure")
|
||||
}
|
||||
s.vecs = append(s.vecs, vecs...)
|
||||
s.payloads = append(s.payloads, payloads...)
|
||||
_ = vecs
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -117,6 +135,30 @@ var _ = Describe("corpus.Manager", func() {
|
||||
_ = os.RemoveAll(dir)
|
||||
})
|
||||
|
||||
It("re-seeds the index when the store comes back empty under an unchanged file", func() {
|
||||
// The local-store backend is an in-memory process the model loader
|
||||
// may evict (active-backend cap) and relaunch empty on the next
|
||||
// request. The file is untouched, so the file fingerprint alone
|
||||
// says "synced" — and the router goes blind: every probe falls back
|
||||
// with similarity 0 while corpus/stats still reports the full count.
|
||||
_, _, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
n, err := mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, store)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(n).To(Equal(0), "live index holds the corpus: nothing to do")
|
||||
|
||||
relaunched := &capturingStore{}
|
||||
n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(n).To(Equal(len(seed)), "empty index under an unchanged file is re-seeded")
|
||||
Expect(relaunched.payloads).To(HaveLen(len(seed)))
|
||||
Expect(embedder.calls).To(Equal(len(seed)), "vectors come from the file, nothing is re-embedded")
|
||||
|
||||
n, err = mgr.EnsureLoaded(ctx, storeName, "embed-1", fingerprint, embedder, relaunched)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(n).To(Equal(0), "and the relaunched index counts as synced again")
|
||||
})
|
||||
|
||||
It("adds entries: embeds, persists, and indexes them", func() {
|
||||
added, skipped, err := mgr.Add(ctx, storeName, "embed-1", fingerprint, embedder, store, seed)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
@@ -109,7 +109,7 @@ const (
|
||||
// model's MaxConcurrent ceiling is full. The Host field carries
|
||||
// the model name (overloading the existing column rather than
|
||||
// adding a new one — admins read it as "the thing that was
|
||||
// busy"); StatusCode is 503.
|
||||
// busy"); StatusCode is 429.
|
||||
KindAdmission EventKind = "admission"
|
||||
)
|
||||
|
||||
|
||||
+2
-1
@@ -59,6 +59,7 @@ services:
|
||||
# capabilities: [gpu, utility]
|
||||
#
|
||||
# For legacy NVIDIA driver (for older NVIDIA Container Toolkit):
|
||||
# Request compute for CUDA libraries (libcuda.so.1) and utility for NVML.
|
||||
# environment:
|
||||
# NVIDIA_DRIVER_CAPABILITIES: "compute,utility"
|
||||
# init: true
|
||||
@@ -68,7 +69,7 @@ services:
|
||||
# devices:
|
||||
# - driver: nvidia
|
||||
# count: 1
|
||||
# capabilities: [gpu, utility]
|
||||
# capabilities: [gpu, compute, utility]
|
||||
|
||||
## Uncomment for PostgreSQL-backed knowledge base (see Agents docs)
|
||||
# postgres:
|
||||
|
||||
@@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model
|
||||
local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master
|
||||
```
|
||||
|
||||
See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files.
|
||||
See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files.
|
||||
|
||||
### Prompt templates
|
||||
|
||||
|
||||
@@ -188,6 +188,8 @@ Each agent has its own configuration that controls its behavior. Key settings in
|
||||
- **Connectors** - external integrations (Slack, Discord, etc.)
|
||||
- **Knowledge Base** - collections of documents for RAG
|
||||
- **MCP Servers** - Model Context Protocol servers for additional tool access
|
||||
- **Allowed / Excluded Tools** (`allowed_tools`, `excluded_tools`) - limit the tools the agent can see, including MCP tools. The agent always keeps its control actions (`send_message`, `stop`, `update_state`). If a tool is in both lists, it is excluded.
|
||||
- **Required Tool Before Finish** (`required_tool_before_finish`) - a tool the agent must call successfully before it can give its final answer, for example a validation or policy check. `required_tool_before_finish_prompt` changes the reminder the model gets when it tries to finish early. `required_tool_before_finish_attempts` sets how many reminders it gets before the answer goes through anyway (default 3).
|
||||
|
||||
The pool-level defaults (API URL, API key, models) can be set via environment variables. Individual agents can further override these in their configuration, allowing them to use different LLM providers (OpenAI, other LocalAI instances, etc.) on a per-agent basis.
|
||||
|
||||
|
||||
@@ -141,6 +141,11 @@ curl http://localhost:8080/api/instructions/config-management?format=json
|
||||
|
||||
An additive, LocalAI-specific superset of `/v1/models`. It returns the same set of models but enriches each entry with the **capabilities** the model supports and the **input/output modalities** it accepts and produces. Use it to decide, before sending a request, whether a given model can take an image, audio, or video attachment directly - or whether the input needs converting/transcribing first.
|
||||
|
||||
The reported `context_size` uses a positive model-level value first.
|
||||
If the model does not set `context_size`, it uses **Settings → Performance → Default Context Size** when positive.
|
||||
Otherwise, it uses the backend fallback of 4096 tokens.
|
||||
For llama.cpp with separate KV caches, the reported value accounts for the number of parallel slots.
|
||||
|
||||
Because it is purely additive, clients that only understand `/v1/models` keep working unchanged; they simply never call this route.
|
||||
|
||||
```bash
|
||||
|
||||
@@ -82,6 +82,8 @@ tags:
|
||||
|
||||
### Verifying OCI Backends
|
||||
|
||||
The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details.
|
||||
|
||||
Backend galleries can require keyless Sigstore signatures for every OCI image
|
||||
they provide. Add a `verification` policy to the gallery configuration, then
|
||||
enable strict integrity mode:
|
||||
|
||||
@@ -952,8 +952,12 @@ usage is reported back to the frontend:
|
||||
NVML library (and therefore `nvidia-smi`) is not available inside the
|
||||
container. CUDA compute still works, but the worker cannot query free VRAM
|
||||
and the Nodes page will show the node as fully used. Set
|
||||
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI
|
||||
runtime, list `capabilities: [gpu, utility]` on the device reservation).
|
||||
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime.
|
||||
For Docker Compose with `driver: nvidia`, use
|
||||
`capabilities: [gpu, compute, utility]` on the device reservation.
|
||||
Docker derives driver capabilities from this reservation, so include `compute`
|
||||
for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone
|
||||
enables monitoring but does not provide CUDA libraries.
|
||||
|
||||
- **Run the container with `init: true` (or `docker run --init`).** The
|
||||
worker process becomes PID 1 in the container and cannot reap zombies on
|
||||
|
||||
@@ -39,6 +39,61 @@ Both views use the same model selection and store the view, search, filter, and
|
||||
selection in the URL. Installing from Explore does not move you away from the
|
||||
catalog; the entry updates in place when the operation finishes.
|
||||
|
||||
## Cyber-Tiel-Coder
|
||||
|
||||
Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp.
|
||||
The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector.
|
||||
To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`.
|
||||
Both configurations use the embedded chat template and default to 32,768 context tokens.
|
||||
The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license.
|
||||
|
||||
## Qwen3.8-27B Agention Precision
|
||||
|
||||
The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of
|
||||
Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image
|
||||
input and use a 32,768-token context by default.
|
||||
|
||||
Install with automatic variant selection:
|
||||
|
||||
```bash
|
||||
local-ai models install qwen3.8-27b-agention-iq4-xs
|
||||
```
|
||||
|
||||
To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs`
|
||||
or `--variant qwen3.8-27b-agention-q4-k-m` to the same command.
|
||||
The files use standard llama.cpp quantization types and the Apache-2.0 license.
|
||||
See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF)
|
||||
for quantization details. These entries do not enable MTP speculative decoding.
|
||||
|
||||
## Swift 1.5 Qwen3.8-27B GSQ-RCO
|
||||
|
||||
Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp.
|
||||
The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model.
|
||||
To select IQ3_S explicitly, run:
|
||||
|
||||
```bash
|
||||
local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
|
||||
```
|
||||
|
||||
The configurations use the embedded chat template and default to 32,768 context tokens.
|
||||
These builds support text chat only: the publisher has no verified vision projector for this release.
|
||||
They use standard GGUF files without MTP decoding.
|
||||
See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms.
|
||||
|
||||
## Sharp-Spark-X2.5-4B
|
||||
|
||||
Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp.
|
||||
The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants.
|
||||
To select the publisher's recommended Q6 build, run:
|
||||
|
||||
```bash
|
||||
local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6
|
||||
```
|
||||
|
||||
All builds use a 32,768-token default context and the embedded Sharp-Spark chat template.
|
||||
That template adds a terseness instruction to the system prompt.
|
||||
See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details.
|
||||
|
||||
## MiMo-V2.6-Distill-Qwen-9B
|
||||
|
||||
Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp.
|
||||
@@ -48,6 +103,35 @@ To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9
|
||||
The configurations default to 32,768 context tokens and use the model's embedded chat template.
|
||||
See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details.
|
||||
|
||||
## Qwopus3.8 Flash V2
|
||||
|
||||
Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection:
|
||||
|
||||
```bash
|
||||
local-ai models install qwopus3.8-27b-flash-v2
|
||||
local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8
|
||||
```
|
||||
|
||||
Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector.
|
||||
Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256.
|
||||
This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks.
|
||||
See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations.
|
||||
|
||||
## ThinkingCap Qwen3.8-27B
|
||||
|
||||
Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input.
|
||||
The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector.
|
||||
LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly:
|
||||
|
||||
```bash
|
||||
local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8
|
||||
```
|
||||
|
||||
Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings.
|
||||
MTP speculative decoding is not enabled by these entries.
|
||||
The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE).
|
||||
Review that license for permitted use.
|
||||
|
||||
## Hemmingway-1
|
||||
|
||||
Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants.
|
||||
@@ -97,7 +181,7 @@ To use a gallery that needs authentication, such as a private GitHub repository
|
||||
|
||||
A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted.
|
||||
|
||||
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
|
||||
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
|
||||
|
||||
```json
|
||||
GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
|
||||
@@ -151,10 +235,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out
|
||||
|
||||
### Signature verification
|
||||
|
||||
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
|
||||
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
|
||||
|
||||
```json
|
||||
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
|
||||
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
|
||||
```
|
||||
|
||||
The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends.
|
||||
@@ -173,9 +257,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK
|
||||
The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}).
|
||||
|
||||
{{% notice warning %}}
|
||||
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
|
||||
`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities.
|
||||
|
||||
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
|
||||
{{% /notice %}}
|
||||
|
||||
### Official gallery publishing
|
||||
|
||||
The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA.
|
||||
|
||||
To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title.
|
||||
|
||||
### Private registries
|
||||
|
||||
A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use:
|
||||
@@ -207,10 +299,10 @@ GALLERIES=[{"name":"<GALLERY_NAME>", "url":"<GALLERY_URL"}]
|
||||
For example, to spell out the default `localai` repository, you can start `local-ai` with:
|
||||
|
||||
```
|
||||
GALLERIES=[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
|
||||
GALLERIES=[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]
|
||||
```
|
||||
|
||||
`https://index.localai.io/models` is a caching mirror of the same index file, and the `github:` entry is the fallback used whenever it cannot be reached. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
|
||||
LocalAI tries `https://index.localai.io/models` first, GitHub second, and the signed OCI gallery last. The OCI artifact includes the repository-local base configurations, so reading those configurations does not require GitHub. Model weights and external URLs still require their original hosts. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
|
||||
|
||||
Note: the url are expanded automatically for `github` and `huggingface`, however `https://` and `http://` prefix works as well.
|
||||
|
||||
|
||||
@@ -340,6 +340,15 @@ curl http://localhost:8080/v1/responses \
|
||||
}'
|
||||
```
|
||||
|
||||
#### Streaming responses
|
||||
|
||||
Set `"stream": true` to receive Server-Sent Events. Each `response.output_item.added` event assigns an `output_index` to an item.
|
||||
Use that index and the item ID to associate later deltas and completion events with the same item.
|
||||
|
||||
If a request without explicit tools produces reasoning, the stream uses separate items for reasoning and answer text.
|
||||
Each item keeps its original index throughout the stream.
|
||||
The `response.completed` event includes both items in the same index order, followed by any automatically parsed tool calls.
|
||||
|
||||
#### Background Processing
|
||||
|
||||
Run requests in the background for long-running tasks:
|
||||
@@ -435,6 +444,11 @@ curl http://localhost:8080/v1/responses \
|
||||
}'
|
||||
```
|
||||
|
||||
For streaming requests with JSON tool output, LocalAI waits for the complete JSON
|
||||
object before emitting a completed `function_call` item. Arguments can span
|
||||
multiple tokens. Read the arguments from the `response.output_item.done` event
|
||||
before executing the tool.
|
||||
|
||||
#### Reasoning Configuration
|
||||
|
||||
Configure reasoning effort and summary style:
|
||||
|
||||
@@ -30,6 +30,10 @@ To install the dependencies follow the instructions below:
|
||||
{{< tabs >}}
|
||||
{{% tab title="Apple" %}}
|
||||
|
||||
To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later.
|
||||
Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts.
|
||||
Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables.
|
||||
|
||||
Install `xcode` from the App Store
|
||||
|
||||
```bash
|
||||
|
||||
@@ -391,6 +391,8 @@ See the [Model Configuration]({{% relref "advanced/model-configuration" %}}) gui
|
||||
|
||||
### List Installed Models
|
||||
|
||||
Ollama clients can list configured models with `GET /api/tags` and loaded models with `GET /api/ps`. Each entry includes `size` in bytes when LocalAI can resolve a non-empty primary weights file on disk. This is the size of that file, not the total size of a multi-file model or its memory use. Unknown sizes are omitted; `/api/ps` also omits `size_vram` because per-model VRAM use is not available.
|
||||
|
||||
```bash
|
||||
# Via API
|
||||
curl http://localhost:8080/v1/models
|
||||
|
||||
@@ -60,9 +60,9 @@ against - and two modes:
|
||||
`proxy.provider` selects the auth scheme and (in translate mode) the wire
|
||||
format. Supported values: `openai`, `anthropic`.
|
||||
|
||||
API keys are loaded from either an environment variable (`api_key_env`) or a
|
||||
file (`api_key_file`). The key never appears in the config file or the admin
|
||||
UI; pick whichever fits your secret-management setup.
|
||||
If the upstream requires an API key, configure either an environment variable
|
||||
(`api_key_env`) or a file (`api_key_file`). The key never appears in the config
|
||||
file or the admin UI. If the upstream requires no API key, omit both fields.
|
||||
|
||||
### OpenAI passthrough
|
||||
|
||||
@@ -126,7 +126,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with
|
||||
|
||||
Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak
|
||||
the OpenAI chat-completions wire format. Use `provider: openai` with the
|
||||
provider's URL and API key:
|
||||
provider's URL and, if required, its API key:
|
||||
|
||||
```yaml
|
||||
name: llama-3-70b-via-together
|
||||
@@ -140,6 +140,37 @@ proxy:
|
||||
upstream_model: meta-llama/Llama-3-70b-chat-hf
|
||||
```
|
||||
|
||||
### Upstreams without an API key
|
||||
|
||||
For an OpenAI-compatible upstream that accepts requests without authentication,
|
||||
omit both `api_key_env` and `api_key_file`:
|
||||
|
||||
```yaml
|
||||
name: internal-chat-proxy
|
||||
backend: cloud-proxy
|
||||
|
||||
proxy:
|
||||
mode: passthrough
|
||||
provider: openai
|
||||
upstream_url: http://inference.internal:8000/v1/chat/completions
|
||||
upstream_model: my-model
|
||||
```
|
||||
|
||||
Replace the example URL and model name with your upstream's values. LocalAI
|
||||
loads this configuration without resolving a key and adds no upstream
|
||||
`Authorization` header. This also applies to OpenAI-compatible upstreams in
|
||||
translate mode.
|
||||
|
||||
Omitting both fields differs from setting `api_key_env` to an empty or unset
|
||||
environment variable: the latter causes a backend load error.
|
||||
|
||||
LocalAI's client authentication is separate. Clients must still authenticate
|
||||
to LocalAI when its authentication is enabled. LocalAI does not forward their
|
||||
`Authorization` header to the upstream.
|
||||
|
||||
An upstream without API keys can still require another authentication or
|
||||
payment protocol. Omitting these fields does not implement that protocol.
|
||||
|
||||
### Translate mode
|
||||
|
||||
In translate mode the cloud-proxy backend converts LocalAI's internal proto
|
||||
|
||||
@@ -558,7 +558,12 @@ The corpus is persisted as one JSONL file per router under
|
||||
`<data path>/router-corpus/` (text, labels, vector, embedding-model name,
|
||||
and embedding fingerprint) — **the file is the source of truth** and
|
||||
survives restarts; the local-store index is rebuilt from it at classifier
|
||||
build time without re-embedding. The fingerprint follows the effective
|
||||
build time without re-embedding. Before each KNN lookup, LocalAI checks a stored
|
||||
vector against the live index. If the store restarts empty after eviction or
|
||||
an idle timeout, LocalAI restores the index from the file without re-embedding.
|
||||
A synchronization error fails the lookup.
|
||||
|
||||
The fingerprint follows the effective
|
||||
embedding-model config and local artifact identity, so changing the model or
|
||||
replacing its local weights re-embeds the corpus on the next process load.
|
||||
For remote embedding services whose weights can change invisibly, bump
|
||||
|
||||
@@ -88,7 +88,9 @@ The `/v1/responses` endpoint returns errors with this structure:
|
||||
| 404 | Not Found | Model or resource does not exist |
|
||||
| 409 | Conflict | Resource already exists (e.g., duplicate token) |
|
||||
| 422 | Unprocessable Entity | Validation failed (e.g., invalid parameter range) |
|
||||
| 429 | Too Many Requests | All backends are saturated (per-model `max_concurrent` or process-wide `--max-concurrent-backend-requests` ceiling reached). Includes a `Retry-After` header and `type: "rate_limit_error"` so OpenAI-compatible clients and harnesses back off automatically |
|
||||
| 500 | Internal Server Error | Backend inference failure, unexpected server errors |
|
||||
| 503 | Service Unavailable | No healthy node available to serve the model (cluster is full, eviction cannot free a slot, or a `node_selector` excludes all candidates). Also used during model-load cooldown and while a model is still cold-loading. Retryable |
|
||||
|
||||
## Global Error Handling
|
||||
|
||||
|
||||
@@ -95,7 +95,7 @@ For more information on VRAM management, see [VRAM and Memory Management]({{%rel
|
||||
| Parameter | Default | Description | Environment Variable |
|
||||
|-----------|---------|-------------|----------------------|
|
||||
| `--address` | `:8080` | Bind address for the API server | `$LOCALAI_ADDRESS`, `$ADDRESS` |
|
||||
| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 503 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
|
||||
| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 429 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
|
||||
| `--cors` | `false` | Enable CORS (Cross-Origin Resource Sharing) | `$LOCALAI_CORS`, `$CORS` |
|
||||
| `--cors-allow-origins` | | Comma-separated list of allowed CORS origins | `$LOCALAI_CORS_ALLOW_ORIGINS`, `$CORS_ALLOW_ORIGINS` |
|
||||
| `--disable-csrf` | `false` | Disable CSRF middleware (enabled by default) | `$LOCALAI_DISABLE_CSRF` |
|
||||
|
||||
@@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things:
|
||||
NVML work inside the container. With `--gpus all` alone (or
|
||||
`--runtime nvidia` without extra flags) only `compute` is wired in on
|
||||
some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility`
|
||||
to your `docker run`, or `capabilities: [gpu, utility]` in compose /
|
||||
Kubernetes device reservations.
|
||||
to your `docker run`. For Docker Compose with `driver: nvidia`, use
|
||||
`capabilities: [gpu, compute, utility]` on the device reservation.
|
||||
Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility`
|
||||
alone only provides monitoring libraries and tools.
|
||||
2. Pass `--init` to `docker run` (or `init: true` in compose) so the
|
||||
container has a proper PID 1 reaper - otherwise short-lived child
|
||||
processes like `nvidia-smi` can intermittently fail with
|
||||
|
||||
@@ -21,8 +21,9 @@ The left column is the literal string as it appears in the LocalAI server log (o
|
||||
| `grpc service not ready` | The backend process was spawned but its gRPC server did not become healthy in time (slow start, crash on startup, or the process died while loading). When a local backend has already exited, the error includes its exit code and last stderr line. | Use the included stderr diagnostic when present; otherwise check the log lines just above. A crash here often means out of memory, a missing shared library, or an incompatible CPU (see `SIGILL`). Increase available RAM/VRAM or pick a smaller quantization. |
|
||||
| `failed to load model: ...` | Returned by the load endpoints and several feature paths (voice, realtime, audio transform) when the model config could not be resolved or the backend load failed. | Confirm the model name exists (`local-ai models list`) and its YAML is valid. The trailing text carries the specific reason. |
|
||||
| HTTP `503` with a `Retry-After` header, after a load failed | Model-load failure cooldown. After a model fails to load, LocalAI refuses new load attempts for that model for a short window so a client that keeps polling a broken model does not respawn a crashing backend on every request. The window starts at `--model-load-failure-cooldown` (default `10s`) and doubles per consecutive failure up to 5m; it resets on the first success. | Fix the underlying load failure (see the rows above), then wait out the `Retry-After` seconds before retrying, or restart LocalAI to clear the cooldown. Set `--model-load-failure-cooldown 0` (or `LOCALAI_MODEL_LOAD_FAILURE_COOLDOWN=0`) to disable the cooldown entirely. See {{% relref "reference/cli-reference" %}}. |
|
||||
| HTTP `503` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `503` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
|
||||
| HTTP `503` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
|
||||
| HTTP `503` with `no available nodes` or `no healthy nodes match selector` | The scheduler could not find any healthy node to serve the model. All nodes are full and eviction cannot free a slot, or a `node_selector` in the model's scheduling config excludes every candidate. | Retry after a node becomes available or an in-flight request completes and frees a slot. In a cluster, add nodes or replicas. If a selector is set, confirm at least one healthy node matches it. |
|
||||
| HTTP `429` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `429` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
|
||||
| HTTP `429` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
|
||||
| `invalid pitch` (with CUDA) | The prompt exceeded the model's context size. | Reduce the prompt length, or raise the model's context size (`context_size:` in the model YAML). |
|
||||
| `SIGILL` (illegal instruction) on startup | The prebuilt backend binary uses CPU instructions your CPU does not have (for example AVX512, AVX2, F16C, FMA). | Rebuild the backend for your CPU. In a container, set `REBUILD=true` and disable the unsupported instructions, for example `CMAKE_ARGS="-DGGML_F16C=OFF -DGGML_AVX512=OFF -DGGML_AVX2=OFF -DGGML_FMA=OFF" make build`. |
|
||||
| CUDA / VRAM out of memory (backend log shows `out of memory`, `CUDA error: out of memory`, or the process is killed loading) | The model plus its KV cache does not fit in GPU memory. | Use a smaller quantization, reduce `context_size:`, offload fewer layers to the GPU (lower `gpu_layers:`), or free VRAM held by other processes. On multi-GPU hosts, confirm the model is not trying to load entirely onto one device. |
|
||||
|
||||
@@ -28,6 +28,29 @@ Returns available backends and currently loaded models.
|
||||
| `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM |
|
||||
| `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against |
|
||||
| `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) |
|
||||
| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes |
|
||||
|
||||
### Per-model VRAM
|
||||
|
||||
On Linux, `size_vram` reports resident device memory for the local backend
|
||||
process and its child processes. LocalAI reads `drm-resident-local*` and
|
||||
`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU.
|
||||
Host-memory regions are excluded. The reading includes buffers attributed
|
||||
to the backend, without separating weights, KV cache, and other allocations.
|
||||
See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html)
|
||||
for these counters.
|
||||
|
||||
The field is omitted when accounting is unavailable or incomplete. This
|
||||
includes external and distributed backends, macOS, proprietary NVIDIA
|
||||
drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters,
|
||||
and unreadable process information.
|
||||
A present value of `0` means the supported counters report zero bytes.
|
||||
Treat an absent field as unknown.
|
||||
|
||||
This is a snapshot of driver accounting, not a memory reservation. Shared
|
||||
buffers can appear in different clients' counters, and allocations can change
|
||||
during collection. Do not treat the sum across models as exclusive physical
|
||||
GPU usage. These readings do not replace capacity checks when scheduling work.
|
||||
|
||||
### Usage
|
||||
|
||||
@@ -49,6 +72,7 @@ curl http://localhost:8080/system
|
||||
{
|
||||
"id": "my-llama-model",
|
||||
"backend": "llama-cpp",
|
||||
"size_vram": 5368709120,
|
||||
"process": {
|
||||
"pid": 48213,
|
||||
"rss_bytes": 5368709120,
|
||||
|
||||
+795
-1
@@ -1,4 +1,96 @@
|
||||
---
|
||||
- name: "ternary-bonsai-2-27b"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
|
||||
- https://github.com/PrismML-Eng/llama.cpp
|
||||
description: |
|
||||
Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary
|
||||
transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits
|
||||
per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a
|
||||
Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's
|
||||
llama.cpp fork) instead of stock llama.cpp.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg
|
||||
overrides:
|
||||
backend: bonsai
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3
|
||||
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
- filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903
|
||||
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
- name: "swift-qwen3.8-27b"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B.
|
||||
The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss.
|
||||
This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding.
|
||||
The weights use the Swift Open License v1.0.
|
||||
license: "swift-open-license-1.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
presence_penalty: 1.5
|
||||
repeat_penalty: 1
|
||||
temperature: 0.7
|
||||
top_k: 20
|
||||
top_p: 0.8
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab
|
||||
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
- filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a
|
||||
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
- name: "ornith-1.5-9b-uncensored"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
@@ -205,7 +297,101 @@
|
||||
files:
|
||||
- filename: ds4flash.gguf
|
||||
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
|
||||
sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7
|
||||
sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
|
||||
- name: "qwopus3.8-27b-flash-v2"
|
||||
variants:
|
||||
- model: qwopus3.8-27b-flash-v2-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
|
||||
description: |
|
||||
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
|
||||
workloads. This Q4_K_M GGUF includes the F32 vision projector and uses
|
||||
llama.cpp's embedded chat template with MTP speculative decoding.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3
|
||||
- vision
|
||||
- multimodal
|
||||
- instruction-tuned
|
||||
- reasoning
|
||||
- mtp
|
||||
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e
|
||||
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
|
||||
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
|
||||
- name: "qwopus3.8-27b-flash-v2-q8"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
|
||||
description: |
|
||||
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
|
||||
workloads. This Q8_0 GGUF includes the F32 vision projector and uses
|
||||
llama.cpp's embedded chat template with MTP speculative decoding.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3
|
||||
- vision
|
||||
- multimodal
|
||||
- instruction-tuned
|
||||
- reasoning
|
||||
- mtp
|
||||
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78
|
||||
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
|
||||
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
|
||||
- name: "qwopus3.8-27b-flash"
|
||||
variants:
|
||||
- model: qwopus3.8-27b-flash-q8
|
||||
@@ -388,6 +574,102 @@
|
||||
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
|
||||
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
- name: thinkingcap-qwen3.8-27b
|
||||
variants:
|
||||
- model: thinkingcap-qwen3.8-27b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
|
||||
This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
|
||||
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
|
||||
license: polyform-small-business-1.0.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- vision
|
||||
- multimodal
|
||||
- reasoning
|
||||
last_checked: "2026-09-27"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
files:
|
||||
- filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
- name: thinkingcap-qwen3.8-27b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
|
||||
This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
|
||||
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
|
||||
license: polyform-small-business-1.0.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- vision
|
||||
- multimodal
|
||||
- reasoning
|
||||
last_checked: "2026-09-27"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
files:
|
||||
- filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
- name: hemmingway-1
|
||||
variants:
|
||||
- model: hemmingway-1-q8
|
||||
@@ -4767,6 +5049,110 @@
|
||||
- filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf
|
||||
uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf
|
||||
sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41
|
||||
- name: cyber-tiel-coder-35b-a3b-q4-mtp
|
||||
variants:
|
||||
- model: cyber-tiel-coder-35b-a3b-q8-mtp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
license: mit
|
||||
urls:
|
||||
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
|
||||
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
|
||||
description: |
|
||||
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
|
||||
based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes
|
||||
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- moe
|
||||
- coding
|
||||
- tools
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
parameters:
|
||||
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a
|
||||
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
|
||||
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
|
||||
- name: cyber-tiel-coder-35b-a3b-q8-mtp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
license: mit
|
||||
urls:
|
||||
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
|
||||
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
|
||||
description: |
|
||||
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
|
||||
based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes
|
||||
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- moe
|
||||
- coding
|
||||
- tools
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
parameters:
|
||||
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961
|
||||
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
|
||||
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
|
||||
- &tiel-coder-35b-a3b
|
||||
name: "tiel-coder-35b-a3b-q4"
|
||||
variants:
|
||||
@@ -5247,6 +5633,104 @@
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf
|
||||
uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf
|
||||
sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545
|
||||
- name: qwen3.8-27b-agention-iq4-xs
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
variants:
|
||||
- model: qwen3.8-27b-agention-q4-k-m
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- name: qwen3.8-27b-agention-q4-k-m
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- &qwen3-8-27b
|
||||
name: "qwen3.8-27b-q4"
|
||||
variants:
|
||||
@@ -5533,6 +6017,178 @@
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf
|
||||
uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf
|
||||
sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco"
|
||||
variants:
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
- !!merge <<: *qwen3-8-27b
|
||||
name: "qwen3.8-27b-gsq-rco-iq2-xs"
|
||||
variants: []
|
||||
@@ -5725,6 +6381,120 @@
|
||||
- filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf
|
||||
uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf
|
||||
sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442
|
||||
- name: sharp-spark-x2.5-4b
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
variants:
|
||||
- model: sharp-spark-x2.5-4b-q5
|
||||
- model: sharp-spark-x2.5-4b-q6
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q4_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q5
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q5_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q6
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q6_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
|
||||
|
||||
- &spark-x2-5-4b
|
||||
name: "spark-x2.5-4b-q4"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
@@ -45212,6 +45982,12 @@
|
||||
sha256: ""
|
||||
uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors
|
||||
- name: kimodo-soma-rp
|
||||
variants:
|
||||
- model: kimodo-soma-rp-bf16
|
||||
- model: kimodo-soma-rp-q4_k
|
||||
- model: kimodo-soma-rp-q4_k_m
|
||||
- model: kimodo-soma-rp-q5_k
|
||||
- model: kimodo-soma-rp-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45399,6 +46175,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-soma-seed
|
||||
variants:
|
||||
- model: kimodo-soma-seed-bf16
|
||||
- model: kimodo-soma-seed-q4_k
|
||||
- model: kimodo-soma-seed-q4_k_m
|
||||
- model: kimodo-soma-seed-q5_k
|
||||
- model: kimodo-soma-seed-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45586,6 +46368,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-g1-rp
|
||||
variants:
|
||||
- model: kimodo-g1-rp-bf16
|
||||
- model: kimodo-g1-rp-q4_k
|
||||
- model: kimodo-g1-rp-q4_k_m
|
||||
- model: kimodo-g1-rp-q5_k
|
||||
- model: kimodo-g1-rp-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45773,6 +46561,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-g1-seed
|
||||
variants:
|
||||
- model: kimodo-g1-seed-bf16
|
||||
- model: kimodo-g1-seed-q4_k
|
||||
- model: kimodo-g1-seed-q4_k_m
|
||||
- model: kimodo-g1-seed-q5_k
|
||||
- model: kimodo-g1-seed-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
|
||||
@@ -83,6 +83,7 @@ require (
|
||||
|
||||
require (
|
||||
filippo.io/bigmod v0.1.1-0.20260103110540-f8a47775ebe5 // indirect
|
||||
filippo.io/edwards25519 v1.1.0 // indirect
|
||||
filippo.io/keygen v0.0.0-20260114151900-8e2790ea4c5b // indirect
|
||||
github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 // indirect
|
||||
github.com/atotto/clipboard v0.1.4 // indirect
|
||||
@@ -106,12 +107,10 @@ require (
|
||||
github.com/cenkalti/backoff/v5 v5.0.3 // indirect
|
||||
github.com/charmbracelet/bubbles v0.21.0 // indirect
|
||||
github.com/charmbracelet/bubbletea v1.3.10 // indirect
|
||||
github.com/chasefleming/elem-go v0.30.0 // indirect
|
||||
github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc // indirect
|
||||
github.com/chromedp/chromedp v0.15.1 // indirect
|
||||
github.com/chromedp/sysutil v1.1.0 // indirect
|
||||
github.com/cyberphone/json-canonicalization v0.0.0-20241213102144-19d51d7fe467 // indirect
|
||||
github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 // indirect
|
||||
github.com/digitorus/pkcs7 v0.0.0-20230818184609-3a137a874352 // indirect
|
||||
github.com/digitorus/timestamp v0.0.0-20231217203849-220c5c2851b7 // indirect
|
||||
github.com/dunglas/httpsfv v1.1.0 // indirect
|
||||
@@ -140,21 +139,17 @@ require (
|
||||
github.com/gobwas/httphead v0.1.0 // indirect
|
||||
github.com/gobwas/pool v0.2.1 // indirect
|
||||
github.com/gobwas/ws v1.4.0 // indirect
|
||||
github.com/gofiber/template v1.8.3 // indirect
|
||||
github.com/gofiber/template/html/v2 v2.1.3 // indirect
|
||||
github.com/gofiber/utils v1.1.0 // indirect
|
||||
github.com/google/certificate-transparency-go v1.3.2 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0 // indirect
|
||||
github.com/in-toto/attestation v1.1.2 // indirect
|
||||
github.com/in-toto/in-toto-golang v0.9.0 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/invopop/jsonschema v0.13.0 // indirect
|
||||
github.com/jinzhu/inflection v1.0.0 // indirect
|
||||
github.com/jinzhu/now v1.1.5 // indirect
|
||||
github.com/jolestar/go-commons-pool/v2 v2.1.2 // indirect
|
||||
github.com/klippa-app/go-pdfium v1.19.2 // indirect
|
||||
github.com/mattn/go-localereader v0.0.1 // indirect
|
||||
github.com/mattn/go-sqlite3 v1.14.28 // indirect
|
||||
github.com/mattn/go-sqlite3 v1.14.32 // indirect
|
||||
github.com/moby/moby/api v1.54.2 // indirect
|
||||
github.com/moby/moby/client v0.4.1 // indirect
|
||||
github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect
|
||||
@@ -167,8 +162,6 @@ require (
|
||||
github.com/sigstore/rekor-tiles/v2 v2.0.1 // indirect
|
||||
github.com/sigstore/sigstore v1.10.0 // indirect
|
||||
github.com/sigstore/timestamp-authority/v2 v2.0.3 // indirect
|
||||
github.com/spf13/cobra v1.10.2 // indirect
|
||||
github.com/spf13/pflag v1.0.10 // indirect
|
||||
github.com/standard-webhooks/standard-webhooks/libraries v0.0.0-20260508151727-1282bb917829 // indirect
|
||||
github.com/stretchr/testify v1.11.1 // indirect
|
||||
github.com/sv-tools/openapi v0.2.1 // indirect
|
||||
@@ -242,7 +235,7 @@ require (
|
||||
github.com/kevinburke/ssh_config v1.2.0 // indirect
|
||||
github.com/labstack/gommon v0.4.2 // indirect
|
||||
github.com/mschoch/smat v0.2.0 // indirect
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1
|
||||
github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca
|
||||
github.com/mudler/localrecall v0.6.5 // indirect
|
||||
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145
|
||||
github.com/olekukonko/tablewriter v0.0.5 // indirect
|
||||
@@ -250,7 +243,7 @@ require (
|
||||
github.com/philippgille/chromem-go v0.7.0 // indirect
|
||||
github.com/pion/transport/v4 v4.0.1 // indirect
|
||||
github.com/pjbgf/sha1cd v0.6.0 // indirect
|
||||
github.com/rs/zerolog v1.31.0 // indirect
|
||||
github.com/rs/zerolog v1.34.0 // indirect
|
||||
github.com/saintfish/chardet v0.0.0-20230101081208-5e3ef4b5456d // indirect
|
||||
github.com/segmentio/asm v1.1.3 // indirect
|
||||
github.com/segmentio/encoding v0.5.4 // indirect
|
||||
@@ -270,14 +263,13 @@ require (
|
||||
github.com/valyala/fasttemplate v1.2.2 // indirect
|
||||
github.com/xanzy/ssh-agent v0.3.3 // indirect
|
||||
go.etcd.io/bbolt v1.4.3 // indirect
|
||||
go.mau.fi/util v0.3.0 // indirect
|
||||
go.mau.fi/util v0.9.2 // indirect
|
||||
go.starlark.net v0.0.0-20250417143717-f57e51f710eb // indirect
|
||||
google.golang.org/appengine v1.6.8 // indirect
|
||||
gopkg.in/warnings.v0 v0.1.2 // indirect
|
||||
gopkg.in/yaml.v2 v2.4.0 // indirect
|
||||
jaytaylor.com/html2text v0.0.0-20230321000545-74c2419ad056 // indirect
|
||||
maunium.net/go/maulogger/v2 v2.4.1 // indirect
|
||||
maunium.net/go/mautrix v0.17.0 // indirect
|
||||
maunium.net/go/mautrix v0.25.2 // indirect
|
||||
mvdan.cc/xurls/v2 v2.6.0 // indirect
|
||||
)
|
||||
|
||||
|
||||
@@ -277,8 +277,6 @@ github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf h1:rLG0Y
|
||||
github.com/charmbracelet/x/exp/slice v0.0.0-20250327172914-2fdc97757edf/go.mod h1:B3UgsnsBZS/eX42BlaNiJkD1pPOUa+oF1IYC6Yd2CEU=
|
||||
github.com/charmbracelet/x/term v0.2.1 h1:AQeHeLZ1OqSXhrAWpYUtZyX1T3zVxfpZuEQMIQaGIAQ=
|
||||
github.com/charmbracelet/x/term v0.2.1/go.mod h1:oQ4enTYFV7QN4m0i9mzHrViD7TQKvNEEkHUMCmsxdUg=
|
||||
github.com/chasefleming/elem-go v0.30.0 h1:BlhV1ekv1RbFiM8XZUQeln1Ikb4D+bu2eDO4agREvok=
|
||||
github.com/chasefleming/elem-go v0.30.0/go.mod h1:hz73qILBIKnTgOujnSMtEj20/epI+f6vg71RUilJAA4=
|
||||
github.com/chengxilo/virtualterm v1.0.4 h1:Z6IpERbRVlfB8WkOmtbHiDbBANU7cimRIof7mk9/PwM=
|
||||
github.com/chengxilo/virtualterm v1.0.4/go.mod h1:DyxxBZz/x1iqJjFxTFcr6/x+jSpqN0iwWCOK1q10rlY=
|
||||
github.com/chromedp/cdproto v0.0.0-20260321001828-e3e3800016bc h1:wkN/LMi5vc60pBRWx6qpbk/aEvq3/ZVNpnMvsw8PVVU=
|
||||
@@ -321,7 +319,6 @@ github.com/cpuguy83/dockercfg v0.3.2 h1:DlJTyZGBDlXqUZ2Dk2Q3xHs/FtnooJJVaad2S9GK
|
||||
github.com/cpuguy83/dockercfg v0.3.2/go.mod h1:sugsbF4//dDlL/i+S+rtpIWp+5h0BHJHfjj5/jFyUJc=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.0-20190314233015-f79a8a8ca69d/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.0/go.mod h1:maD7wRr/U5Z6m/iR4s+kqSMx2CaBsrgA7czyZG/E6dU=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
|
||||
github.com/creachadair/mds v0.21.3 h1:RRgEAPIb52cU0q7UxGyN+13QlCVTZIL4slRr0cYYQfA=
|
||||
github.com/creachadair/mds v0.21.3/go.mod h1:1ltMWZd9yXhaHEoZwBialMaviWVUpRPvMwVP7saFAzM=
|
||||
github.com/creachadair/otp v0.5.0 h1:q3Th7CXm2zlmCdBjw5tEPFOj4oWJMnVL5HXlq0sNKS0=
|
||||
@@ -334,8 +331,6 @@ github.com/cyphar/filepath-securejoin v0.6.1 h1:5CeZ1jPXEiYt3+Z6zqprSAgSWiggmpVy
|
||||
github.com/cyphar/filepath-securejoin v0.6.1/go.mod h1:A8hd4EnAeyujCJRrICiOWqjS1AX0a9kM5XL+NwKoYSc=
|
||||
github.com/danieljoos/wincred v1.2.2 h1:774zMFJrqaeYCK2W57BgAem/MLi6mtSE47MB6BOJ0i0=
|
||||
github.com/danieljoos/wincred v1.2.2/go.mod h1:w7w4Utbrz8lqeMbDAK0lkNJUv5sAOkFi7nd/ogr0Uh8=
|
||||
github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2 h1:flLYmnQFZNo04x2NPehMbf30m7Pli57xwZ0NFqR/hb0=
|
||||
github.com/dave-gray101/v2keyauth v0.0.0-20240624150259-c45d584d25e2/go.mod h1:NtWqRzAp/1tw+twkW8uuBenEVVYndEAZACWU3F3xdoQ=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM=
|
||||
@@ -556,12 +551,6 @@ github.com/godbus/dbus/v5 v5.1.0 h1:4KLkAxT3aOY8Li4FRJe/KvhoNFFxo0m6fNuFUO8QJUk=
|
||||
github.com/godbus/dbus/v5 v5.1.0/go.mod h1:xhWf0FNVPg57R7Z0UbKHbJfkEywrmjJnf7w5xrFpKfA=
|
||||
github.com/gofiber/fiber/v2 v2.52.13 h1:TOKP64iqC9b5P49VrBW5tHhUOvDyrtJ0xePEfzJbCbk=
|
||||
github.com/gofiber/fiber/v2 v2.52.13/go.mod h1:YEcBbO/FB+5M1IZNBP9FO3J9281zgPAreiI1oqg8nDw=
|
||||
github.com/gofiber/template v1.8.3 h1:hzHdvMwMo/T2kouz2pPCA0zGiLCeMnoGsQZBTSYgZxc=
|
||||
github.com/gofiber/template v1.8.3/go.mod h1:bs/2n0pSNPOkRa5VJ8zTIvedcI/lEYxzV3+YPXdBvq8=
|
||||
github.com/gofiber/template/html/v2 v2.1.3 h1:n1LYBtmr9C0V/k/3qBblXyMxV5B0o/gpb6dFLp8ea+o=
|
||||
github.com/gofiber/template/html/v2 v2.1.3/go.mod h1:U5Fxgc5KpyujU9OqKzy6Kn6Qup6Tm7zdsISR+VpnHRE=
|
||||
github.com/gofiber/utils v1.1.0 h1:vdEBpn7AzIUJRhe+CiTOJdUcTg4Q9RK+pEa0KPbLdrM=
|
||||
github.com/gofiber/utils v1.1.0/go.mod h1:poZpsnhBykfnY1Mc0KeEa6mSHrS3dV0+oBWyeQmb2e0=
|
||||
github.com/gofrs/flock v0.13.0 h1:95JolYOvGMqeH31+FC7D2+uULf6mG61mEZ/A8dRYMzw=
|
||||
github.com/gofrs/flock v0.13.0/go.mod h1:jxeyy9R1auM5S6JYDBhDt+E2TCo7DkratH4Pgi8P+Z0=
|
||||
github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q=
|
||||
@@ -928,8 +917,8 @@ github.com/mattn/go-runewidth v0.0.9/go.mod h1:H031xJmbD/WCDINGzjvQ9THkh0rPKHF+m
|
||||
github.com/mattn/go-runewidth v0.0.12/go.mod h1:RAqKPSqVFrSLVXbA8x7dzmKdmGzieGRCM46jaSJTDAk=
|
||||
github.com/mattn/go-runewidth v0.0.17 h1:78v8ZlW0bP43XfmAfPsdXcoNCelfMHsDmd/pkENfrjQ=
|
||||
github.com/mattn/go-runewidth v0.0.17/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
|
||||
github.com/mattn/go-sqlite3 v1.14.28 h1:ThEiQrnbtumT+QMknw63Befp/ce/nUPgBPMlRFEum7A=
|
||||
github.com/mattn/go-sqlite3 v1.14.28/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y=
|
||||
github.com/mattn/go-sqlite3 v1.14.32 h1:JD12Ag3oLy1zQA+BNn74xRgaBbdhbNIDYvQUEuuErjs=
|
||||
github.com/mattn/go-sqlite3 v1.14.32/go.mod h1:Uh1q+B4BYcTPb+yiD3kU8Ct7aC0hY9fxUwlHK0RXw+Y=
|
||||
github.com/mdelapenya/tlscert v0.2.0 h1:7H81W6Z/4weDvZBNOfQte5GpIMo0lGYEeWbkGp5LJHI=
|
||||
github.com/mdelapenya/tlscert v0.2.0/go.mod h1:O4njj3ELLnJjGdkN7M/vIVCpZ+Cf0L6muqOG4tLSl8o=
|
||||
github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3vE=
|
||||
@@ -1005,10 +994,8 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI=
|
||||
github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8=
|
||||
github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM=
|
||||
github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxFoYsAObvAuOqXBakRPMD0PWxWG5EE=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca h1:bHlzSuOc5cKvHGF21ZjFFUV7IlkS3wt9YkBsWtkcaC4=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg=
|
||||
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY=
|
||||
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4=
|
||||
github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU=
|
||||
@@ -1017,8 +1004,6 @@ github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc h1:RxwneJl1VgvikiX
|
||||
github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc/go.mod h1:O7SwdSWMilAWhBZMK9N9Y/oBDyMMzshE3ju8Xkexwig=
|
||||
github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6 h1:/nFm1Ttf8g1BnWtEth986JR34pCh9rzae5A2vKBZosc=
|
||||
github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6/go.mod h1:h6kmHUZeafr+k5hRYpGLMzJFH4hItHffgpRo2QIkP+o=
|
||||
github.com/mudler/localrecall v0.6.3 h1:uXOrP9JmetzxgVKzSrawviyBHZfAcvPBBIrvVUdZjDA=
|
||||
github.com/mudler/localrecall v0.6.3/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU=
|
||||
github.com/mudler/localrecall v0.6.5 h1:Q0atTJFFAyumKZG5dbGSrvQ+wsuA88hywIOfHdxpBEU=
|
||||
github.com/mudler/localrecall v0.6.5/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU=
|
||||
github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 h1:Ry8RiWy8fZ6Ff4E7dPmjRsBrnHOnPeOOj2LhCgyjQu0=
|
||||
@@ -1202,13 +1187,12 @@ github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6L
|
||||
github.com/rogpeppe/go-internal v1.3.0/go.mod h1:M8bDsm7K2OlrFYOpmOWEs/qY81heoFRclV5y23lUDJ4=
|
||||
github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ=
|
||||
github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc=
|
||||
github.com/rs/xid v1.5.0/go.mod h1:trrq9SKmegXys3aeAKXMUTdJsYXVwGY3RLcfgqegfbg=
|
||||
github.com/rs/zerolog v1.31.0 h1:FcTR3NnLWW+NnTwwhFWiJSZr4ECLpqCm6QsEnyvbV4A=
|
||||
github.com/rs/zerolog v1.31.0/go.mod h1:/7mN4D5sKwJLZQ2b/znpjC3/GQWY/xaDXUM0kKWRHss=
|
||||
github.com/rs/xid v1.6.0/go.mod h1:7XoLgs4eV+QndskICGsho+ADou8ySMSjJKDIan90Nz0=
|
||||
github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY=
|
||||
github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ=
|
||||
github.com/russross/blackfriday v1.6.0 h1:KqfZb0pUVN2lYqZUYRddxF4OR8ZMURnJIG5Y3VRLtww=
|
||||
github.com/russross/blackfriday v1.6.0/go.mod h1:ti0ldHuxg49ri4ksnFxlkCfN+hvslNlmVHqNRXXJNAY=
|
||||
github.com/russross/blackfriday/v2 v2.0.1/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
|
||||
github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
|
||||
github.com/ruudk/golang-pdf417 v0.0.0-20181029194003-1af4ab5afa58/go.mod h1:6lfFZQK844Gfx8o5WFuvpxWRwnSoipWe/p622j1v06w=
|
||||
github.com/ryanuber/columnize v0.0.0-20160712163229-9b3edd62028f/go.mod h1:sm1tb6uqfes/u+d4ooFouqFdy9/2g9QGwK3SQygK0Ts=
|
||||
github.com/ryanuber/go-glob v1.0.0 h1:iQh3xXAumdQ+4Ufa5b25cRpC5TYKlno6hsv6Cb3pkBk=
|
||||
@@ -1301,7 +1285,6 @@ github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
|
||||
github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4=
|
||||
github.com/spf13/jwalterweatherman v1.1.0/go.mod h1:aNWZUN0dPAAO/Ljvb5BEdw96iTZ0EXowPYD95IqWIGo=
|
||||
github.com/spf13/pflag v1.0.5/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||
github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||
github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk=
|
||||
github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||
github.com/spf13/viper v1.8.1/go.mod h1:o0Pch8wJ9BVSWGQMbra6iw0oQ5oktSIBaujf1rJH9Ns=
|
||||
@@ -1449,8 +1432,8 @@ go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
|
||||
go.etcd.io/etcd/api/v3 v3.5.0/go.mod h1:cbVKeC6lCfl7j/8jBhAK6aIYO9XOjdptoxU/nLQcPvs=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.5.0/go.mod h1:IJHfcCEKxYu1Os13ZdwCwIUTUVGYTSAM3YSwc9/Ac1g=
|
||||
go.etcd.io/etcd/client/v2 v2.305.0/go.mod h1:h9puh54ZTgAKtEbut2oe9P4L/oqKCVB6xsXlzd7alYQ=
|
||||
go.mau.fi/util v0.3.0 h1:Lt3lbRXP6ZBqTINK0EieRWor3zEwwwrDT14Z5N8RUCs=
|
||||
go.mau.fi/util v0.3.0/go.mod h1:9dGsBCCbZJstx16YgnVMVi3O2bOizELoKpugLD4FoGs=
|
||||
go.mau.fi/util v0.9.2 h1:+S4Z03iCsGqU2WY8X2gySFsFjaLlUHFRDVCYvVwynKM=
|
||||
go.mau.fi/util v0.9.2/go.mod h1:055elBBCJSdhRsmub7ci9hXZPgGr1U6dYg44cSgRgoU=
|
||||
go.mongodb.org/mongo-driver v1.17.6 h1:87JUG1wZfWsr6rIz3ZmpH90rL5tea7O3IHuSwHUpsss=
|
||||
go.mongodb.org/mongo-driver v1.17.6/go.mod h1:Hy04i7O2kC4RS06ZrhPRqj/u4DTYkFDAAccj+rVKqgQ=
|
||||
go.opencensus.io v0.21.0/go.mod h1:mSImk1erAIZhrmZN+AvHh14ztQfjbGwt4TtuofqLduU=
|
||||
@@ -2005,10 +1988,8 @@ k8s.io/klog/v2 v2.130.1 h1:n9Xl7H1Xvksem4KFG4PYbdQCQxqc/tTUyrgXaOhHSzk=
|
||||
k8s.io/klog/v2 v2.130.1/go.mod h1:3Jpz1GvMt720eyJH1ckRHK1EDfpxISzJ7I9OYgaDtPE=
|
||||
lukechampine.com/blake3 v1.4.1 h1:I3Smz7gso8w4/TunLKec6K2fn+kyKtDxr/xcQEN84Wg=
|
||||
lukechampine.com/blake3 v1.4.1/go.mod h1:QFosUxmjB8mnrWFSNwKmvxHpfY72bmD2tQ0kBMM3kwo=
|
||||
maunium.net/go/maulogger/v2 v2.4.1 h1:N7zSdd0mZkB2m2JtFUsiGTQQAdP0YeFWT7YMc80yAL8=
|
||||
maunium.net/go/maulogger/v2 v2.4.1/go.mod h1:omPuYwYBILeVQobz8uO3XC8DIRuEb5rXYlQSuqrbCho=
|
||||
maunium.net/go/mautrix v0.17.0 h1:scc1qlUbzPn+wc+3eAPquyD+3gZwwy/hBANBm+iGKK8=
|
||||
maunium.net/go/mautrix v0.17.0/go.mod h1:j+puTEQCEydlVxhJ/dQP5chfa26TdvBO7X6F3Ataav8=
|
||||
maunium.net/go/mautrix v0.25.2 h1:CUG23zp754yGOTMh9Q4mVSENS9FyweE/G+6ZsPDMCUU=
|
||||
maunium.net/go/mautrix v0.25.2/go.mod h1:EWgYyp2iFZP7pnSm+rufHlO8YVnA2KnoNBDpwekiAwI=
|
||||
mvdan.cc/xurls/v2 v2.6.0 h1:3NTZpeTxYVWNSokW3MKeyVkz/j7uYXYiMtXRUfmjbgI=
|
||||
mvdan.cc/xurls/v2 v2.6.0/go.mod h1:bCvEZ1XvdA6wDnxY7jPPjEmigDtvtvPXAD/Exa9IMSk=
|
||||
oras.land/oras-go/v2 v2.6.0 h1:X4ELRsiGkrbeox69+9tzTu492FMUu7zJQW6eJU+I2oc=
|
||||
|
||||
@@ -89,6 +89,11 @@ func (f Functions) ToJSONStructure(name, args string) JSONFunctionStructure {
|
||||
return js
|
||||
}
|
||||
|
||||
// ToJSONStructure converts functions using the configured property keys.
|
||||
func (c FunctionsConfig) ToJSONStructure(functions Functions) JSONFunctionStructure {
|
||||
return functions.ToJSONStructure(c.FunctionNameKey, c.FunctionArgumentsKey)
|
||||
}
|
||||
|
||||
// Select returns a list of functions containing the function with the given name
|
||||
func (f Functions) Select(name string) Functions {
|
||||
var funcs Functions
|
||||
|
||||
@@ -65,6 +65,33 @@ var _ = Describe("LocalAI grammar functions", func() {
|
||||
Expect(fnName.Const).To(Equal("search"))
|
||||
Expect(fnArgs.Properties["query"].(map[string]any)["type"]).To(Equal("string"))
|
||||
})
|
||||
|
||||
It("keeps the name and the arguments in separate properties when both keys are customized", func() {
|
||||
var functions Functions = []Function{
|
||||
{
|
||||
Name: "get_weather",
|
||||
Parameters: map[string]any{
|
||||
"properties": map[string]any{
|
||||
"city": map[string]any{
|
||||
"type": "string",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
config := FunctionsConfig{
|
||||
FunctionNameKey: "function",
|
||||
FunctionArgumentsKey: "parameters",
|
||||
}
|
||||
js := config.ToJSONStructure(functions)
|
||||
Expect(js.OneOf[0].Properties).To(HaveLen(2))
|
||||
|
||||
fnName := js.OneOf[0].Properties["function"].(FunctionName)
|
||||
fnArgs := js.OneOf[0].Properties["parameters"].(Argument)
|
||||
Expect(fnName.Const).To(Equal("get_weather"))
|
||||
Expect(fnArgs.Properties["city"].(map[string]any)["type"]).To(Equal("string"))
|
||||
})
|
||||
})
|
||||
Context("Select()", func() {
|
||||
It("selects one of the functions and returns a list containing only the selected one", func() {
|
||||
|
||||
@@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
|
||||
skippedFiles := 0
|
||||
skippedBytes := int64(0)
|
||||
tasks := make([]downloader.FileTask, 0, len(snapshot.Files))
|
||||
// Sibling manifests are read once, before the staging loop, so the
|
||||
// per-file reuse lookups below never re-read or re-parse them.
|
||||
siblings := loadSiblingCandidates(modelsPath, spec, layout)
|
||||
for index, file := range snapshot.Files {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return Result{}, err
|
||||
@@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
|
||||
skippedBytes += file.Size
|
||||
continue
|
||||
}
|
||||
// Before reaching for the network, consult committed sibling trees for the
|
||||
// same Source (type+endpoint+repo+revision). A narrower allow_patterns
|
||||
// request gets a different CacheKey, so committedResult misses even though a
|
||||
// broader sibling already holds this exact file; reusing it avoids a
|
||||
// redundant re-download of tens of gigabytes. The match is re-verified
|
||||
// through verifyDownloadedFile (full SHA-256), never size-only, and a broader
|
||||
// request can never inherit a narrower sibling's gaps because each file is
|
||||
// matched individually against the sibling's manifest.
|
||||
if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok {
|
||||
manifest.Files[taskIndex] = entry
|
||||
completedBytes.Add(file.Size)
|
||||
skippedFiles++
|
||||
skippedBytes += file.Size
|
||||
continue
|
||||
}
|
||||
nameSum := sha256.Sum256([]byte(file.Path))
|
||||
blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:]))
|
||||
blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel))
|
||||
@@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest
|
||||
return entry, true
|
||||
}
|
||||
|
||||
// siblingCandidate is one committed sibling artifact tree that shares this
|
||||
// request's Source (type+endpoint+repo+revision), with its manifest files
|
||||
// indexed by path.
|
||||
type siblingCandidate struct {
|
||||
final string
|
||||
filesByPath map[string][]ManifestFile
|
||||
}
|
||||
|
||||
// loadSiblingCandidates reads the committed sibling manifest set once, before
|
||||
// the staging loop. Doing it per file instead would re-read and re-parse every
|
||||
// sibling manifest for every file — 20 committed siblings and a 300-file
|
||||
// snapshot means 6000 manifest reads before the first byte is fetched.
|
||||
//
|
||||
// The current artifact's own committed tree is excluded: it is either absent
|
||||
// (the reason materializeLocked is running) or already handled by
|
||||
// committedResult's exact-key fast path.
|
||||
func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate {
|
||||
if spec.Resolved == nil || layout.Final == "" {
|
||||
return nil
|
||||
}
|
||||
siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface")
|
||||
entries, err := os.ReadDir(siblingsRoot)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var candidates []siblingCandidate
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() {
|
||||
continue
|
||||
}
|
||||
siblingFinal := filepath.Join(siblingsRoot, entry.Name())
|
||||
if siblingFinal == layout.Final {
|
||||
continue
|
||||
}
|
||||
siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json"))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
siblingArtifact := siblingManifest.Artifact
|
||||
if siblingArtifact.Resolved == nil ||
|
||||
siblingArtifact.Source.Type != spec.Source.Type ||
|
||||
siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint ||
|
||||
siblingArtifact.Source.Repo != spec.Source.Repo ||
|
||||
siblingArtifact.Resolved.Revision != spec.Resolved.Revision {
|
||||
continue
|
||||
}
|
||||
byPath := make(map[string][]ManifestFile, len(siblingManifest.Files))
|
||||
for _, f := range siblingManifest.Files {
|
||||
byPath[f.Path] = append(byPath[f.Path], f)
|
||||
}
|
||||
candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath})
|
||||
}
|
||||
return candidates
|
||||
}
|
||||
|
||||
// reuseFromCommittedSibling looks for a file already committed under a sibling
|
||||
// artifact tree — same Source (type+endpoint+repo+revision), different
|
||||
// allow/ignore patterns — and stages it for this writer instead of fetching.
|
||||
// A narrower allow_patterns request gets a different CacheKey (path.go:62), so
|
||||
// committedResult misses and materializeLocked would otherwise re-download
|
||||
// files an already-committed broader sibling already holds.
|
||||
//
|
||||
// The match is never size-only: the sibling file is re-hashed through the
|
||||
// shared verifyDownloadedFile against the current request's SnapshotFile (its
|
||||
// LFS or git blob OID), so the staged entry is byte-for-byte identical to a
|
||||
// fresh download. A broader request can never stand in for files a narrower
|
||||
// sibling lacks, because each requested file is matched individually against
|
||||
// the sibling's manifest file set. Hard-link keeps the shared models volume
|
||||
// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the
|
||||
// kernel cannot hard-link.
|
||||
func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) {
|
||||
snapshotRel := path.Join("snapshot", file.Path)
|
||||
snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel))
|
||||
for _, sibling := range candidates {
|
||||
for _, siblingFile := range sibling.filesByPath[file.Path] {
|
||||
if siblingFile.Size != file.Size {
|
||||
continue
|
||||
}
|
||||
siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path))
|
||||
verified, err := verifyDownloadedFile(siblingPath, file)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil {
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
_ = root.Remove(snapshotRel)
|
||||
if err := linkOrCopy(siblingPath, snapshotAbs); err != nil {
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
return verified, true
|
||||
}
|
||||
}
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
|
||||
// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only
|
||||
// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking
|
||||
// keeps the shared models volume neutral — a narrowed request does not double
|
||||
// the storage of a broad sibling's files.
|
||||
func linkOrCopy(src, dst string) error {
|
||||
if err := os.Link(src, dst); err == nil {
|
||||
return nil
|
||||
} else if !errors.Is(err, syscall.EXDEV) {
|
||||
return err
|
||||
}
|
||||
in, err := os.Open(src)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = in.Close() }()
|
||||
out, err := os.Create(dst)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := io.Copy(out, in); err != nil {
|
||||
_ = out.Close()
|
||||
_ = os.Remove(dst)
|
||||
return err
|
||||
}
|
||||
return out.Close()
|
||||
}
|
||||
|
||||
func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) {
|
||||
file, err := os.Open(fileName)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
package modelartifacts_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
hfapi "github.com/mudler/LocalAI/pkg/huggingface-api"
|
||||
"github.com/mudler/LocalAI/pkg/modelartifacts"
|
||||
)
|
||||
|
||||
const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567"
|
||||
|
||||
// recordingResolver serves a fixed full file set filtered by each request's
|
||||
// allow/ignore patterns, so a narrower request genuinely resolves to a strict
|
||||
// subset of a broader sibling's files. The HTTP server behind it records every
|
||||
// fetch, which is the signal the sibling-reuse fix is verified through. The
|
||||
// function under fix is never mocked: a real Manager drives the real staging +
|
||||
// commit path against this stub collaborator.
|
||||
type recordingResolver struct {
|
||||
endpoint string
|
||||
repo string
|
||||
files []hfapi.SnapshotFile
|
||||
server *httptest.Server
|
||||
|
||||
mu sync.Mutex
|
||||
fetched map[string]int
|
||||
}
|
||||
|
||||
func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver {
|
||||
r := &recordingResolver{
|
||||
endpoint: "https://huggingface.co",
|
||||
repo: "owner/repo",
|
||||
files: files,
|
||||
fetched: map[string]int{},
|
||||
}
|
||||
r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) {
|
||||
name := strings.TrimPrefix(req.URL.Path, "/file/")
|
||||
body, ok := contents[name]
|
||||
if !ok {
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
return
|
||||
}
|
||||
r.mu.Lock()
|
||||
r.fetched[name]++
|
||||
r.mu.Unlock()
|
||||
w.Header().Set("Content-Length", strconv.Itoa(len(body)))
|
||||
_, _ = w.Write(body)
|
||||
}))
|
||||
return r
|
||||
}
|
||||
|
||||
func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) {
|
||||
files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns)
|
||||
if err != nil {
|
||||
return hfapi.Snapshot{}, err
|
||||
}
|
||||
out := make([]hfapi.SnapshotFile, len(files))
|
||||
for i, f := range files {
|
||||
f.URL = r.server.URL + "/file/" + f.Path
|
||||
out[i] = f
|
||||
}
|
||||
return hfapi.Snapshot{
|
||||
Endpoint: r.endpoint, Repo: r.repo,
|
||||
RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (r *recordingResolver) fetchCount(path string) int {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
return r.fetched[path]
|
||||
}
|
||||
|
||||
func (r *recordingResolver) resetFetches() {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
r.fetched = map[string]int{}
|
||||
}
|
||||
|
||||
func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile {
|
||||
paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"}
|
||||
files := make([]hfapi.SnapshotFile, 0, len(paths))
|
||||
for _, p := range paths {
|
||||
sum := sha256.Sum256(contents[p])
|
||||
files = append(files, hfapi.SnapshotFile{
|
||||
Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]),
|
||||
})
|
||||
}
|
||||
return files
|
||||
}
|
||||
|
||||
// The narrow-request case proves the fix for #11047:
|
||||
// a request with narrower allow_patterns (a strict subset) reuses files an
|
||||
// already-committed broader sibling holds, hard-linking instead of re-fetching.
|
||||
//
|
||||
// On master this is RED: a narrower allow_patterns set hashes to a different
|
||||
// CacheKey (path.go:62), so committedResult misses and materializeLocked
|
||||
// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On
|
||||
// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling,
|
||||
// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0,
|
||||
// os.SameFile true).
|
||||
var _ = Describe("committed sibling reuse", func() {
|
||||
It("reuses files from a broader committed sibling", func() {
|
||||
contents := map[string][]byte{
|
||||
"a/first.bin": []byte("first-file-bytes"),
|
||||
"b/second.bin": []byte("second-file-bytes-longer"),
|
||||
"c/third.bin": []byte("third-file"),
|
||||
}
|
||||
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
|
||||
defer resolver.server.Close()
|
||||
|
||||
modelsPath := GinkgoT().TempDir()
|
||||
manager := modelartifacts.NewManager(resolver,
|
||||
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
|
||||
|
||||
// Commit the broad sibling: all three files, fetched from the resolver.
|
||||
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
}}
|
||||
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(broad.CacheHit).To(BeFalse())
|
||||
Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0),
|
||||
"the broad sibling must have fetched a/first.bin to commit it")
|
||||
|
||||
resolver.resetFetches()
|
||||
|
||||
// Narrowed request: a strict subset of the broad sibling's file set.
|
||||
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
AllowPatterns: []string{"a/first.bin"},
|
||||
}}
|
||||
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(narrow.CacheHit).To(BeFalse())
|
||||
|
||||
// (b) The sibling-present file must NOT be re-fetched: zero fetches. This is
|
||||
// the assertion that is RED on master (one fetch) and GREEN on the branch.
|
||||
Expect(resolver.fetchCount("a/first.bin")).To(Equal(0),
|
||||
"a/first.bin must be reused from the committed broad sibling, not re-fetched")
|
||||
|
||||
// (a) The narrowed tree's staged file is the same inode as the broad
|
||||
// sibling's file (hard-link), not a freshly downloaded second copy. RED on
|
||||
// master (separate file), GREEN on the branch (hard-link).
|
||||
broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin")
|
||||
narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin")
|
||||
broadInfo, err := os.Stat(broadFile)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
narrowInfo, err := os.Stat(narrowFile)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(),
|
||||
"the narrowed request must hard-link the broad sibling's file rather than store a second copy")
|
||||
|
||||
// The reused bytes are intact end to end.
|
||||
Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"]))
|
||||
})
|
||||
|
||||
// The broader-request case is the manifest file-set guard: a broader request
|
||||
// against a narrower committed sibling must still fetch the files the sibling
|
||||
// lacks and commit a complete tree. Sibling-reuse can never serve an incomplete
|
||||
// model as complete, because each requested file is matched individually against
|
||||
// the sibling's manifest.
|
||||
It("fetches files missing from a narrower committed sibling", func() {
|
||||
contents := map[string][]byte{
|
||||
"a/first.bin": []byte("first-file-bytes"),
|
||||
"b/second.bin": []byte("second-file-bytes-longer"),
|
||||
"c/third.bin": []byte("third-file"),
|
||||
}
|
||||
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
|
||||
defer resolver.server.Close()
|
||||
|
||||
modelsPath := GinkgoT().TempDir()
|
||||
manager := modelartifacts.NewManager(resolver,
|
||||
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
|
||||
|
||||
// Commit a NARROW sibling first: only a/first.bin and b/second.bin.
|
||||
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
AllowPatterns: []string{"a/first.bin", "b/second.bin"},
|
||||
}}
|
||||
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
narrowPaths := make([]string, 0, len(narrow.Manifest.Files))
|
||||
for _, f := range narrow.Manifest.Files {
|
||||
narrowPaths = append(narrowPaths, f.Path)
|
||||
}
|
||||
Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"}))
|
||||
|
||||
resolver.resetFetches()
|
||||
|
||||
// A BROADER request asks for all three files, including c/third.bin which the
|
||||
// narrow sibling does not hold.
|
||||
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
}}
|
||||
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
// The file the narrow sibling lacks MUST be fetched: sibling-reuse must not
|
||||
// inherit a narrower tree's gaps as if the broad request were complete.
|
||||
Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0),
|
||||
"c/third.bin is absent from the narrow sibling and must be fetched, not served as complete")
|
||||
|
||||
// The broad tree's manifest file set is exactly the full set — never the
|
||||
// narrow sibling's subset. This file-set comparison proves no incomplete model
|
||||
// is ever served as complete via sibling-reuse.
|
||||
broadPaths := make([]string, 0, len(broad.Manifest.Files))
|
||||
for _, f := range broad.Manifest.Files {
|
||||
broadPaths = append(broadPaths, f.Path)
|
||||
}
|
||||
Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"}))
|
||||
|
||||
// Every file is present on disk with the right bytes after commit.
|
||||
for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} {
|
||||
Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))).
|
||||
To(Equal(contents[p]))
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,165 @@
|
||||
//go:build linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ProcessVRAM reports device-local resident bytes accounted to a process tree
|
||||
// by DRM. Unsupported or incomplete accounting returns false, not a measured zero.
|
||||
func ProcessVRAM(pid int) (uint64, bool) {
|
||||
return processVRAM("/proc", pid)
|
||||
}
|
||||
|
||||
func processVRAM(procRoot string, pid int) (uint64, bool) {
|
||||
if pid <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
clients := map[string]uint64{}
|
||||
seen := map[int]bool{}
|
||||
pending := []int{pid}
|
||||
for len(pending) > 0 {
|
||||
current := pending[len(pending)-1]
|
||||
pending = pending[:len(pending)-1]
|
||||
if seen[current] {
|
||||
continue
|
||||
}
|
||||
seen[current] = true
|
||||
base := filepath.Join(procRoot, strconv.Itoa(current))
|
||||
fds, err := os.ReadDir(filepath.Join(base, "fd"))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
for _, fd := range fds {
|
||||
target, err := os.Readlink(filepath.Join(base, "fd", fd.Name()))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
// A mixed DRM/NVIDIA tree cannot provide a complete DRM reading.
|
||||
if strings.HasPrefix(target, "/dev/nvidia") {
|
||||
return 0, false
|
||||
}
|
||||
if !strings.HasPrefix(target, "/dev/dri/render") {
|
||||
// Primary nodes can also own allocations. Until their device
|
||||
// identity is resolved, omitting them would undercount the tree.
|
||||
if strings.HasPrefix(target, "/dev/dri/") {
|
||||
return 0, false
|
||||
}
|
||||
continue
|
||||
}
|
||||
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
|
||||
// base adds an integer PID, and fd.Name comes from os.ReadDir.
|
||||
// The kernel supplies these path components, not request input.
|
||||
data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name()))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
client, used, ok := drmResidentClient(data)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
key := target + ":" + client
|
||||
// dup() and fork() can expose the same client more than once. The
|
||||
// snapshot is not atomic; retain its largest observed reading.
|
||||
clients[key] = max(clients[key], used)
|
||||
}
|
||||
|
||||
// A worker may be spawned by any thread, not just the thread leader.
|
||||
tasks, err := os.ReadDir(filepath.Join(base, "task"))
|
||||
if err != nil || len(tasks) == 0 {
|
||||
return 0, false
|
||||
}
|
||||
for _, task := range tasks {
|
||||
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
|
||||
// base adds an integer PID, and task.Name comes from os.ReadDir.
|
||||
// The kernel supplies these path components, not request input.
|
||||
data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children"))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
for _, raw := range strings.Fields(string(data)) {
|
||||
child, err := strconv.Atoi(raw)
|
||||
if err != nil || child <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
pending = append(pending, child)
|
||||
}
|
||||
}
|
||||
}
|
||||
var total uint64
|
||||
for _, used := range clients {
|
||||
if used > math.MaxUint64-total {
|
||||
return 0, false
|
||||
}
|
||||
total += used
|
||||
}
|
||||
return total, len(clients) > 0
|
||||
}
|
||||
|
||||
func drmResidentClient(data []byte) (string, uint64, bool) {
|
||||
var client string
|
||||
var total uint64
|
||||
found := false
|
||||
scanner := bufio.NewScanner(bytes.NewReader(data))
|
||||
for scanner.Scan() {
|
||||
key, value, ok := strings.Cut(scanner.Text(), ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if key == "drm-client-id" {
|
||||
id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64)
|
||||
if err != nil {
|
||||
return "", 0, false
|
||||
}
|
||||
client = strconv.FormatUint(id, 10)
|
||||
}
|
||||
region, resident := strings.CutPrefix(key, "drm-resident-")
|
||||
if !resident || !isVRAMRegion(region) {
|
||||
continue
|
||||
}
|
||||
used, ok := drmResidentBytes(value)
|
||||
if !ok || used > math.MaxUint64-total {
|
||||
return "", 0, false
|
||||
}
|
||||
total += used
|
||||
found = true
|
||||
}
|
||||
return client, total, scanner.Err() == nil && client != "" && found
|
||||
}
|
||||
|
||||
func drmResidentBytes(value string) (uint64, bool) {
|
||||
fields := strings.Fields(value)
|
||||
if len(fields) == 0 || len(fields) > 2 {
|
||||
return 0, false
|
||||
}
|
||||
n, err := strconv.ParseUint(fields[0], 10, 64)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
unit := uint64(1)
|
||||
if len(fields) == 2 {
|
||||
switch strings.ToLower(fields[1]) {
|
||||
case "b":
|
||||
case "kib":
|
||||
unit = 1 << 10
|
||||
case "mib":
|
||||
unit = 1 << 20
|
||||
case "gib":
|
||||
unit = 1 << 30
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
if n > math.MaxUint64/unit {
|
||||
return 0, false
|
||||
}
|
||||
return n * unit, true
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
//go:build linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("ProcessVRAM", func() {
|
||||
var root string
|
||||
write := func(path, contents string) {
|
||||
Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed())
|
||||
Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed())
|
||||
}
|
||||
addProcess := func(pid int, children string) {
|
||||
base := filepath.Join(root, strconv.Itoa(pid))
|
||||
Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed())
|
||||
write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children)
|
||||
}
|
||||
addFD := func(pid, fd int, render, info string) {
|
||||
base := filepath.Join(root, strconv.Itoa(pid))
|
||||
name := strconv.Itoa(fd)
|
||||
Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed())
|
||||
write(filepath.Join(base, "fdinfo", name), info)
|
||||
}
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
root, err = os.MkdirTemp("", "process-vram-")
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
DeferCleanup(os.RemoveAll, root)
|
||||
addProcess(100, "")
|
||||
})
|
||||
|
||||
It("sums resident device memory across GPUs and child processes without duplicate clients", func() {
|
||||
write(filepath.Join(root, "100/task/101/children"), "200")
|
||||
addProcess(200, "")
|
||||
info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n"
|
||||
addFD(100, 3, "renderD128", info)
|
||||
addFD(100, 4, "renderD128", info)
|
||||
addFD(200, 3, "renderD128", info)
|
||||
addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n")
|
||||
used, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(used).To(Equal(uint64(384 * 1024 * 1024)))
|
||||
})
|
||||
|
||||
It("distinguishes a measured zero from unavailable accounting", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n")
|
||||
used, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(used).To(BeZero())
|
||||
})
|
||||
|
||||
DescribeTable("does not invent readings from unsupported or invalid accounting",
|
||||
func(info string) {
|
||||
addFD(100, 3, "renderD128", info)
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
},
|
||||
Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"),
|
||||
Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"),
|
||||
Entry("no client identity", "drm-resident-vram0: 128 MiB\n"),
|
||||
Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"),
|
||||
Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"),
|
||||
Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"),
|
||||
)
|
||||
|
||||
It("omits a partial reading if a child cannot be inspected", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
write(filepath.Join(root, "100/task/100/children"), "200")
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
})
|
||||
|
||||
It("omits a partial reading if another DRM client lacks accounting", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
addFD(100, 4, "renderD129", "drm-client-id: 8\n")
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
})
|
||||
|
||||
DescribeTable("omits mixed readings with unsupported GPU descriptors",
|
||||
func(target string) {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed())
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
},
|
||||
Entry("primary DRM node", "/dev/dri/card0"),
|
||||
Entry("NVIDIA device", "/dev/nvidia0"),
|
||||
)
|
||||
|
||||
It("returns unavailable for missing processes or no DRM descriptors", func() {
|
||||
for _, pid := range []int{-1, 0, 100, 999} {
|
||||
_, ok := processVRAM(root, pid)
|
||||
Expect(ok).To(BeFalse())
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,9 @@
|
||||
//go:build !linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting.
|
||||
func ProcessVRAM(pid int) (uint64, bool) {
|
||||
return 0, false
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
// Package the official index and its repository-local base configurations.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
func main() {
|
||||
if len(os.Args) != 4 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gallery REPOSITORY {gallery|backend} OUTPUT")
|
||||
os.Exit(1)
|
||||
}
|
||||
if err := packageGallery(os.Args[1], os.Args[2], os.Args[3]); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func packageGallery(root, source, output string) error {
|
||||
if source != "gallery" && source != "backend" {
|
||||
return fmt.Errorf("unsupported gallery directory %q", source)
|
||||
}
|
||||
repository, err := os.OpenRoot(root)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = repository.Close() }()
|
||||
body, err := repository.ReadFile(filepath.Join(source, "index.yaml"))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var doc yaml.Node
|
||||
if err := yaml.Unmarshal(body, &doc); err != nil {
|
||||
return err
|
||||
}
|
||||
// The build operator explicitly selects the output directory via the CLI.
|
||||
if err := os.MkdirAll(output, 0700); err != nil { // #nosec G703 -- caller-selected output root
|
||||
return err
|
||||
}
|
||||
destination, err := os.OpenRoot(output)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = destination.Close() }()
|
||||
// Keep the tree relative to the repository root so repeated base configs
|
||||
// share a layer, even when an index refers outside its own directory.
|
||||
const prefix = "github:mudler/LocalAI/"
|
||||
var walk func(*yaml.Node) error
|
||||
walk = func(n *yaml.Node) error {
|
||||
if n.Kind == yaml.MappingNode {
|
||||
for i := 0; i < len(n.Content); i += 2 {
|
||||
value := n.Content[i+1]
|
||||
if n.Content[i].Value != "url" || value.Kind != yaml.ScalarNode || !strings.HasPrefix(value.Value, prefix) || !strings.HasSuffix(value.Value, "@master") {
|
||||
continue
|
||||
}
|
||||
path := strings.TrimSuffix(strings.TrimPrefix(value.Value, prefix), "@master")
|
||||
if !filepath.IsLocal(path) {
|
||||
return fmt.Errorf("base config escapes repository: %q", path)
|
||||
}
|
||||
config, err := repository.ReadFile(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := destination.MkdirAll(filepath.Dir(path), 0700); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := destination.WriteFile(path, config, 0600); err != nil {
|
||||
return err
|
||||
}
|
||||
value.Value = filepath.ToSlash(path)
|
||||
}
|
||||
}
|
||||
for _, child := range n.Content {
|
||||
if err := walk(child); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := walk(&doc); err != nil {
|
||||
return err
|
||||
}
|
||||
body, err = yaml.Marshal(&doc)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return destination.WriteFile("index.yaml", body, 0600)
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package main
|
||||
|
||||
import (
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
"gopkg.in/yaml.v3"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGalleryPackage(t *testing.T) { RegisterFailHandler(Fail); RunSpecs(t, "Gallery packaging") }
|
||||
|
||||
var _ = Describe("Gallery packaging", func() {
|
||||
It("packages both official indexes with every repository-local base available offline", func() {
|
||||
for _, source := range []string{"gallery", "backend"} {
|
||||
out := GinkgoT().TempDir()
|
||||
Expect(packageGallery("../../..", source, out)).To(Succeed())
|
||||
body, err := os.ReadFile(filepath.Join(out, "index.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
var entries []map[string]any
|
||||
Expect(yaml.Unmarshal(body, &entries)).To(Succeed())
|
||||
Expect(entries).ToNot(BeEmpty())
|
||||
for _, entry := range entries {
|
||||
url, _ := entry["url"].(string)
|
||||
Expect(url).ToNot(HavePrefix("github:mudler/LocalAI/"))
|
||||
if strings.HasPrefix(url, "gallery/") {
|
||||
Expect(filepath.Join(out, url)).To(BeAnExistingFile())
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
It("bundles local base configs and preserves external URLs and YAML aliases", func() {
|
||||
root := GinkgoT().TempDir()
|
||||
Expect(os.MkdirAll(filepath.Join(root, "gallery"), 0755)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/base.yaml"), []byte("backend: llama-cpp\n"), 0644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- &base\n name: first\n url: github:mudler/LocalAI/gallery/base.yaml@master\n- <<: *base\n name: second\n- name: external\n url: https://example.com/config.yaml\n"), 0644)).To(Succeed())
|
||||
out := filepath.Join(root, "out")
|
||||
Expect(packageGallery(root, "gallery", out)).To(Succeed())
|
||||
data, err := os.ReadFile(filepath.Join(out, "index.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
var entries []map[string]any
|
||||
Expect(yaml.Unmarshal(data, &entries)).To(Succeed())
|
||||
Expect(entries[0]["url"]).To(Equal("gallery/base.yaml"))
|
||||
Expect(entries[1]["url"]).To(Equal("gallery/base.yaml"))
|
||||
Expect(entries[2]["url"]).To(Equal("https://example.com/config.yaml"))
|
||||
body, err := os.ReadFile(filepath.Join(out, "gallery/base.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(body)).To(Equal("backend: llama-cpp\n"))
|
||||
})
|
||||
It("fails if a referenced config is missing or escapes the repository", func() {
|
||||
for _, ref := range []string{"missing.yaml", "../../outside.yaml"} {
|
||||
root := GinkgoT().TempDir()
|
||||
Expect(os.Mkdir(filepath.Join(root, "gallery"), 0755)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- name: broken\n url: github:mudler/LocalAI/gallery/"+ref+"@master\n"), 0644)).To(Succeed())
|
||||
Expect(packageGallery(root, "gallery", filepath.Join(root, "out"))).ToNot(Succeed())
|
||||
}
|
||||
})
|
||||
It("rejects symlink escapes when reading configs or writing the bundle", func() {
|
||||
for _, location := range []string{"source", "output"} {
|
||||
root, out, outside := GinkgoT().TempDir(), GinkgoT().TempDir(), GinkgoT().TempDir()
|
||||
for _, dir := range []string{filepath.Join(root, "gallery"), filepath.Join(out, "gallery")} {
|
||||
Expect(os.Mkdir(dir, 0700)).To(Succeed())
|
||||
}
|
||||
index := []byte("- name: test\n url: github:mudler/LocalAI/gallery/base.yaml@master\n")
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), index, 0600)).To(Succeed())
|
||||
outsideFile := filepath.Join(outside, "base.yaml")
|
||||
Expect(os.WriteFile(outsideFile, []byte("outside"), 0600)).To(Succeed())
|
||||
link := filepath.Join(root, "gallery/base.yaml")
|
||||
if location == "output" {
|
||||
Expect(os.WriteFile(link, []byte("inside"), 0600)).To(Succeed())
|
||||
link = filepath.Join(out, "gallery/base.yaml")
|
||||
}
|
||||
Expect(os.Symlink(outsideFile, link)).To(Succeed())
|
||||
Expect(packageGallery(root, "gallery", out)).ToNot(Succeed(), location)
|
||||
data, err := os.ReadFile(outsideFile)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(data)).To(Equal("outside"))
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -7313,6 +7313,9 @@ const docTemplate = `{
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"metadata": {
|
||||
"type": "object"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
@@ -7807,6 +7810,10 @@ const docTemplate = `{
|
||||
},
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -7310,6 +7310,9 @@
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"metadata": {
|
||||
"type": "object"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
@@ -7804,6 +7807,10 @@
|
||||
},
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
Loaded 100 of 102 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user