mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-30 01:54:31 -04:00
Merge remote-tracking branch 'origin/master' into feat/failover-chains
Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Assisted-by: Claude:claude-opus-5-5 [Claude Code]
This commit is contained in:
commit
dae9a431e8
69 files changed
+2725
-227
No files matched your search
@@ -355,7 +355,7 @@ jobs:
|
||||
with:
|
||||
backend: ${{ matrix.backend }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
go-version: "1.25.x"
|
||||
go-version: "1.27.x"
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
lang: ${{ matrix.lang || 'python' }}
|
||||
use-pip: ${{ matrix.backend == 'diffusers' }}
|
||||
|
||||
@@ -252,7 +252,8 @@ jobs:
|
||||
name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
# Release matrices and their retries can outlive a one-day artifact.
|
||||
retention-days: 7
|
||||
|
||||
- name: Build (PR)
|
||||
uses: docker/build-push-action@v7
|
||||
|
||||
@@ -22,7 +22,8 @@ on:
|
||||
type: string
|
||||
go-version:
|
||||
description: 'Go version to use'
|
||||
default: '1.24.x'
|
||||
# Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs.
|
||||
default: '1.27.x'
|
||||
type: string
|
||||
tag-suffix:
|
||||
description: 'Tag suffix for the built image'
|
||||
|
||||
@@ -281,7 +281,7 @@ jobs:
|
||||
with:
|
||||
backend: ${{ matrix.backend }}
|
||||
build-type: ${{ matrix.build-type }}
|
||||
go-version: "1.25.x"
|
||||
go-version: "1.27.x"
|
||||
tag-suffix: ${{ matrix.tag-suffix }}
|
||||
lang: ${{ matrix.lang || 'python' }}
|
||||
use-pip: ${{ matrix.backend == 'diffusers' }}
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: Publish official OCI galleries
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [master]
|
||||
paths:
|
||||
- 'gallery/**'
|
||||
- 'backend/index.yaml'
|
||||
- 'scripts/build/gallery/**'
|
||||
- '.github/workflows/gallery_publish.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: publish-official-galleries
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
env:
|
||||
COSIGN_EXPERIMENTAL: '1'
|
||||
GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- source: gallery
|
||||
tag: gallery-models
|
||||
- source: backend
|
||||
tag: gallery-backends
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Test and package gallery
|
||||
env:
|
||||
GALLERY_SOURCE: ${{ matrix.source }}
|
||||
run: |
|
||||
go test ./scripts/build/gallery -count=1
|
||||
go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery"
|
||||
- uses: oras-project/setup-oras@v1
|
||||
with:
|
||||
version: '1.3.0'
|
||||
- uses: sigstore/cosign-installer@v3
|
||||
with:
|
||||
cosign-release: 'v2.6.5'
|
||||
- name: Login to Quay.io
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
registry: quay.io
|
||||
username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
|
||||
password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
|
||||
- name: Publish and sign gallery
|
||||
shell: bash
|
||||
env:
|
||||
GALLERY_TAG: ${{ matrix.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
cd "$RUNNER_TEMP/gallery"
|
||||
files=()
|
||||
while IFS= read -r -d '' file; do
|
||||
files+=("${file#./}:application/yaml")
|
||||
done < <(find . -type f -print0 | sort -z)
|
||||
# Publish an immutable revision, then expose latest only after signing.
|
||||
ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA"
|
||||
oras push --artifact-type application/vnd.localai.gallery.v1 \
|
||||
--format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json"
|
||||
digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json")
|
||||
cosign sign --yes --new-bundle-format \
|
||||
--registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest"
|
||||
oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG"
|
||||
@@ -40,7 +40,7 @@ jobs:
|
||||
# fetch their own toolchains, and no step uses sudo, apt, make or unzip.
|
||||
runs-on: ${{ github.repository == 'mudler/LocalAI' && 'arc-runner-set' || 'ubuntu-latest' }}
|
||||
env:
|
||||
HUGO_VERSION: "0.146.3"
|
||||
HUGO_VERSION: "0.166.0"
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v7
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
|
||||
# rebuild and so the bump bot can see the pin.
|
||||
|
||||
AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22
|
||||
AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
|
||||
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
|
||||
|
||||
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc
|
||||
IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
|
||||
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4
|
||||
LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234
|
||||
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
|
||||
# Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant.
|
||||
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
|
||||
TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64
|
||||
TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d
|
||||
LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -1,52 +0,0 @@
|
||||
diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
index 680fd12..ffd6604 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
+++ b/ggml/src/ggml-cuda/fattn-vec.cuh
|
||||
@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0);
|
||||
extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
-extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu
|
||||
index 5c614a9..d765cfc 100644
|
||||
--- a/ggml/src/ggml-cuda/fattn.cu
|
||||
+++ b/ggml/src/ggml-cuda/fattn.cu
|
||||
@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16)
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)
|
||||
FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0)
|
||||
- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0)
|
||||
|
||||
#ifdef GGML_CUDA_FA_ALL_QUANTS
|
||||
FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16)
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
index a93be56..3630d87 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0);
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
index 3c806c2..c8a4d9f 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0);
|
||||
diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
index 180902f..1646ef0 100644
|
||||
--- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu
|
||||
@@ -5,4 +5,3 @@
|
||||
DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
-DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0);
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# CrispASR version (release tag)
|
||||
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
|
||||
CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee
|
||||
CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
|
||||
SO_TARGET?=libgocrispasr.so
|
||||
|
||||
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# parakeet-cpp backend Makefile.
|
||||
#
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
|
||||
# (.github/bump_deps.sh) can find and update it - matches the
|
||||
# whisper.cpp / ds4 / vibevoice-cpp convention.
|
||||
#
|
||||
@@ -15,7 +15,7 @@
|
||||
# That's what the L0 smoke test uses. The default target below does the
|
||||
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
|
||||
|
||||
PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8
|
||||
PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
|
||||
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
|
||||
|
||||
GOCMD?=go
|
||||
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# stablediffusion.cpp (ggml)
|
||||
STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
|
||||
STABLEDIFFUSION_GGML_VERSION?=b167b942f77ecb17e7f78e163a8c32ff7ac95c10
|
||||
STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551
|
||||
|
||||
CMAKE_ARGS+=-DGGML_MAX_NAME=128
|
||||
|
||||
|
||||
@@ -710,14 +710,14 @@ void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled) {
|
||||
params->enabled = enabled;
|
||||
}
|
||||
|
||||
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y) {
|
||||
params->tile_size_x = tile_size_x;
|
||||
params->tile_size_y = tile_size_y;
|
||||
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h) {
|
||||
params->tile_size_w = tile_size_w;
|
||||
params->tile_size_h = tile_size_h;
|
||||
}
|
||||
|
||||
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y) {
|
||||
params->rel_size_x = rel_size_x;
|
||||
params->rel_size_y = rel_size_y;
|
||||
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h) {
|
||||
params->rel_size_w = rel_size_w;
|
||||
params->rel_size_h = rel_size_h;
|
||||
}
|
||||
|
||||
void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap) {
|
||||
|
||||
@@ -6,8 +6,8 @@ extern "C" {
|
||||
#endif
|
||||
|
||||
void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled);
|
||||
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y);
|
||||
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y);
|
||||
void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h);
|
||||
void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h);
|
||||
void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap);
|
||||
sd_tiling_params_t* sd_img_gen_params_get_vae_tiling_params(sd_img_gen_params_t *params);
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e
|
||||
|
||||
# vllm.cpp version
|
||||
VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp
|
||||
VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb
|
||||
VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c
|
||||
|
||||
# MLX GEMM provider (darwin/metal only; see the metal branch below for why).
|
||||
# Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun
|
||||
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
accelerate
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@
|
||||
torch==2.9.0
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -1,11 +1,11 @@
|
||||
--extra-index-url https://download.pytorch.org/whl/rocm7.0
|
||||
torch==2.10.0+rocm7.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -3,9 +3,9 @@ torch
|
||||
optimum[openvino]
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -2,9 +2,9 @@ torch==2.7.1
|
||||
llvmlite==0.49.0
|
||||
numba==0.67.0
|
||||
accelerate
|
||||
transformers>=5.15.1
|
||||
transformers>=5.17.0
|
||||
bitsandbytes
|
||||
sentence-transformers==5.7.0
|
||||
sentence-transformers==6.1.0
|
||||
diffusers
|
||||
soundfile
|
||||
protobuf==7.36.1
|
||||
@@ -1,6 +1,6 @@
|
||||
grpcio==1.83.0
|
||||
grpcio==1.84.0
|
||||
protobuf==7.36.1
|
||||
certifi
|
||||
setuptools
|
||||
scipy==1.18.0
|
||||
numpy>=2.5.2
|
||||
numpy>=2.5.3
|
||||
@@ -132,6 +132,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: "Kokoros TTS model loaded".into(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -180,11 +181,13 @@ impl Backend for KokorosService {
|
||||
return Ok(Response::new(backend::Result {
|
||||
success: false,
|
||||
message: format!("Failed to write WAV: {}", e),
|
||||
..Default::default()
|
||||
}));
|
||||
}
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: String::new(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
Err(e) => {
|
||||
@@ -192,6 +195,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: false,
|
||||
message: format!("TTS error: {}", e),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
}
|
||||
@@ -292,6 +296,7 @@ impl Backend for KokorosService {
|
||||
Ok(Response::new(backend::Result {
|
||||
success: true,
|
||||
message: "Model freed".into(),
|
||||
..Default::default()
|
||||
}))
|
||||
}
|
||||
|
||||
@@ -348,6 +353,13 @@ impl Backend for KokorosService {
|
||||
Err(Status::unimplemented("Not supported"))
|
||||
}
|
||||
|
||||
async fn animate3_d(
|
||||
&self,
|
||||
_: Request<backend::Animate3DRequest>,
|
||||
) -> Result<Response<backend::Result>, Status> {
|
||||
Err(Status::unimplemented("Not supported"))
|
||||
}
|
||||
|
||||
async fn audio_transcription(
|
||||
&self,
|
||||
_: Request<backend::TranscriptRequest>,
|
||||
|
||||
+11
-1
@@ -47,10 +47,13 @@ type Gallery struct {
|
||||
// fallback for availability, not a load-balancing pool: the primary is
|
||||
// always preferred, and a mirror is only consulted after the one before
|
||||
// it fails. Any URI the gallery loader understands works here
|
||||
// (https://, github:, file://).
|
||||
// (https://, github:, file://, oci://).
|
||||
Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"`
|
||||
Name string `json:"name" yaml:"name"`
|
||||
Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"`
|
||||
// ArtifactVerification overrides Verification only for the gallery OCI artifact.
|
||||
// Backend images keep their separate Verification policy.
|
||||
ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"`
|
||||
}
|
||||
|
||||
// Equal reports whether two gallery entries describe the same gallery.
|
||||
@@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool {
|
||||
if !slices.Equal(g.Mirrors, other.Mirrors) {
|
||||
return false
|
||||
}
|
||||
if g.ArtifactVerification == nil || other.ArtifactVerification == nil {
|
||||
if g.ArtifactVerification != other.ArtifactVerification {
|
||||
return false
|
||||
}
|
||||
} else if *g.ArtifactVerification != *other.ArtifactVerification {
|
||||
return false
|
||||
}
|
||||
if g.Verification == nil || other.Verification == nil {
|
||||
return g.Verification == other.Verification
|
||||
}
|
||||
|
||||
@@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() {
|
||||
Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery"))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("Gallery artifact verification", func() {
|
||||
It("compares artifact policies by value and preserves them in JSON and YAML", func() {
|
||||
a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
|
||||
b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}}
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
b.ArtifactVerification.Identity = "another-workflow"
|
||||
Expect(a.Equal(b)).To(BeFalse())
|
||||
b.ArtifactVerification = nil
|
||||
Expect(a.Equal(b)).To(BeFalse())
|
||||
raw, err := json.Marshal(a)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(json.Unmarshal(raw, &b)).To(Succeed())
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
raw, err = yaml.Marshal(a)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
b = config.Gallery{}
|
||||
Expect(yaml.Unmarshal(raw, &b)).To(Succeed())
|
||||
Expect(a.Equal(b)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
@@ -17,8 +17,8 @@ import (
|
||||
// a caching mirror of the files below. The GitHub URI stays as a mirror so an
|
||||
// install still resolves its gallery unchanged whenever the primary is
|
||||
// unreachable - see the fallback chain in core/gallery/gallery_mirrors.go.
|
||||
const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]`
|
||||
const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]`
|
||||
const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
|
||||
const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]`
|
||||
|
||||
func mustGalleries(jsonList string) []Gallery {
|
||||
var g []Gallery
|
||||
|
||||
@@ -10,22 +10,22 @@ import (
|
||||
)
|
||||
|
||||
var _ = Describe("default galleries", func() {
|
||||
It("serves the model gallery from index.localai.io with GitHub as a mirror", func() {
|
||||
It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() {
|
||||
var galleries []config.Gallery
|
||||
Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed())
|
||||
Expect(galleries).To(HaveLen(1))
|
||||
Expect(galleries[0].Name).To(Equal("localai"))
|
||||
Expect(galleries[0].URL).To(Equal("https://index.localai.io/models"))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"}))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"}))
|
||||
})
|
||||
|
||||
It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() {
|
||||
It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() {
|
||||
var galleries []config.Gallery
|
||||
Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed())
|
||||
Expect(galleries).To(HaveLen(1))
|
||||
Expect(galleries[0].Name).To(Equal("localai"))
|
||||
Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends"))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"}))
|
||||
Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"}))
|
||||
})
|
||||
|
||||
// The mirror is the whole reason this default is safe to ship: if
|
||||
@@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() {
|
||||
Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed())
|
||||
for _, g := range galleries {
|
||||
Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name)
|
||||
Expect(g.ArtifactVerification).ToNot(BeNil())
|
||||
Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"))
|
||||
Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com"))
|
||||
Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust")
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
@@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string {
|
||||
if !looksLikeOCIGallery(candidate) {
|
||||
continue
|
||||
}
|
||||
dir := ociGalleryCacheDir(basePath, candidate, g.Verification)
|
||||
dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g))
|
||||
if dir == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
+31
-19
@@ -4,7 +4,6 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -19,6 +18,7 @@ import (
|
||||
"github.com/mudler/LocalAI/pkg/vram"
|
||||
"github.com/mudler/LocalAI/pkg/xsync"
|
||||
"github.com/mudler/xlog"
|
||||
"golang.org/x/sync/singleflight"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
@@ -276,13 +276,12 @@ func FindGalleryElement[T GalleryElement](models []T, name string) T {
|
||||
func AvailableGalleryModels(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryModel], error) {
|
||||
var models []*GalleryModel
|
||||
|
||||
isInstalled := installedConfigs(systemState.Model.ModelsPath)
|
||||
|
||||
// Get models from galleries
|
||||
for _, gallery := range galleries {
|
||||
galleryModels, err := getGalleryElements(gallery, systemState.Model.ModelsPath, systemState.RequireBackendIntegrity, func(model *GalleryModel) bool {
|
||||
if _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", model.GetName()))); err == nil {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
return isInstalled(model.GetName())
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -351,6 +350,7 @@ var (
|
||||
// same cache-defeating loop the refresh interval exists to stop.
|
||||
availableModelsLoaded bool
|
||||
refreshing atomic.Bool
|
||||
coldLoad singleflight.Group
|
||||
galleryGeneration atomic.Uint64
|
||||
lastRefreshUnixNano atomic.Int64
|
||||
)
|
||||
@@ -429,12 +429,15 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste
|
||||
availableModelsMu.RUnlock()
|
||||
|
||||
if loaded {
|
||||
// The directory is read before taking the lock. Held across the
|
||||
// filesystem work, the lock serialized every caller behind it, and a
|
||||
// page view is dozens of concurrent callers.
|
||||
isInstalled := installedConfigs(systemState.Model.ModelsPath)
|
||||
// Refresh installed status under write lock to avoid races with
|
||||
// concurrent readers and the background refresh goroutine.
|
||||
availableModelsMu.Lock()
|
||||
for _, m := range cached {
|
||||
_, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", m.GetName())))
|
||||
m.SetInstalled(err == nil)
|
||||
m.SetInstalled(isInstalled(m.GetName()))
|
||||
}
|
||||
availableModelsMu.Unlock()
|
||||
// Trigger a background refresh if one is not already running.
|
||||
@@ -442,20 +445,29 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste
|
||||
return cached, nil
|
||||
}
|
||||
|
||||
// No cache yet — must do a blocking load.
|
||||
models, err := AvailableGalleryModels(galleries, systemState)
|
||||
// No cache yet, so the load blocks. Callers arriving while it runs wait
|
||||
// for it instead of each starting their own: a page view on a fresh
|
||||
// server is the listing plus one estimate per row at once, and each load
|
||||
// fetches the gallery index and every config it references.
|
||||
v, err, _ := coldLoad.Do("gallery", func() (any, error) {
|
||||
models, err := AvailableGalleryModels(galleries, systemState)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
availableModelsMu.Lock()
|
||||
availableModelsCache = models
|
||||
availableModelsLoaded = true
|
||||
galleryGeneration.Add(1)
|
||||
availableModelsMu.Unlock()
|
||||
lastRefreshUnixNano.Store(time.Now().UnixNano())
|
||||
|
||||
return models, nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
availableModelsMu.Lock()
|
||||
availableModelsCache = models
|
||||
availableModelsLoaded = true
|
||||
galleryGeneration.Add(1)
|
||||
availableModelsMu.Unlock()
|
||||
lastRefreshUnixNano.Store(time.Now().UnixNano())
|
||||
|
||||
return models, nil
|
||||
return v.(GalleryElements[*GalleryModel]), nil
|
||||
}
|
||||
|
||||
// triggerGalleryRefresh starts a background goroutine that refreshes the
|
||||
@@ -634,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]()
|
||||
// would also point relative entry urls at an unpacked tree the new policy has
|
||||
// not produced yet, so they could not be installed.
|
||||
func galleryIndexCacheKey(g config.Gallery) string {
|
||||
return g.Name + "-" + galleryCacheName(g.URL, g.Verification)
|
||||
return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g))
|
||||
}
|
||||
|
||||
func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) {
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
package gallery_test
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/gallery"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
)
|
||||
|
||||
// The models directory is often network storage (SMB, NFS), where every
|
||||
// filesystem call is a round trip. The cached listing is read by the gallery
|
||||
// page and by one VRAM estimate per row, so whatever it costs is paid dozens
|
||||
// of times per page view.
|
||||
var _ = Describe("Gallery cache installed status", func() {
|
||||
const index = `
|
||||
- name: plain
|
||||
backend: llama-cpp
|
||||
- name: linked
|
||||
backend: llama-cpp
|
||||
- name: dangling
|
||||
backend: llama-cpp
|
||||
- name: later
|
||||
backend: llama-cpp
|
||||
- name: absent
|
||||
backend: llama-cpp
|
||||
`
|
||||
|
||||
var (
|
||||
modelsDir string
|
||||
state *system.SystemState
|
||||
galleries []config.Gallery
|
||||
hits atomic.Int32
|
||||
delay time.Duration
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
modelsDir, err = os.MkdirTemp("", "gallery-installed")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
DeferCleanup(func() { _ = os.RemoveAll(modelsDir) })
|
||||
state, err = system.GetSystemState(system.WithModelPath(modelsDir))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
hits.Store(0)
|
||||
delay = 0
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hits.Add(1)
|
||||
time.Sleep(delay)
|
||||
_, _ = w.Write([]byte(index))
|
||||
}))
|
||||
DeferCleanup(server.Close)
|
||||
galleries = []config.Gallery{{Name: "test", URL: server.URL + "/index.yaml"}}
|
||||
|
||||
gallery.ResetGalleryModelCache()
|
||||
DeferCleanup(gallery.ResetGalleryModelCache)
|
||||
})
|
||||
|
||||
installed := func(models gallery.GalleryElements[*gallery.GalleryModel]) map[string]bool {
|
||||
out := map[string]bool{}
|
||||
for _, m := range models {
|
||||
out[m.Name] = m.Installed
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
It("reports what os.Stat would, for files, symlinks and dangling symlinks", func() {
|
||||
Expect(os.WriteFile(filepath.Join(modelsDir, "plain.yaml"), []byte("name: plain\n"), 0o644)).To(Succeed())
|
||||
target := filepath.Join(modelsDir, "target.txt")
|
||||
Expect(os.WriteFile(target, []byte("name: linked\n"), 0o644)).To(Succeed())
|
||||
Expect(os.Symlink(target, filepath.Join(modelsDir, "linked.yaml"))).To(Succeed())
|
||||
Expect(os.Symlink(filepath.Join(modelsDir, "missing"), filepath.Join(modelsDir, "dangling.yaml"))).To(Succeed())
|
||||
|
||||
// Both the blocking first load and the cached path set the flag, and
|
||||
// they must agree.
|
||||
for range 2 {
|
||||
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(installed(models)).To(Equal(map[string]bool{
|
||||
"plain": true,
|
||||
"linked": true,
|
||||
"dangling": false,
|
||||
"later": false,
|
||||
"absent": false,
|
||||
}))
|
||||
}
|
||||
})
|
||||
|
||||
It("picks up a config written after the gallery was cached", func() {
|
||||
_, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
Expect(os.WriteFile(filepath.Join(modelsDir, "later.yaml"), []byte("name: later\n"), 0o644)).To(Succeed())
|
||||
|
||||
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(installed(models)).To(HaveKeyWithValue("later", true))
|
||||
})
|
||||
|
||||
It("reports nothing installed when the models directory is gone", func() {
|
||||
_, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(os.RemoveAll(modelsDir)).To(Succeed())
|
||||
|
||||
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(installed(models)).To(HaveEach(BeFalse()))
|
||||
})
|
||||
|
||||
It("shares one upstream load between concurrent callers on a cold cache", func() {
|
||||
// Slow enough that every caller arrives while the first load is still
|
||||
// in flight, which is what a page view does to a freshly started
|
||||
// server: the listing and every row's estimate at once.
|
||||
delay = 300 * time.Millisecond
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for range 8 {
|
||||
wg.Go(func() {
|
||||
defer GinkgoRecover()
|
||||
models, err := gallery.AvailableGalleryModelsCached(galleries, state)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(models).To(HaveLen(5))
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
Expect(hits.Load()).To(Equal(int32(1)))
|
||||
})
|
||||
})
|
||||
@@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
|
||||
if !looksLikeOCIGallery(g.URL) {
|
||||
return nil
|
||||
}
|
||||
return g.Verification
|
||||
return galleryArtifactPolicy(g)
|
||||
}
|
||||
|
||||
// verifiableCandidates drops the candidates that cannot answer for a signed
|
||||
@@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification {
|
||||
// at, and after a refusal it would turn "this artifact is not trusted" into
|
||||
// "use this other, unchecked copy instead".
|
||||
func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string {
|
||||
if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) {
|
||||
if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) {
|
||||
return candidates
|
||||
}
|
||||
out := make([]string, 0, len(candidates))
|
||||
|
||||
@@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) {
|
||||
// later fetch served would hand the user a truncated gallery with no sign that
|
||||
// anything went wrong.
|
||||
func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) {
|
||||
policy := galleryArtifactPolicy(g)
|
||||
// Checked before the cache: a copy unpacked while strict integrity was
|
||||
// off was never verified, and turning strict integrity on must not keep
|
||||
// serving it for the rest of its TTL.
|
||||
if g.Verification == nil && requireIntegrity {
|
||||
if policy == nil && requireIntegrity {
|
||||
return nil, &galleryVerificationError{
|
||||
strict: true,
|
||||
err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
|
||||
err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate),
|
||||
}
|
||||
}
|
||||
|
||||
cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification)
|
||||
cacheDir := ociGalleryCacheDir(basePath, candidate, policy)
|
||||
if cacheDir == "" {
|
||||
return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate)
|
||||
}
|
||||
@@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
|
||||
pullRef := downloader.URI(candidate).OCIReference()
|
||||
|
||||
if g.Verification != nil {
|
||||
if policy != nil {
|
||||
// Resolve first, verify the digest, then pull that same digest.
|
||||
// Nothing has been fetched at this point beyond the manifest, so a
|
||||
// policy failure leaves no content anywhere.
|
||||
@@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil {
|
||||
if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil {
|
||||
// Only a decision about the artifact is a refusal. The
|
||||
// verifier also reaches the Sigstore TUF mirror and the
|
||||
// registry, and a timeout or a 5xx there says nothing about
|
||||
@@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base
|
||||
|
||||
return body, nil
|
||||
}
|
||||
|
||||
func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification {
|
||||
if g.ArtifactVerification != nil {
|
||||
return g.ArtifactVerification
|
||||
}
|
||||
return g.Verification
|
||||
}
|
||||
@@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() {
|
||||
})
|
||||
})
|
||||
|
||||
It("uses the artifact policy without replacing backend image verification", func() {
|
||||
srv, _, _ := ociRegistry()
|
||||
url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}})
|
||||
backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"}
|
||||
artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"}
|
||||
var seen *config.GalleryVerification
|
||||
stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil })
|
||||
g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy}
|
||||
_, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true)
|
||||
Expect(source).To(Equal(url))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(seen).To(Equal(artifactPolicy))
|
||||
Expect(g.Verification).To(Equal(backendPolicy))
|
||||
})
|
||||
|
||||
It("refuses an unsigned gallery in strict integrity mode", func() {
|
||||
srv, _, blobs := ociRegistry()
|
||||
url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
package gallery
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
const modelConfigExt = ".yaml"
|
||||
|
||||
// installedConfigs answers "does <modelsPath>/<name>.yaml exist?" for every
|
||||
// entry of a gallery from a single read of the models directory.
|
||||
//
|
||||
// The question used to be asked with one os.Stat per gallery entry. The gallery
|
||||
// holds thousands of entries and the models directory is often network storage
|
||||
// (SMB, NFS), where each Stat is a round trip, so one listing cost seconds. The
|
||||
// listing is read by the gallery page and by one VRAM estimate per row, which
|
||||
// turned a page view into minutes.
|
||||
//
|
||||
// Answers match os.Stat on the same path: a symlink counts only when its target
|
||||
// exists, and anything else carrying the name counts, directories included.
|
||||
// Names that are not a plain file name are checked with os.Stat directly, since
|
||||
// they point outside the listed directory.
|
||||
func installedConfigs(modelsPath string) func(name string) bool {
|
||||
statInstalled := func(name string) bool {
|
||||
_, err := os.Stat(filepath.Join(modelsPath, name+modelConfigExt))
|
||||
return err == nil
|
||||
}
|
||||
|
||||
entries, err := os.ReadDir(modelsPath)
|
||||
if err != nil {
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return func(string) bool { return false }
|
||||
}
|
||||
// A directory that exists but cannot be listed may still answer a
|
||||
// Stat, so fall back rather than report everything as not installed.
|
||||
return statInstalled
|
||||
}
|
||||
|
||||
present := make(map[string]struct{}, len(entries))
|
||||
for _, e := range entries {
|
||||
base, ok := strings.CutSuffix(e.Name(), modelConfigExt)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if e.Type()&fs.ModeSymlink != 0 && !statInstalled(base) {
|
||||
continue
|
||||
}
|
||||
present[base] = struct{}{}
|
||||
}
|
||||
|
||||
return func(name string) bool {
|
||||
if strings.ContainsRune(name, '/') || strings.ContainsRune(name, filepath.Separator) {
|
||||
return statInstalled(name)
|
||||
}
|
||||
_, ok := present[name]
|
||||
return ok
|
||||
}
|
||||
}
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
"github.com/mudler/LocalAI/core/services/monitoring"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/xsysinfo"
|
||||
)
|
||||
|
||||
// SystemInformations returns the system informations
|
||||
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
|
||||
entry.Process = proc
|
||||
}
|
||||
}
|
||||
if pid, ok := localPID(m); ok {
|
||||
if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
|
||||
entry.SizeVRAM = &used
|
||||
}
|
||||
}
|
||||
sysmodels = append(sysmodels, entry)
|
||||
}
|
||||
if sampler != nil {
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package localai_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/http/endpoints/localai"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
process "github.com/mudler/go-processmanager"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SystemInformations memory", func() {
|
||||
It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
|
||||
path, err := os.MkdirTemp("", "system-info-")
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
DeferCleanup(os.RemoveAll, path)
|
||||
configFile := filepath.Join(path, "remote.yaml")
|
||||
Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
|
||||
cl := config.NewModelConfigLoader(path)
|
||||
Expect(cl.ReadModelConfig(configFile)).To(Succeed())
|
||||
ml := model.NewModelLoader(&system.SystemState{})
|
||||
store := model.NewInMemoryModelStore()
|
||||
store.Set("remote", model.NewModel("remote", "worker:50051", nil))
|
||||
store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
|
||||
ml.SetModelStore(store)
|
||||
app := echo.New()
|
||||
app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
|
||||
rec := httptest.NewRecorder()
|
||||
app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
var response struct {
|
||||
Models []map[string]any `json:"loaded_models"`
|
||||
}
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
|
||||
Expect(response.Models).To(ConsistOf(
|
||||
map[string]any{"id": "remote", "backend": "llama-cpp"},
|
||||
map[string]any{"id": "stopped"},
|
||||
))
|
||||
})
|
||||
})
|
||||
@@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
return true
|
||||
}
|
||||
|
||||
// Try JSON parsing as fallback
|
||||
jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true)
|
||||
if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount {
|
||||
// Only completed JSON calls can be emitted as completed SSE items.
|
||||
jsonResults := parseStreamingJSONToolCalls(cleanedResult)
|
||||
if len(jsonResults) > lastEmittedToolCallCount {
|
||||
for i := lastEmittedToolCallCount; i < len(jsonResults); i++ {
|
||||
jsonObj := jsonResults[i]
|
||||
if name, ok := jsonObj["name"].(string); ok && name != "" {
|
||||
args := "{}"
|
||||
if argsVal, ok := jsonObj["arguments"]; ok {
|
||||
if argsStr, ok := argsVal.(string); ok {
|
||||
args = argsStr
|
||||
} else {
|
||||
argsBytes, _ := json.Marshal(argsVal)
|
||||
args = string(argsBytes)
|
||||
}
|
||||
}
|
||||
tc := jsonResults[i]
|
||||
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
|
||||
outputIndex++
|
||||
|
||||
toolCallID := fmt.Sprintf("fc_%s", uuid.New().String())
|
||||
outputIndex++
|
||||
|
||||
functionCallItem := &schema.ORItemField{
|
||||
Type: "function_call",
|
||||
ID: toolCallID,
|
||||
Status: "completed",
|
||||
CallID: toolCallID,
|
||||
Name: name,
|
||||
Arguments: args,
|
||||
}
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
functionCallItem := &schema.ORItemField{
|
||||
Type: "function_call",
|
||||
ID: toolCallID,
|
||||
Status: "completed",
|
||||
CallID: toolCallID,
|
||||
Name: tc.Name,
|
||||
Arguments: tc.Arguments,
|
||||
}
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
Item: functionCallItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
}
|
||||
lastEmittedToolCallCount = len(jsonResults)
|
||||
c.Response().Flush()
|
||||
@@ -2424,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
}
|
||||
|
||||
// Non-tool-call streaming path
|
||||
messageOutputIndex := outputIndex
|
||||
var reasoningOutputIndex int
|
||||
// Emit output_item.added for message
|
||||
currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String())
|
||||
messageItem := &schema.ORItemField{
|
||||
@@ -2436,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
Item: messageItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2448,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Part: &emptyTextPart,
|
||||
})
|
||||
@@ -2471,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
}
|
||||
|
||||
// Handle reasoning item
|
||||
if extractor.Reasoning() != "" {
|
||||
if extractor.Reasoning() != "" || reasoningDelta != "" {
|
||||
// Check if we need to create reasoning item
|
||||
if currentReasoningID == "" {
|
||||
outputIndex++
|
||||
reasoningOutputIndex = outputIndex
|
||||
currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String())
|
||||
reasoningItem := &schema.ORItemField{
|
||||
Type: "reasoning",
|
||||
@@ -2484,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
Item: reasoningItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2496,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.added",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Part: &emptyPart,
|
||||
})
|
||||
@@ -2509,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.delta",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Delta: strPtr(reasoningDelta),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2526,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.delta",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Delta: strPtr(contentDelta),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2595,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Text: strPtr(finalReasoning),
|
||||
Logprobs: emptyLogprobs(),
|
||||
@@ -2608,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentReasoningID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
ContentIndex: ¤tReasoningContentIndex,
|
||||
Part: &reasoningPart,
|
||||
})
|
||||
@@ -2624,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &reasoningOutputIndex,
|
||||
Item: reasoningItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2658,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.output_text.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Text: strPtr(result),
|
||||
Logprobs: logprobsPtr(mcpStreamLogprobs),
|
||||
@@ -2671,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
Type: "response.content_part.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
ItemID: currentMessageID,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
ContentIndex: ¤tContentIndex,
|
||||
Part: &resultPart,
|
||||
})
|
||||
@@ -2683,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
sendSSEEvent(c, &schema.ORStreamEvent{
|
||||
Type: "response.output_item.done",
|
||||
SequenceNumber: sequenceNumber,
|
||||
OutputIndex: &outputIndex,
|
||||
OutputIndex: &messageOutputIndex,
|
||||
Item: messageItem,
|
||||
})
|
||||
sequenceNumber++
|
||||
@@ -2723,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6
|
||||
// Emit response.completed
|
||||
now := time.Now().Unix()
|
||||
|
||||
// Collect final output items (reasoning first, then messages, then tool calls)
|
||||
var finalOutputItems []schema.ORItemField
|
||||
// Add reasoning item if it exists
|
||||
if currentReasoningID != "" && finalReasoning != "" {
|
||||
finalOutputItems = append(finalOutputItems, schema.ORItemField{
|
||||
Type: "reasoning",
|
||||
ID: currentReasoningID,
|
||||
Status: "completed",
|
||||
Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)},
|
||||
})
|
||||
}
|
||||
// Add message item
|
||||
if len(collectedOutputItems) > 0 {
|
||||
// Use collected items (may include reasoning already)
|
||||
for _, item := range collectedOutputItems {
|
||||
if item.Type == "message" {
|
||||
finalOutputItems = append(finalOutputItems, item)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
finalOutputItems = append(finalOutputItems, *messageItem)
|
||||
}
|
||||
// Add function_call items from fallback
|
||||
for _, item := range collectedOutputItems {
|
||||
if item.Type == "function_call" {
|
||||
finalOutputItems = append(finalOutputItems, item)
|
||||
}
|
||||
}
|
||||
// The final output array must use the indices announced in the stream.
|
||||
// The message is opened first, followed by reasoning and fallback calls.
|
||||
finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...)
|
||||
responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{
|
||||
InputTokens: noToolTokenUsage.Prompt,
|
||||
OutputTokens: noToolTokenUsage.Completion,
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/backend"
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Responses stream item consistency", func() {
|
||||
DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) {
|
||||
originalInference := backend.ModelInferenceFunc
|
||||
DeferCleanup(func() { backend.ModelInferenceFunc = originalInference })
|
||||
backend.ModelInferenceFunc = func(
|
||||
ctx context.Context, prompt string, messages schema.Messages,
|
||||
images, videos, audios []string, loader *model.ModelLoader,
|
||||
cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig,
|
||||
tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string,
|
||||
logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string,
|
||||
) (func() (backend.LLMResponse, error), error) {
|
||||
return func() (backend.LLMResponse, error) {
|
||||
for i, token := range tokens {
|
||||
usage := backend.TokenUsage{}
|
||||
if len(chatDeltas) > 0 {
|
||||
usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]}
|
||||
}
|
||||
if !tokenCallback(token, usage) {
|
||||
break
|
||||
}
|
||||
}
|
||||
return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil
|
||||
}, nil
|
||||
}
|
||||
cfg := &config.ModelConfig{}
|
||||
cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback
|
||||
cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)<tool_call>(.*?)</tool_call>`}
|
||||
recorder := httptest.NewRecorder()
|
||||
request := httptest.NewRequest("POST", "/v1/responses", nil)
|
||||
c := echo.New().NewContext(request, recorder)
|
||||
input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true}
|
||||
err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n"))
|
||||
|
||||
var events []schema.ORStreamEvent
|
||||
var completed *schema.ORResponseResource
|
||||
for _, line := range strings.Split(recorder.Body.String(), "\n") {
|
||||
if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" {
|
||||
continue
|
||||
}
|
||||
var event schema.ORStreamEvent
|
||||
Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed())
|
||||
Expect(event.Type).NotTo(Equal("error"))
|
||||
events = append(events, event)
|
||||
if event.Type == "response.completed" {
|
||||
completed = event.Response
|
||||
}
|
||||
}
|
||||
Expect(completed).NotTo(BeNil())
|
||||
wantCount := 1
|
||||
if wantReasoning != "" {
|
||||
wantCount++
|
||||
}
|
||||
if fallback {
|
||||
wantCount++
|
||||
}
|
||||
Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls")
|
||||
|
||||
indices := map[string]int{}
|
||||
done := map[string]int{}
|
||||
deltas := map[string]string{}
|
||||
for i, event := range events {
|
||||
Expect(event.SequenceNumber).To(Equal(i))
|
||||
if event.Type == "response.output_item.added" {
|
||||
Expect(event.Item).NotTo(BeNil())
|
||||
Expect(event.OutputIndex).NotTo(BeNil())
|
||||
Expect(indices).NotTo(HaveKey(event.Item.ID))
|
||||
Expect(*event.OutputIndex).To(Equal(len(indices)))
|
||||
indices[event.Item.ID] = *event.OutputIndex
|
||||
}
|
||||
id := event.ItemID
|
||||
if event.Item != nil {
|
||||
id = event.Item.ID
|
||||
}
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
Expect(indices).To(HaveKey(id))
|
||||
Expect(event.OutputIndex).NotTo(BeNil())
|
||||
Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id)
|
||||
Expect(completed.Output[indices[id]].ID).To(Equal(id))
|
||||
if event.Type == "response.output_item.done" {
|
||||
done[id]++
|
||||
Expect(event.Item.Status).To(Equal("completed"))
|
||||
Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type))
|
||||
if event.Item.Type == "function_call" {
|
||||
Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name))
|
||||
Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments))
|
||||
} else {
|
||||
Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content))
|
||||
}
|
||||
}
|
||||
if event.Type == "response.output_text.delta" {
|
||||
deltas[id] += *event.Delta
|
||||
}
|
||||
}
|
||||
Expect(indices).To(HaveLen(wantCount))
|
||||
for _, item := range completed.Output {
|
||||
Expect(done[item.ID]).To(Equal(1))
|
||||
switch item.Type {
|
||||
case "message", "reasoning":
|
||||
want := wantAnswer
|
||||
if item.Type == "reasoning" {
|
||||
want = wantReasoning
|
||||
}
|
||||
parts, ok := item.Content.([]any)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(parts).To(HaveLen(1))
|
||||
Expect(parts[0].(map[string]any)["text"]).To(Equal(want))
|
||||
if !fallback {
|
||||
Expect(deltas[item.ID]).To(Equal(want))
|
||||
}
|
||||
case "function_call":
|
||||
Expect(item.Name).To(Equal("get_weather"))
|
||||
Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`))
|
||||
Expect(item.CallID).NotTo(BeEmpty())
|
||||
default:
|
||||
Fail("unexpected output item type: " + item.Type)
|
||||
}
|
||||
}
|
||||
},
|
||||
Entry("tagged reasoning and answer", []string{"<think>", "Let me think.", "</think>", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false),
|
||||
Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false),
|
||||
Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false),
|
||||
Entry("automatic fallback tool call", []string{`<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "", "", true),
|
||||
Entry("reasoning and automatic fallback tool call", []string{"<think>", "Let me think.", "</think>", `<tool_call>{"name":"get_weather","arguments":{"city":"Rome"}}</tool_call>`}, nil, "Let me think.", "", true),
|
||||
)
|
||||
})
|
||||
@@ -0,0 +1,35 @@
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/mudler/LocalAI/pkg/functions"
|
||||
)
|
||||
|
||||
func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults {
|
||||
// Partial parsing heals unfinished arguments. The caller emits terminal
|
||||
// events and never revisits emitted calls, so only accept complete JSON.
|
||||
// Keep completed objects returned before an unfinished trailing object.
|
||||
objects, _ := functions.ParseJSONIterative(text, false)
|
||||
var calls []functions.FuncCallResults
|
||||
for _, object := range objects {
|
||||
name, ok := object["name"].(string)
|
||||
if !ok || name == "" {
|
||||
continue
|
||||
}
|
||||
arguments := "{}"
|
||||
if value, ok := object["arguments"]; ok {
|
||||
if s, ok := value.(string); ok {
|
||||
arguments = s
|
||||
} else {
|
||||
data, err := json.Marshal(value)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
arguments = string(data)
|
||||
}
|
||||
}
|
||||
calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments})
|
||||
}
|
||||
return calls
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package openresponses
|
||||
|
||||
import (
|
||||
"github.com/mudler/LocalAI/pkg/functions"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Streaming JSON tool calls", func() {
|
||||
It("waits for the arguments before completing a split call", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty())
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty())
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("does not complete a call at any intermediate token boundary", func() {
|
||||
text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}`
|
||||
for end := 1; end < len(text); end++ {
|
||||
Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end])
|
||||
}
|
||||
Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1))
|
||||
})
|
||||
|
||||
It("keeps completed calls while the next call is incomplete", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("preserves string arguments and calls that take no arguments", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "Bash", Arguments: `{"command":"ls -la"}`},
|
||||
{Name: "status", Arguments: `{}`},
|
||||
}))
|
||||
})
|
||||
|
||||
It("does not count unrelated JSON objects as emitted calls", func() {
|
||||
Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{
|
||||
{Name: "status", Arguments: `{}`},
|
||||
}))
|
||||
})
|
||||
})
|
||||
@@ -208,6 +208,9 @@ type SysInfoModel struct {
|
||||
// when the model has no local process (a distributed worker holds it) or
|
||||
// the process could not be read.
|
||||
Process *SysInfoProcess `json:"process,omitempty"`
|
||||
// SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
|
||||
// the backend process tree has no complete supported reading.
|
||||
SizeVRAM *uint64 `json:"size_vram,omitempty"`
|
||||
}
|
||||
|
||||
// SysInfoProcess is a point-in-time reading of one backend process.
|
||||
|
||||
@@ -99,7 +99,7 @@ type OpenAIResponse struct {
|
||||
// OpenAI-SDK consumers that filter on a truthy `result.usage`
|
||||
// (continuedev/continue, Kilo Code, Roo Code, etc.).
|
||||
Usage *OpenAIUsage `json:"usage,omitempty"`
|
||||
Metadata json.RawMessage `json:"metadata,omitempty"`
|
||||
Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"`
|
||||
}
|
||||
|
||||
// StreamOptions mirrors OpenAI's `stream_options` request field. The only
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package schema_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SysInfoModel memory", func() {
|
||||
It("omits unavailable VRAM while preserving a measured zero", func() {
|
||||
entry := schema.SysInfoModel{ID: "model"}
|
||||
encoded, err := json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
|
||||
|
||||
zero := uint64(0)
|
||||
entry.SizeVRAM = &zero
|
||||
encoded, err = json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
|
||||
})
|
||||
})
|
||||
+2
-1
@@ -59,6 +59,7 @@ services:
|
||||
# capabilities: [gpu, utility]
|
||||
#
|
||||
# For legacy NVIDIA driver (for older NVIDIA Container Toolkit):
|
||||
# Request compute for CUDA libraries (libcuda.so.1) and utility for NVML.
|
||||
# environment:
|
||||
# NVIDIA_DRIVER_CAPABILITIES: "compute,utility"
|
||||
# init: true
|
||||
@@ -68,7 +69,7 @@ services:
|
||||
# devices:
|
||||
# - driver: nvidia
|
||||
# count: 1
|
||||
# capabilities: [gpu, utility]
|
||||
# capabilities: [gpu, compute, utility]
|
||||
|
||||
## Uncomment for PostgreSQL-backed knowledge base (see Agents docs)
|
||||
# postgres:
|
||||
|
||||
@@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model
|
||||
local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master
|
||||
```
|
||||
|
||||
See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files.
|
||||
See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files.
|
||||
|
||||
### Prompt templates
|
||||
|
||||
|
||||
@@ -82,6 +82,8 @@ tags:
|
||||
|
||||
### Verifying OCI Backends
|
||||
|
||||
The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details.
|
||||
|
||||
Backend galleries can require keyless Sigstore signatures for every OCI image
|
||||
they provide. Add a `verification` policy to the gallery configuration, then
|
||||
enable strict integrity mode:
|
||||
|
||||
@@ -417,8 +417,12 @@ usage is reported back to the frontend:
|
||||
NVML library (and therefore `nvidia-smi`) is not available inside the
|
||||
container. CUDA compute still works, but the worker cannot query free VRAM
|
||||
and the Nodes page will show the node as fully used. Set
|
||||
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI
|
||||
runtime, list `capabilities: [gpu, utility]` on the device reservation).
|
||||
`NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime.
|
||||
For Docker Compose with `driver: nvidia`, use
|
||||
`capabilities: [gpu, compute, utility]` on the device reservation.
|
||||
Docker derives driver capabilities from this reservation, so include `compute`
|
||||
for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone
|
||||
enables monitoring but does not provide CUDA libraries.
|
||||
|
||||
- **Run the container with `init: true` (or `docker run --init`).** The
|
||||
worker process becomes PID 1 in the container and cannot reap zombies on
|
||||
|
||||
@@ -39,6 +39,99 @@ Both views use the same model selection and store the view, search, filter, and
|
||||
selection in the URL. Installing from Explore does not move you away from the
|
||||
catalog; the entry updates in place when the operation finishes.
|
||||
|
||||
## Cyber-Tiel-Coder
|
||||
|
||||
Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp.
|
||||
The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector.
|
||||
To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`.
|
||||
Both configurations use the embedded chat template and default to 32,768 context tokens.
|
||||
The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license.
|
||||
|
||||
## Qwen3.8-27B Agention Precision
|
||||
|
||||
The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of
|
||||
Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image
|
||||
input and use a 32,768-token context by default.
|
||||
|
||||
Install with automatic variant selection:
|
||||
|
||||
```bash
|
||||
local-ai models install qwen3.8-27b-agention-iq4-xs
|
||||
```
|
||||
|
||||
To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs`
|
||||
or `--variant qwen3.8-27b-agention-q4-k-m` to the same command.
|
||||
The files use standard llama.cpp quantization types and the Apache-2.0 license.
|
||||
See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF)
|
||||
for quantization details. These entries do not enable MTP speculative decoding.
|
||||
|
||||
## Swift 1.5 Qwen3.8-27B GSQ-RCO
|
||||
|
||||
Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp.
|
||||
The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model.
|
||||
To select IQ3_S explicitly, run:
|
||||
|
||||
```bash
|
||||
local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
|
||||
```
|
||||
|
||||
The configurations use the embedded chat template and default to 32,768 context tokens.
|
||||
These builds support text chat only: the publisher has no verified vision projector for this release.
|
||||
They use standard GGUF files without MTP decoding.
|
||||
See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms.
|
||||
|
||||
## Sharp-Spark-X2.5-4B
|
||||
|
||||
Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp.
|
||||
The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants.
|
||||
To select the publisher's recommended Q6 build, run:
|
||||
|
||||
```bash
|
||||
local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6
|
||||
```
|
||||
|
||||
All builds use a 32,768-token default context and the embedded Sharp-Spark chat template.
|
||||
That template adds a terseness instruction to the system prompt.
|
||||
See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details.
|
||||
|
||||
## MiMo-V2.6-Distill-Qwen-9B
|
||||
|
||||
Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp.
|
||||
This MIT-licensed 9B Qwen3.5 fine-tune targets coding, agent tasks, and visual coding.
|
||||
The gallery groups Q4_K_M and Q8_0 builds as variants; both include the F16 vision projector.
|
||||
To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9b --variant mimo-v2.6-distill-qwen-9b-q8`.
|
||||
The configurations default to 32,768 context tokens and use the model's embedded chat template.
|
||||
See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details.
|
||||
|
||||
## Qwopus3.8 Flash V2
|
||||
|
||||
Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection:
|
||||
|
||||
```bash
|
||||
local-ai models install qwopus3.8-27b-flash-v2
|
||||
local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8
|
||||
```
|
||||
|
||||
Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector.
|
||||
Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256.
|
||||
This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks.
|
||||
See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations.
|
||||
|
||||
## ThinkingCap Qwen3.8-27B
|
||||
|
||||
Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input.
|
||||
The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector.
|
||||
LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly:
|
||||
|
||||
```bash
|
||||
local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8
|
||||
```
|
||||
|
||||
Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings.
|
||||
MTP speculative decoding is not enabled by these entries.
|
||||
The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE).
|
||||
Review that license for permitted use.
|
||||
|
||||
## Hemmingway-1
|
||||
|
||||
Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants.
|
||||
@@ -88,7 +181,7 @@ To use a gallery that needs authentication, such as a private GitHub repository
|
||||
|
||||
A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted.
|
||||
|
||||
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
|
||||
Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory.
|
||||
|
||||
```json
|
||||
GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
|
||||
@@ -142,10 +235,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out
|
||||
|
||||
### Signature verification
|
||||
|
||||
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
|
||||
An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use:
|
||||
|
||||
```json
|
||||
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
|
||||
GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}]
|
||||
```
|
||||
|
||||
The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends.
|
||||
@@ -164,9 +257,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK
|
||||
The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}).
|
||||
|
||||
{{% notice warning %}}
|
||||
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
|
||||
`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities.
|
||||
|
||||
With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log.
|
||||
{{% /notice %}}
|
||||
|
||||
### Official gallery publishing
|
||||
|
||||
The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA.
|
||||
|
||||
To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title.
|
||||
|
||||
### Private registries
|
||||
|
||||
A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use:
|
||||
@@ -198,10 +299,10 @@ GALLERIES=[{"name":"<GALLERY_NAME>", "url":"<GALLERY_URL"}]
|
||||
For example, to spell out the default `localai` repository, you can start `local-ai` with:
|
||||
|
||||
```
|
||||
GALLERIES=[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]
|
||||
GALLERIES=[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]
|
||||
```
|
||||
|
||||
`https://index.localai.io/models` is a caching mirror of the same index file, and the `github:` entry is the fallback used whenever it cannot be reached. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
|
||||
LocalAI tries `https://index.localai.io/models` first, GitHub second, and the signed OCI gallery last. The OCI artifact includes the repository-local base configurations, so reading those configurations does not require GitHub. Model weights and external URLs still require their original hosts. `github:mudler/LocalAI/gallery/index.yaml@master` is expanded automatically to `https://raw.githubusercontent.com/mudler/LocalAI/master/gallery/index.yaml`.
|
||||
|
||||
Note: the url are expanded automatically for `github` and `huggingface`, however `https://` and `http://` prefix works as well.
|
||||
|
||||
|
||||
@@ -340,6 +340,15 @@ curl http://localhost:8080/v1/responses \
|
||||
}'
|
||||
```
|
||||
|
||||
#### Streaming responses
|
||||
|
||||
Set `"stream": true` to receive Server-Sent Events. Each `response.output_item.added` event assigns an `output_index` to an item.
|
||||
Use that index and the item ID to associate later deltas and completion events with the same item.
|
||||
|
||||
If a request without explicit tools produces reasoning, the stream uses separate items for reasoning and answer text.
|
||||
Each item keeps its original index throughout the stream.
|
||||
The `response.completed` event includes both items in the same index order, followed by any automatically parsed tool calls.
|
||||
|
||||
#### Background Processing
|
||||
|
||||
Run requests in the background for long-running tasks:
|
||||
@@ -434,6 +443,11 @@ curl http://localhost:8080/v1/responses \
|
||||
}'
|
||||
```
|
||||
|
||||
For streaming requests with JSON tool output, LocalAI waits for the complete JSON
|
||||
object before emitting a completed `function_call` item. Arguments can span
|
||||
multiple tokens. Read the arguments from the `response.output_item.done` event
|
||||
before executing the tool.
|
||||
|
||||
#### Reasoning Configuration
|
||||
|
||||
Configure reasoning effort and summary style:
|
||||
|
||||
@@ -30,6 +30,10 @@ To install the dependencies follow the instructions below:
|
||||
{{< tabs >}}
|
||||
{{% tab title="Apple" %}}
|
||||
|
||||
To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later.
|
||||
Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts.
|
||||
Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables.
|
||||
|
||||
Install `xcode` from the App Store
|
||||
|
||||
```bash
|
||||
|
||||
@@ -63,9 +63,9 @@ against - and two modes:
|
||||
`proxy.provider` selects the auth scheme and (in translate mode) the wire
|
||||
format. Supported values: `openai`, `anthropic`.
|
||||
|
||||
API keys are loaded from either an environment variable (`api_key_env`) or a
|
||||
file (`api_key_file`). The key never appears in the config file or the admin
|
||||
UI; pick whichever fits your secret-management setup.
|
||||
If the upstream requires an API key, configure either an environment variable
|
||||
(`api_key_env`) or a file (`api_key_file`). The key never appears in the config
|
||||
file or the admin UI. If the upstream requires no API key, omit both fields.
|
||||
|
||||
### OpenAI passthrough
|
||||
|
||||
@@ -129,7 +129,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with
|
||||
|
||||
Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak
|
||||
the OpenAI chat-completions wire format. Use `provider: openai` with the
|
||||
provider's URL and API key:
|
||||
provider's URL and, if required, its API key:
|
||||
|
||||
```yaml
|
||||
name: llama-3-70b-via-together
|
||||
@@ -143,6 +143,37 @@ proxy:
|
||||
upstream_model: meta-llama/Llama-3-70b-chat-hf
|
||||
```
|
||||
|
||||
### Upstreams without an API key
|
||||
|
||||
For an OpenAI-compatible upstream that accepts requests without authentication,
|
||||
omit both `api_key_env` and `api_key_file`:
|
||||
|
||||
```yaml
|
||||
name: internal-chat-proxy
|
||||
backend: cloud-proxy
|
||||
|
||||
proxy:
|
||||
mode: passthrough
|
||||
provider: openai
|
||||
upstream_url: http://inference.internal:8000/v1/chat/completions
|
||||
upstream_model: my-model
|
||||
```
|
||||
|
||||
Replace the example URL and model name with your upstream's values. LocalAI
|
||||
loads this configuration without resolving a key and adds no upstream
|
||||
`Authorization` header. This also applies to OpenAI-compatible upstreams in
|
||||
translate mode.
|
||||
|
||||
Omitting both fields differs from setting `api_key_env` to an empty or unset
|
||||
environment variable: the latter causes a backend load error.
|
||||
|
||||
LocalAI's client authentication is separate. Clients must still authenticate
|
||||
to LocalAI when its authentication is enabled. LocalAI does not forward their
|
||||
`Authorization` header to the upstream.
|
||||
|
||||
An upstream without API keys can still require another authentication or
|
||||
payment protocol. Omitting these fields does not implement that protocol.
|
||||
|
||||
### Translate mode
|
||||
|
||||
In translate mode the cloud-proxy backend converts LocalAI's internal proto
|
||||
|
||||
@@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things:
|
||||
NVML work inside the container. With `--gpus all` alone (or
|
||||
`--runtime nvidia` without extra flags) only `compute` is wired in on
|
||||
some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility`
|
||||
to your `docker run`, or `capabilities: [gpu, utility]` in compose /
|
||||
Kubernetes device reservations.
|
||||
to your `docker run`. For Docker Compose with `driver: nvidia`, use
|
||||
`capabilities: [gpu, compute, utility]` on the device reservation.
|
||||
Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility`
|
||||
alone only provides monitoring libraries and tools.
|
||||
2. Pass `--init` to `docker run` (or `init: true` in compose) so the
|
||||
container has a proper PID 1 reaper - otherwise short-lived child
|
||||
processes like `nvidia-smi` can intermittently fail with
|
||||
|
||||
@@ -28,6 +28,29 @@ Returns available backends and currently loaded models.
|
||||
| `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM |
|
||||
| `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against |
|
||||
| `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) |
|
||||
| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes |
|
||||
|
||||
### Per-model VRAM
|
||||
|
||||
On Linux, `size_vram` reports resident device memory for the local backend
|
||||
process and its child processes. LocalAI reads `drm-resident-local*` and
|
||||
`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU.
|
||||
Host-memory regions are excluded. The reading includes buffers attributed
|
||||
to the backend, without separating weights, KV cache, and other allocations.
|
||||
See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html)
|
||||
for these counters.
|
||||
|
||||
The field is omitted when accounting is unavailable or incomplete. This
|
||||
includes external and distributed backends, macOS, proprietary NVIDIA
|
||||
drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters,
|
||||
and unreadable process information.
|
||||
A present value of `0` means the supported counters report zero bytes.
|
||||
Treat an absent field as unknown.
|
||||
|
||||
This is a snapshot of driver accounting, not a memory reservation. Shared
|
||||
buffers can appear in different clients' counters, and allocations can change
|
||||
during collection. Do not treat the sum across models as exclusive physical
|
||||
GPU usage. These readings do not replace capacity checks when scheduling work.
|
||||
|
||||
### Usage
|
||||
|
||||
@@ -49,6 +72,7 @@ curl http://localhost:8080/system
|
||||
{
|
||||
"id": "my-llama-model",
|
||||
"backend": "llama-cpp",
|
||||
"size_vram": 5368709120,
|
||||
"process": {
|
||||
"pid": 48213,
|
||||
"rss_bytes": 5368709120,
|
||||
|
||||
+881
-1
@@ -1,4 +1,96 @@
|
||||
---
|
||||
- name: "ternary-bonsai-2-27b"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
|
||||
- https://github.com/PrismML-Eng/llama.cpp
|
||||
description: |
|
||||
Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary
|
||||
transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits
|
||||
per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a
|
||||
Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's
|
||||
llama.cpp fork) instead of stock llama.cpp.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg
|
||||
overrides:
|
||||
backend: bonsai
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3
|
||||
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf
|
||||
- filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903
|
||||
uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf
|
||||
- name: "swift-qwen3.8-27b"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B.
|
||||
The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss.
|
||||
This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding.
|
||||
The weights use the Swift Open License v1.0.
|
||||
license: "swift-open-license-1.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
presence_penalty: 1.5
|
||||
repeat_penalty: 1
|
||||
temperature: 0.7
|
||||
top_k: 20
|
||||
top_p: 0.8
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab
|
||||
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf
|
||||
- filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a
|
||||
uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf
|
||||
- name: "ornith-1.5-9b-uncensored"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
@@ -205,7 +297,101 @@
|
||||
files:
|
||||
- filename: ds4flash.gguf
|
||||
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
|
||||
sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7
|
||||
sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
|
||||
- name: "qwopus3.8-27b-flash-v2"
|
||||
variants:
|
||||
- model: qwopus3.8-27b-flash-v2-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
|
||||
description: |
|
||||
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
|
||||
workloads. This Q4_K_M GGUF includes the F32 vision projector and uses
|
||||
llama.cpp's embedded chat template with MTP speculative decoding.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3
|
||||
- vision
|
||||
- multimodal
|
||||
- instruction-tuned
|
||||
- reasoning
|
||||
- mtp
|
||||
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf
|
||||
sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e
|
||||
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
|
||||
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
|
||||
- name: "qwopus3.8-27b-flash-v2-q8"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash
|
||||
- https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF
|
||||
description: |
|
||||
Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent
|
||||
workloads. This Q8_0 GGUF includes the F32 vision projector and uses
|
||||
llama.cpp's embedded chat template with MTP speculative decoding.
|
||||
license: "apache-2.0"
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3
|
||||
- vision
|
||||
- multimodal
|
||||
- instruction-tuned
|
||||
- reasoning
|
||||
- mtp
|
||||
icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
- spec_n_max:6
|
||||
- spec_p_min:0.75
|
||||
parameters:
|
||||
model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf
|
||||
sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78
|
||||
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf
|
||||
sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6
|
||||
- name: "qwopus3.8-27b-flash"
|
||||
variants:
|
||||
- model: qwopus3.8-27b-flash-q8
|
||||
@@ -302,6 +488,188 @@
|
||||
- filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-MTP-Q4_K_M/mmproj-F32.gguf
|
||||
uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-GGUF/resolve/e146d61e88782677805b3b68ad3adf8674dde80d/mmproj-F32.gguf
|
||||
sha256: 52e6818e4d18eea010c50e5245eaa10a8cc3dcc30efea4ff60cbad8abf5669e1
|
||||
- name: mimo-v2.6-distill-qwen-9b
|
||||
variants:
|
||||
- model: mimo-v2.6-distill-qwen-9b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B
|
||||
- https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF
|
||||
description: |
|
||||
MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding.
|
||||
This Q4_K_M GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector.
|
||||
license: mit
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- vision
|
||||
- multimodal
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
|
||||
files:
|
||||
- filename: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
|
||||
sha256: 4bca6f18c73f72270c7a20c2ea2bea581de8246e318714277120369d34048c81
|
||||
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf
|
||||
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
|
||||
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
- name: mimo-v2.6-distill-qwen-9b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B
|
||||
- https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF
|
||||
description: |
|
||||
MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding.
|
||||
This Q8_0 GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector.
|
||||
license: mit
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- vision
|
||||
- multimodal
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
|
||||
files:
|
||||
- filename: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
|
||||
sha256: 2fad0aa11bb9e7aa491ff12f768954f9dd0a6e7d4ce4a897ca73ec420f3b90ae
|
||||
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf
|
||||
- filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576
|
||||
uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf
|
||||
- name: thinkingcap-qwen3.8-27b
|
||||
variants:
|
||||
- model: thinkingcap-qwen3.8-27b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
|
||||
This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
|
||||
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
|
||||
license: polyform-small-business-1.0.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- vision
|
||||
- multimodal
|
||||
- reasoning
|
||||
last_checked: "2026-09-27"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
files:
|
||||
- filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf
|
||||
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
- name: thinkingcap-qwen3.8-27b-q8
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B
|
||||
- https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF
|
||||
description: |
|
||||
ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input.
|
||||
This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector.
|
||||
Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use.
|
||||
license: polyform-small-business-1.0.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- vision
|
||||
- multimodal
|
||||
- reasoning
|
||||
last_checked: "2026-09-27"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
parameters:
|
||||
model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
files:
|
||||
- filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf
|
||||
- filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1
|
||||
uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf
|
||||
- name: hemmingway-1
|
||||
variants:
|
||||
- model: hemmingway-1-q8
|
||||
@@ -4681,6 +5049,110 @@
|
||||
- filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf
|
||||
uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf
|
||||
sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41
|
||||
- name: cyber-tiel-coder-35b-a3b-q4-mtp
|
||||
variants:
|
||||
- model: cyber-tiel-coder-35b-a3b-q8-mtp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
license: mit
|
||||
urls:
|
||||
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
|
||||
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
|
||||
description: |
|
||||
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
|
||||
based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes
|
||||
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- moe
|
||||
- coding
|
||||
- tools
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
parameters:
|
||||
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf
|
||||
sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a
|
||||
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
|
||||
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
|
||||
- name: cyber-tiel-coder-35b-a3b-q8-mtp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
license: mit
|
||||
urls:
|
||||
- https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated
|
||||
- https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP
|
||||
description: |
|
||||
Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters,
|
||||
based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes
|
||||
MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- moe
|
||||
- coding
|
||||
- tools
|
||||
- vision
|
||||
- multimodal
|
||||
- mtp
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
- spec_type:draft-mtp
|
||||
parameters:
|
||||
model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf
|
||||
sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961
|
||||
- filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf
|
||||
sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837
|
||||
- &tiel-coder-35b-a3b
|
||||
name: "tiel-coder-35b-a3b-q4"
|
||||
variants:
|
||||
@@ -5161,6 +5633,104 @@
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf
|
||||
uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf
|
||||
sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545
|
||||
- name: qwen3.8-27b-agention-iq4-xs
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
variants:
|
||||
- model: qwen3.8-27b-agention-q4-k-m
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- name: qwen3.8-27b-agention-q4-k-m
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- &qwen3-8-27b
|
||||
name: "qwen3.8-27b-q4"
|
||||
variants:
|
||||
@@ -5447,6 +6017,178 @@
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf
|
||||
uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf
|
||||
sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco"
|
||||
variants:
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs
|
||||
- model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf
|
||||
- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
urls:
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b
|
||||
- https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF
|
||||
license: "swift-open-license-1.0"
|
||||
description: |
|
||||
Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks.
|
||||
This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp.
|
||||
Text chat only; the publisher provides no verified vision projector for this release.
|
||||
The weights use the Swift Open License v1.0.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- reasoning
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
min_p: 0
|
||||
model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
presence_penalty: 0
|
||||
repeat_penalty: 1
|
||||
temperature: 1
|
||||
top_k: 20
|
||||
top_p: 0.95
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786
|
||||
uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf
|
||||
- !!merge <<: *qwen3-8-27b
|
||||
name: "qwen3.8-27b-gsq-rco-iq2-xs"
|
||||
variants: []
|
||||
@@ -5639,6 +6381,120 @@
|
||||
- filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf
|
||||
uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf
|
||||
sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442
|
||||
- name: sharp-spark-x2.5-4b
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
variants:
|
||||
- model: sharp-spark-x2.5-4b-q5
|
||||
- model: sharp-spark-x2.5-4b-q6
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q4_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q5
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q5_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q6
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/XHToken/Spark-X2.5-4B
|
||||
- https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF
|
||||
description: |
|
||||
Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model
|
||||
with an adjusted chat template for coding. This Q6_K_XL build uses the
|
||||
embedded Sharp-Spark template and a 32K-token default context.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- coding
|
||||
- reasoning
|
||||
last_checked: "2026-09-26"
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
|
||||
|
||||
- &spark-x2-5-4b
|
||||
name: "spark-x2.5-4b-q4"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
@@ -45126,6 +45982,12 @@
|
||||
sha256: ""
|
||||
uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors
|
||||
- name: kimodo-soma-rp
|
||||
variants:
|
||||
- model: kimodo-soma-rp-bf16
|
||||
- model: kimodo-soma-rp-q4_k
|
||||
- model: kimodo-soma-rp-q4_k_m
|
||||
- model: kimodo-soma-rp-q5_k
|
||||
- model: kimodo-soma-rp-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45313,6 +46175,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-soma-seed
|
||||
variants:
|
||||
- model: kimodo-soma-seed-bf16
|
||||
- model: kimodo-soma-seed-q4_k
|
||||
- model: kimodo-soma-seed-q4_k_m
|
||||
- model: kimodo-soma-seed-q5_k
|
||||
- model: kimodo-soma-seed-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45500,6 +46368,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-g1-rp
|
||||
variants:
|
||||
- model: kimodo-g1-rp-bf16
|
||||
- model: kimodo-g1-rp-q4_k
|
||||
- model: kimodo-g1-rp-q4_k_m
|
||||
- model: kimodo-g1-rp-q5_k
|
||||
- model: kimodo-g1-rp-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
@@ -45687,6 +46561,12 @@
|
||||
uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf
|
||||
sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f
|
||||
- name: kimodo-g1-seed
|
||||
variants:
|
||||
- model: kimodo-g1-seed-bf16
|
||||
- model: kimodo-g1-seed-q4_k
|
||||
- model: kimodo-g1-seed-q4_k_m
|
||||
- model: kimodo-g1-seed-q5_k
|
||||
- model: kimodo-g1-seed-q6_k
|
||||
url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master
|
||||
backend: kimodocpp
|
||||
urls:
|
||||
|
||||
@@ -259,7 +259,7 @@ require (
|
||||
github.com/kevinburke/ssh_config v1.2.0 // indirect
|
||||
github.com/labstack/gommon v0.4.2 // indirect
|
||||
github.com/mschoch/smat v0.2.0 // indirect
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1
|
||||
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e
|
||||
github.com/mudler/localrecall v0.6.5 // indirect
|
||||
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145
|
||||
github.com/olekukonko/tablewriter v0.0.5 // indirect
|
||||
@@ -535,7 +535,7 @@ require (
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect
|
||||
golang.org/x/mod v0.36.0 // indirect
|
||||
golang.org/x/sync v0.20.0
|
||||
golang.org/x/sys v0.45.0 // indirect
|
||||
golang.org/x/sys v0.45.0
|
||||
golang.org/x/term v0.43.0
|
||||
golang.org/x/text v0.37.0
|
||||
golang.org/x/tools v0.45.0 // indirect
|
||||
|
||||
@@ -1032,6 +1032,8 @@ github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxF
|
||||
github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q=
|
||||
github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY=
|
||||
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY=
|
||||
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4=
|
||||
github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU=
|
||||
|
||||
@@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
|
||||
skippedFiles := 0
|
||||
skippedBytes := int64(0)
|
||||
tasks := make([]downloader.FileTask, 0, len(snapshot.Files))
|
||||
// Sibling manifests are read once, before the staging loop, so the
|
||||
// per-file reuse lookups below never re-read or re-parse them.
|
||||
siblings := loadSiblingCandidates(modelsPath, spec, layout)
|
||||
for index, file := range snapshot.Files {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return Result{}, err
|
||||
@@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec
|
||||
skippedBytes += file.Size
|
||||
continue
|
||||
}
|
||||
// Before reaching for the network, consult committed sibling trees for the
|
||||
// same Source (type+endpoint+repo+revision). A narrower allow_patterns
|
||||
// request gets a different CacheKey, so committedResult misses even though a
|
||||
// broader sibling already holds this exact file; reusing it avoids a
|
||||
// redundant re-download of tens of gigabytes. The match is re-verified
|
||||
// through verifyDownloadedFile (full SHA-256), never size-only, and a broader
|
||||
// request can never inherit a narrower sibling's gaps because each file is
|
||||
// matched individually against the sibling's manifest.
|
||||
if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok {
|
||||
manifest.Files[taskIndex] = entry
|
||||
completedBytes.Add(file.Size)
|
||||
skippedFiles++
|
||||
skippedBytes += file.Size
|
||||
continue
|
||||
}
|
||||
nameSum := sha256.Sum256([]byte(file.Path))
|
||||
blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:]))
|
||||
blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel))
|
||||
@@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest
|
||||
return entry, true
|
||||
}
|
||||
|
||||
// siblingCandidate is one committed sibling artifact tree that shares this
|
||||
// request's Source (type+endpoint+repo+revision), with its manifest files
|
||||
// indexed by path.
|
||||
type siblingCandidate struct {
|
||||
final string
|
||||
filesByPath map[string][]ManifestFile
|
||||
}
|
||||
|
||||
// loadSiblingCandidates reads the committed sibling manifest set once, before
|
||||
// the staging loop. Doing it per file instead would re-read and re-parse every
|
||||
// sibling manifest for every file — 20 committed siblings and a 300-file
|
||||
// snapshot means 6000 manifest reads before the first byte is fetched.
|
||||
//
|
||||
// The current artifact's own committed tree is excluded: it is either absent
|
||||
// (the reason materializeLocked is running) or already handled by
|
||||
// committedResult's exact-key fast path.
|
||||
func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate {
|
||||
if spec.Resolved == nil || layout.Final == "" {
|
||||
return nil
|
||||
}
|
||||
siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface")
|
||||
entries, err := os.ReadDir(siblingsRoot)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var candidates []siblingCandidate
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() {
|
||||
continue
|
||||
}
|
||||
siblingFinal := filepath.Join(siblingsRoot, entry.Name())
|
||||
if siblingFinal == layout.Final {
|
||||
continue
|
||||
}
|
||||
siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json"))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
siblingArtifact := siblingManifest.Artifact
|
||||
if siblingArtifact.Resolved == nil ||
|
||||
siblingArtifact.Source.Type != spec.Source.Type ||
|
||||
siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint ||
|
||||
siblingArtifact.Source.Repo != spec.Source.Repo ||
|
||||
siblingArtifact.Resolved.Revision != spec.Resolved.Revision {
|
||||
continue
|
||||
}
|
||||
byPath := make(map[string][]ManifestFile, len(siblingManifest.Files))
|
||||
for _, f := range siblingManifest.Files {
|
||||
byPath[f.Path] = append(byPath[f.Path], f)
|
||||
}
|
||||
candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath})
|
||||
}
|
||||
return candidates
|
||||
}
|
||||
|
||||
// reuseFromCommittedSibling looks for a file already committed under a sibling
|
||||
// artifact tree — same Source (type+endpoint+repo+revision), different
|
||||
// allow/ignore patterns — and stages it for this writer instead of fetching.
|
||||
// A narrower allow_patterns request gets a different CacheKey (path.go:62), so
|
||||
// committedResult misses and materializeLocked would otherwise re-download
|
||||
// files an already-committed broader sibling already holds.
|
||||
//
|
||||
// The match is never size-only: the sibling file is re-hashed through the
|
||||
// shared verifyDownloadedFile against the current request's SnapshotFile (its
|
||||
// LFS or git blob OID), so the staged entry is byte-for-byte identical to a
|
||||
// fresh download. A broader request can never stand in for files a narrower
|
||||
// sibling lacks, because each requested file is matched individually against
|
||||
// the sibling's manifest file set. Hard-link keeps the shared models volume
|
||||
// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the
|
||||
// kernel cannot hard-link.
|
||||
func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) {
|
||||
snapshotRel := path.Join("snapshot", file.Path)
|
||||
snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel))
|
||||
for _, sibling := range candidates {
|
||||
for _, siblingFile := range sibling.filesByPath[file.Path] {
|
||||
if siblingFile.Size != file.Size {
|
||||
continue
|
||||
}
|
||||
siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path))
|
||||
verified, err := verifyDownloadedFile(siblingPath, file)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil {
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
_ = root.Remove(snapshotRel)
|
||||
if err := linkOrCopy(siblingPath, snapshotAbs); err != nil {
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
return verified, true
|
||||
}
|
||||
}
|
||||
return ManifestFile{}, false
|
||||
}
|
||||
|
||||
// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only
|
||||
// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking
|
||||
// keeps the shared models volume neutral — a narrowed request does not double
|
||||
// the storage of a broad sibling's files.
|
||||
func linkOrCopy(src, dst string) error {
|
||||
if err := os.Link(src, dst); err == nil {
|
||||
return nil
|
||||
} else if !errors.Is(err, syscall.EXDEV) {
|
||||
return err
|
||||
}
|
||||
in, err := os.Open(src)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = in.Close() }()
|
||||
out, err := os.Create(dst)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := io.Copy(out, in); err != nil {
|
||||
_ = out.Close()
|
||||
_ = os.Remove(dst)
|
||||
return err
|
||||
}
|
||||
return out.Close()
|
||||
}
|
||||
|
||||
func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) {
|
||||
file, err := os.Open(fileName)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
package modelartifacts_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
hfapi "github.com/mudler/LocalAI/pkg/huggingface-api"
|
||||
"github.com/mudler/LocalAI/pkg/modelartifacts"
|
||||
)
|
||||
|
||||
const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567"
|
||||
|
||||
// recordingResolver serves a fixed full file set filtered by each request's
|
||||
// allow/ignore patterns, so a narrower request genuinely resolves to a strict
|
||||
// subset of a broader sibling's files. The HTTP server behind it records every
|
||||
// fetch, which is the signal the sibling-reuse fix is verified through. The
|
||||
// function under fix is never mocked: a real Manager drives the real staging +
|
||||
// commit path against this stub collaborator.
|
||||
type recordingResolver struct {
|
||||
endpoint string
|
||||
repo string
|
||||
files []hfapi.SnapshotFile
|
||||
server *httptest.Server
|
||||
|
||||
mu sync.Mutex
|
||||
fetched map[string]int
|
||||
}
|
||||
|
||||
func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver {
|
||||
r := &recordingResolver{
|
||||
endpoint: "https://huggingface.co",
|
||||
repo: "owner/repo",
|
||||
files: files,
|
||||
fetched: map[string]int{},
|
||||
}
|
||||
r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) {
|
||||
name := strings.TrimPrefix(req.URL.Path, "/file/")
|
||||
body, ok := contents[name]
|
||||
if !ok {
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
return
|
||||
}
|
||||
r.mu.Lock()
|
||||
r.fetched[name]++
|
||||
r.mu.Unlock()
|
||||
w.Header().Set("Content-Length", strconv.Itoa(len(body)))
|
||||
_, _ = w.Write(body)
|
||||
}))
|
||||
return r
|
||||
}
|
||||
|
||||
func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) {
|
||||
files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns)
|
||||
if err != nil {
|
||||
return hfapi.Snapshot{}, err
|
||||
}
|
||||
out := make([]hfapi.SnapshotFile, len(files))
|
||||
for i, f := range files {
|
||||
f.URL = r.server.URL + "/file/" + f.Path
|
||||
out[i] = f
|
||||
}
|
||||
return hfapi.Snapshot{
|
||||
Endpoint: r.endpoint, Repo: r.repo,
|
||||
RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (r *recordingResolver) fetchCount(path string) int {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
return r.fetched[path]
|
||||
}
|
||||
|
||||
func (r *recordingResolver) resetFetches() {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
r.fetched = map[string]int{}
|
||||
}
|
||||
|
||||
func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile {
|
||||
paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"}
|
||||
files := make([]hfapi.SnapshotFile, 0, len(paths))
|
||||
for _, p := range paths {
|
||||
sum := sha256.Sum256(contents[p])
|
||||
files = append(files, hfapi.SnapshotFile{
|
||||
Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]),
|
||||
})
|
||||
}
|
||||
return files
|
||||
}
|
||||
|
||||
// The narrow-request case proves the fix for #11047:
|
||||
// a request with narrower allow_patterns (a strict subset) reuses files an
|
||||
// already-committed broader sibling holds, hard-linking instead of re-fetching.
|
||||
//
|
||||
// On master this is RED: a narrower allow_patterns set hashes to a different
|
||||
// CacheKey (path.go:62), so committedResult misses and materializeLocked
|
||||
// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On
|
||||
// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling,
|
||||
// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0,
|
||||
// os.SameFile true).
|
||||
var _ = Describe("committed sibling reuse", func() {
|
||||
It("reuses files from a broader committed sibling", func() {
|
||||
contents := map[string][]byte{
|
||||
"a/first.bin": []byte("first-file-bytes"),
|
||||
"b/second.bin": []byte("second-file-bytes-longer"),
|
||||
"c/third.bin": []byte("third-file"),
|
||||
}
|
||||
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
|
||||
defer resolver.server.Close()
|
||||
|
||||
modelsPath := GinkgoT().TempDir()
|
||||
manager := modelartifacts.NewManager(resolver,
|
||||
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
|
||||
|
||||
// Commit the broad sibling: all three files, fetched from the resolver.
|
||||
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
}}
|
||||
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(broad.CacheHit).To(BeFalse())
|
||||
Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0),
|
||||
"the broad sibling must have fetched a/first.bin to commit it")
|
||||
|
||||
resolver.resetFetches()
|
||||
|
||||
// Narrowed request: a strict subset of the broad sibling's file set.
|
||||
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
AllowPatterns: []string{"a/first.bin"},
|
||||
}}
|
||||
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(narrow.CacheHit).To(BeFalse())
|
||||
|
||||
// (b) The sibling-present file must NOT be re-fetched: zero fetches. This is
|
||||
// the assertion that is RED on master (one fetch) and GREEN on the branch.
|
||||
Expect(resolver.fetchCount("a/first.bin")).To(Equal(0),
|
||||
"a/first.bin must be reused from the committed broad sibling, not re-fetched")
|
||||
|
||||
// (a) The narrowed tree's staged file is the same inode as the broad
|
||||
// sibling's file (hard-link), not a freshly downloaded second copy. RED on
|
||||
// master (separate file), GREEN on the branch (hard-link).
|
||||
broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin")
|
||||
narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin")
|
||||
broadInfo, err := os.Stat(broadFile)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
narrowInfo, err := os.Stat(narrowFile)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(),
|
||||
"the narrowed request must hard-link the broad sibling's file rather than store a second copy")
|
||||
|
||||
// The reused bytes are intact end to end.
|
||||
Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"]))
|
||||
})
|
||||
|
||||
// The broader-request case is the manifest file-set guard: a broader request
|
||||
// against a narrower committed sibling must still fetch the files the sibling
|
||||
// lacks and commit a complete tree. Sibling-reuse can never serve an incomplete
|
||||
// model as complete, because each requested file is matched individually against
|
||||
// the sibling's manifest.
|
||||
It("fetches files missing from a narrower committed sibling", func() {
|
||||
contents := map[string][]byte{
|
||||
"a/first.bin": []byte("first-file-bytes"),
|
||||
"b/second.bin": []byte("second-file-bytes-longer"),
|
||||
"c/third.bin": []byte("third-file"),
|
||||
}
|
||||
resolver := newRecordingResolver(siblingReuseFiles(contents), contents)
|
||||
defer resolver.server.Close()
|
||||
|
||||
modelsPath := GinkgoT().TempDir()
|
||||
manager := modelartifacts.NewManager(resolver,
|
||||
modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} }))
|
||||
|
||||
// Commit a NARROW sibling first: only a/first.bin and b/second.bin.
|
||||
narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
AllowPatterns: []string{"a/first.bin", "b/second.bin"},
|
||||
}}
|
||||
narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
narrowPaths := make([]string, 0, len(narrow.Manifest.Files))
|
||||
for _, f := range narrow.Manifest.Files {
|
||||
narrowPaths = append(narrowPaths, f.Path)
|
||||
}
|
||||
Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"}))
|
||||
|
||||
resolver.resetFetches()
|
||||
|
||||
// A BROADER request asks for all three files, including c/third.bin which the
|
||||
// narrow sibling does not hold.
|
||||
broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{
|
||||
Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo",
|
||||
}}
|
||||
broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
// The file the narrow sibling lacks MUST be fetched: sibling-reuse must not
|
||||
// inherit a narrower tree's gaps as if the broad request were complete.
|
||||
Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0),
|
||||
"c/third.bin is absent from the narrow sibling and must be fetched, not served as complete")
|
||||
|
||||
// The broad tree's manifest file set is exactly the full set — never the
|
||||
// narrow sibling's subset. This file-set comparison proves no incomplete model
|
||||
// is ever served as complete via sibling-reuse.
|
||||
broadPaths := make([]string, 0, len(broad.Manifest.Files))
|
||||
for _, f := range broad.Manifest.Files {
|
||||
broadPaths = append(broadPaths, f.Path)
|
||||
}
|
||||
Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"}))
|
||||
|
||||
// Every file is present on disk with the right bytes after commit.
|
||||
for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} {
|
||||
Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))).
|
||||
To(Equal(contents[p]))
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,165 @@
|
||||
//go:build linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ProcessVRAM reports device-local resident bytes accounted to a process tree
|
||||
// by DRM. Unsupported or incomplete accounting returns false, not a measured zero.
|
||||
func ProcessVRAM(pid int) (uint64, bool) {
|
||||
return processVRAM("/proc", pid)
|
||||
}
|
||||
|
||||
func processVRAM(procRoot string, pid int) (uint64, bool) {
|
||||
if pid <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
clients := map[string]uint64{}
|
||||
seen := map[int]bool{}
|
||||
pending := []int{pid}
|
||||
for len(pending) > 0 {
|
||||
current := pending[len(pending)-1]
|
||||
pending = pending[:len(pending)-1]
|
||||
if seen[current] {
|
||||
continue
|
||||
}
|
||||
seen[current] = true
|
||||
base := filepath.Join(procRoot, strconv.Itoa(current))
|
||||
fds, err := os.ReadDir(filepath.Join(base, "fd"))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
for _, fd := range fds {
|
||||
target, err := os.Readlink(filepath.Join(base, "fd", fd.Name()))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
// A mixed DRM/NVIDIA tree cannot provide a complete DRM reading.
|
||||
if strings.HasPrefix(target, "/dev/nvidia") {
|
||||
return 0, false
|
||||
}
|
||||
if !strings.HasPrefix(target, "/dev/dri/render") {
|
||||
// Primary nodes can also own allocations. Until their device
|
||||
// identity is resolved, omitting them would undercount the tree.
|
||||
if strings.HasPrefix(target, "/dev/dri/") {
|
||||
return 0, false
|
||||
}
|
||||
continue
|
||||
}
|
||||
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
|
||||
// base adds an integer PID, and fd.Name comes from os.ReadDir.
|
||||
// The kernel supplies these path components, not request input.
|
||||
data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name()))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
client, used, ok := drmResidentClient(data)
|
||||
if !ok {
|
||||
return 0, false
|
||||
}
|
||||
key := target + ":" + client
|
||||
// dup() and fork() can expose the same client more than once. The
|
||||
// snapshot is not atomic; retain its largest observed reading.
|
||||
clients[key] = max(clients[key], used)
|
||||
}
|
||||
|
||||
// A worker may be spawned by any thread, not just the thread leader.
|
||||
tasks, err := os.ReadDir(filepath.Join(base, "task"))
|
||||
if err != nil || len(tasks) == 0 {
|
||||
return 0, false
|
||||
}
|
||||
for _, task := range tasks {
|
||||
// #nosec G304 -- procRoot is /proc in production (a temp dir in tests);
|
||||
// base adds an integer PID, and task.Name comes from os.ReadDir.
|
||||
// The kernel supplies these path components, not request input.
|
||||
data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children"))
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
for _, raw := range strings.Fields(string(data)) {
|
||||
child, err := strconv.Atoi(raw)
|
||||
if err != nil || child <= 0 {
|
||||
return 0, false
|
||||
}
|
||||
pending = append(pending, child)
|
||||
}
|
||||
}
|
||||
}
|
||||
var total uint64
|
||||
for _, used := range clients {
|
||||
if used > math.MaxUint64-total {
|
||||
return 0, false
|
||||
}
|
||||
total += used
|
||||
}
|
||||
return total, len(clients) > 0
|
||||
}
|
||||
|
||||
func drmResidentClient(data []byte) (string, uint64, bool) {
|
||||
var client string
|
||||
var total uint64
|
||||
found := false
|
||||
scanner := bufio.NewScanner(bytes.NewReader(data))
|
||||
for scanner.Scan() {
|
||||
key, value, ok := strings.Cut(scanner.Text(), ":")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if key == "drm-client-id" {
|
||||
id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64)
|
||||
if err != nil {
|
||||
return "", 0, false
|
||||
}
|
||||
client = strconv.FormatUint(id, 10)
|
||||
}
|
||||
region, resident := strings.CutPrefix(key, "drm-resident-")
|
||||
if !resident || !isVRAMRegion(region) {
|
||||
continue
|
||||
}
|
||||
used, ok := drmResidentBytes(value)
|
||||
if !ok || used > math.MaxUint64-total {
|
||||
return "", 0, false
|
||||
}
|
||||
total += used
|
||||
found = true
|
||||
}
|
||||
return client, total, scanner.Err() == nil && client != "" && found
|
||||
}
|
||||
|
||||
func drmResidentBytes(value string) (uint64, bool) {
|
||||
fields := strings.Fields(value)
|
||||
if len(fields) == 0 || len(fields) > 2 {
|
||||
return 0, false
|
||||
}
|
||||
n, err := strconv.ParseUint(fields[0], 10, 64)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
unit := uint64(1)
|
||||
if len(fields) == 2 {
|
||||
switch strings.ToLower(fields[1]) {
|
||||
case "b":
|
||||
case "kib":
|
||||
unit = 1 << 10
|
||||
case "mib":
|
||||
unit = 1 << 20
|
||||
case "gib":
|
||||
unit = 1 << 30
|
||||
default:
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
if n > math.MaxUint64/unit {
|
||||
return 0, false
|
||||
}
|
||||
return n * unit, true
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
//go:build linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("ProcessVRAM", func() {
|
||||
var root string
|
||||
write := func(path, contents string) {
|
||||
Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed())
|
||||
Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed())
|
||||
}
|
||||
addProcess := func(pid int, children string) {
|
||||
base := filepath.Join(root, strconv.Itoa(pid))
|
||||
Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed())
|
||||
write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children)
|
||||
}
|
||||
addFD := func(pid, fd int, render, info string) {
|
||||
base := filepath.Join(root, strconv.Itoa(pid))
|
||||
name := strconv.Itoa(fd)
|
||||
Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed())
|
||||
write(filepath.Join(base, "fdinfo", name), info)
|
||||
}
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
root, err = os.MkdirTemp("", "process-vram-")
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
DeferCleanup(os.RemoveAll, root)
|
||||
addProcess(100, "")
|
||||
})
|
||||
|
||||
It("sums resident device memory across GPUs and child processes without duplicate clients", func() {
|
||||
write(filepath.Join(root, "100/task/101/children"), "200")
|
||||
addProcess(200, "")
|
||||
info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n"
|
||||
addFD(100, 3, "renderD128", info)
|
||||
addFD(100, 4, "renderD128", info)
|
||||
addFD(200, 3, "renderD128", info)
|
||||
addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n")
|
||||
used, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(used).To(Equal(uint64(384 * 1024 * 1024)))
|
||||
})
|
||||
|
||||
It("distinguishes a measured zero from unavailable accounting", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n")
|
||||
used, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(used).To(BeZero())
|
||||
})
|
||||
|
||||
DescribeTable("does not invent readings from unsupported or invalid accounting",
|
||||
func(info string) {
|
||||
addFD(100, 3, "renderD128", info)
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
},
|
||||
Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"),
|
||||
Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"),
|
||||
Entry("no client identity", "drm-resident-vram0: 128 MiB\n"),
|
||||
Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"),
|
||||
Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"),
|
||||
Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"),
|
||||
)
|
||||
|
||||
It("omits a partial reading if a child cannot be inspected", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
write(filepath.Join(root, "100/task/100/children"), "200")
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
})
|
||||
|
||||
It("omits a partial reading if another DRM client lacks accounting", func() {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
addFD(100, 4, "renderD129", "drm-client-id: 8\n")
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
})
|
||||
|
||||
DescribeTable("omits mixed readings with unsupported GPU descriptors",
|
||||
func(target string) {
|
||||
addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n")
|
||||
Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed())
|
||||
_, ok := processVRAM(root, 100)
|
||||
Expect(ok).To(BeFalse())
|
||||
},
|
||||
Entry("primary DRM node", "/dev/dri/card0"),
|
||||
Entry("NVIDIA device", "/dev/nvidia0"),
|
||||
)
|
||||
|
||||
It("returns unavailable for missing processes or no DRM descriptors", func() {
|
||||
for _, pid := range []int{-1, 0, 100, 999} {
|
||||
_, ok := processVRAM(root, pid)
|
||||
Expect(ok).To(BeFalse())
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,9 @@
|
||||
//go:build !linux
|
||||
|
||||
// SPDX-License-Identifier: MIT
|
||||
package xsysinfo
|
||||
|
||||
// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting.
|
||||
func ProcessVRAM(pid int) (uint64, bool) {
|
||||
return 0, false
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
// Package the official index and its repository-local base configurations.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
func main() {
|
||||
if len(os.Args) != 4 {
|
||||
fmt.Fprintln(os.Stderr, "usage: gallery REPOSITORY {gallery|backend} OUTPUT")
|
||||
os.Exit(1)
|
||||
}
|
||||
if err := packageGallery(os.Args[1], os.Args[2], os.Args[3]); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func packageGallery(root, source, output string) error {
|
||||
if source != "gallery" && source != "backend" {
|
||||
return fmt.Errorf("unsupported gallery directory %q", source)
|
||||
}
|
||||
repository, err := os.OpenRoot(root)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = repository.Close() }()
|
||||
body, err := repository.ReadFile(filepath.Join(source, "index.yaml"))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var doc yaml.Node
|
||||
if err := yaml.Unmarshal(body, &doc); err != nil {
|
||||
return err
|
||||
}
|
||||
// The build operator explicitly selects the output directory via the CLI.
|
||||
if err := os.MkdirAll(output, 0700); err != nil { // #nosec G703 -- caller-selected output root
|
||||
return err
|
||||
}
|
||||
destination, err := os.OpenRoot(output)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = destination.Close() }()
|
||||
// Keep the tree relative to the repository root so repeated base configs
|
||||
// share a layer, even when an index refers outside its own directory.
|
||||
const prefix = "github:mudler/LocalAI/"
|
||||
var walk func(*yaml.Node) error
|
||||
walk = func(n *yaml.Node) error {
|
||||
if n.Kind == yaml.MappingNode {
|
||||
for i := 0; i < len(n.Content); i += 2 {
|
||||
value := n.Content[i+1]
|
||||
if n.Content[i].Value != "url" || value.Kind != yaml.ScalarNode || !strings.HasPrefix(value.Value, prefix) || !strings.HasSuffix(value.Value, "@master") {
|
||||
continue
|
||||
}
|
||||
path := strings.TrimSuffix(strings.TrimPrefix(value.Value, prefix), "@master")
|
||||
if !filepath.IsLocal(path) {
|
||||
return fmt.Errorf("base config escapes repository: %q", path)
|
||||
}
|
||||
config, err := repository.ReadFile(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := destination.MkdirAll(filepath.Dir(path), 0700); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := destination.WriteFile(path, config, 0600); err != nil {
|
||||
return err
|
||||
}
|
||||
value.Value = filepath.ToSlash(path)
|
||||
}
|
||||
}
|
||||
for _, child := range n.Content {
|
||||
if err := walk(child); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if err := walk(&doc); err != nil {
|
||||
return err
|
||||
}
|
||||
body, err = yaml.Marshal(&doc)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return destination.WriteFile("index.yaml", body, 0600)
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package main
|
||||
|
||||
import (
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
"gopkg.in/yaml.v3"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestGalleryPackage(t *testing.T) { RegisterFailHandler(Fail); RunSpecs(t, "Gallery packaging") }
|
||||
|
||||
var _ = Describe("Gallery packaging", func() {
|
||||
It("packages both official indexes with every repository-local base available offline", func() {
|
||||
for _, source := range []string{"gallery", "backend"} {
|
||||
out := GinkgoT().TempDir()
|
||||
Expect(packageGallery("../../..", source, out)).To(Succeed())
|
||||
body, err := os.ReadFile(filepath.Join(out, "index.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
var entries []map[string]any
|
||||
Expect(yaml.Unmarshal(body, &entries)).To(Succeed())
|
||||
Expect(entries).ToNot(BeEmpty())
|
||||
for _, entry := range entries {
|
||||
url, _ := entry["url"].(string)
|
||||
Expect(url).ToNot(HavePrefix("github:mudler/LocalAI/"))
|
||||
if strings.HasPrefix(url, "gallery/") {
|
||||
Expect(filepath.Join(out, url)).To(BeAnExistingFile())
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
It("bundles local base configs and preserves external URLs and YAML aliases", func() {
|
||||
root := GinkgoT().TempDir()
|
||||
Expect(os.MkdirAll(filepath.Join(root, "gallery"), 0755)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/base.yaml"), []byte("backend: llama-cpp\n"), 0644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- &base\n name: first\n url: github:mudler/LocalAI/gallery/base.yaml@master\n- <<: *base\n name: second\n- name: external\n url: https://example.com/config.yaml\n"), 0644)).To(Succeed())
|
||||
out := filepath.Join(root, "out")
|
||||
Expect(packageGallery(root, "gallery", out)).To(Succeed())
|
||||
data, err := os.ReadFile(filepath.Join(out, "index.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
var entries []map[string]any
|
||||
Expect(yaml.Unmarshal(data, &entries)).To(Succeed())
|
||||
Expect(entries[0]["url"]).To(Equal("gallery/base.yaml"))
|
||||
Expect(entries[1]["url"]).To(Equal("gallery/base.yaml"))
|
||||
Expect(entries[2]["url"]).To(Equal("https://example.com/config.yaml"))
|
||||
body, err := os.ReadFile(filepath.Join(out, "gallery/base.yaml"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(body)).To(Equal("backend: llama-cpp\n"))
|
||||
})
|
||||
It("fails if a referenced config is missing or escapes the repository", func() {
|
||||
for _, ref := range []string{"missing.yaml", "../../outside.yaml"} {
|
||||
root := GinkgoT().TempDir()
|
||||
Expect(os.Mkdir(filepath.Join(root, "gallery"), 0755)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- name: broken\n url: github:mudler/LocalAI/gallery/"+ref+"@master\n"), 0644)).To(Succeed())
|
||||
Expect(packageGallery(root, "gallery", filepath.Join(root, "out"))).ToNot(Succeed())
|
||||
}
|
||||
})
|
||||
It("rejects symlink escapes when reading configs or writing the bundle", func() {
|
||||
for _, location := range []string{"source", "output"} {
|
||||
root, out, outside := GinkgoT().TempDir(), GinkgoT().TempDir(), GinkgoT().TempDir()
|
||||
for _, dir := range []string{filepath.Join(root, "gallery"), filepath.Join(out, "gallery")} {
|
||||
Expect(os.Mkdir(dir, 0700)).To(Succeed())
|
||||
}
|
||||
index := []byte("- name: test\n url: github:mudler/LocalAI/gallery/base.yaml@master\n")
|
||||
Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), index, 0600)).To(Succeed())
|
||||
outsideFile := filepath.Join(outside, "base.yaml")
|
||||
Expect(os.WriteFile(outsideFile, []byte("outside"), 0600)).To(Succeed())
|
||||
link := filepath.Join(root, "gallery/base.yaml")
|
||||
if location == "output" {
|
||||
Expect(os.WriteFile(link, []byte("inside"), 0600)).To(Succeed())
|
||||
link = filepath.Join(out, "gallery/base.yaml")
|
||||
}
|
||||
Expect(os.Symlink(outsideFile, link)).To(Succeed())
|
||||
Expect(packageGallery(root, "gallery", out)).ToNot(Succeed(), location)
|
||||
data, err := os.ReadFile(outsideFile)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(data)).To(Equal("outside"))
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -7313,6 +7313,9 @@ const docTemplate = `{
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"metadata": {
|
||||
"type": "object"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
@@ -7807,6 +7810,10 @@ const docTemplate = `{
|
||||
},
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -7310,6 +7310,9 @@
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"metadata": {
|
||||
"type": "object"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
@@ -7804,6 +7807,10 @@
|
||||
},
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -2266,6 +2266,8 @@ definitions:
|
||||
type: array
|
||||
id:
|
||||
type: string
|
||||
metadata:
|
||||
type: object
|
||||
model:
|
||||
type: string
|
||||
object:
|
||||
@@ -2652,6 +2654,11 @@ definitions:
|
||||
type: string
|
||||
id:
|
||||
type: string
|
||||
size_vram:
|
||||
description: |-
|
||||
SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
|
||||
the backend process tree has no complete supported reading.
|
||||
type: integer
|
||||
type: object
|
||||
schema.SystemInformationResponse:
|
||||
properties:
|
||||
|
||||
@@ -3,10 +3,10 @@
|
||||
# The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh,
|
||||
# which runs weekly from .github/workflows/refresh-site-counters.yml. Editing
|
||||
# them by hand works but will be overwritten on the next run.
|
||||
stars: 48949
|
||||
forks: 4430
|
||||
contributors: 237
|
||||
releases: 135
|
||||
stars: 49204
|
||||
forks: 4459
|
||||
contributors: 245
|
||||
releases: 136
|
||||
|
||||
# The GitHub API cannot answer for this one, so it is maintained by hand and
|
||||
# the refresh script carries it through untouched.
|
||||
|
||||
Reference in new issue
Block a user