diff --git a/.github/workflows/backend.yml b/.github/workflows/backend.yml index 13e67c6fe..d4bb0bf59 100644 --- a/.github/workflows/backend.yml +++ b/.github/workflows/backend.yml @@ -355,7 +355,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/.github/workflows/backend_build.yml b/.github/workflows/backend_build.yml index 05d50cf82..3e3af89f0 100644 --- a/.github/workflows/backend_build.yml +++ b/.github/workflows/backend_build.yml @@ -252,7 +252,8 @@ jobs: name: digests${{ inputs.tag-suffix }}--${{ inputs.platform-tag || 'single' }} path: /tmp/digests/* if-no-files-found: error - retention-days: 1 + # Release matrices and their retries can outlive a one-day artifact. + retention-days: 7 - name: Build (PR) uses: docker/build-push-action@v7 diff --git a/.github/workflows/backend_build_darwin.yml b/.github/workflows/backend_build_darwin.yml index 6b8b2a89b..19952a5ef 100644 --- a/.github/workflows/backend_build_darwin.yml +++ b/.github/workflows/backend_build_darwin.yml @@ -22,7 +22,8 @@ on: type: string go-version: description: 'Go version to use' - default: '1.24.x' + # Go 1.27 stamps pure-Go hosts with SDK metadata that supports modern Metal APIs. + default: '1.27.x' type: string tag-suffix: description: 'Tag suffix for the built image' diff --git a/.github/workflows/backend_pr.yml b/.github/workflows/backend_pr.yml index c13c444c4..2626f87e8 100644 --- a/.github/workflows/backend_pr.yml +++ b/.github/workflows/backend_pr.yml @@ -281,7 +281,7 @@ jobs: with: backend: ${{ matrix.backend }} build-type: ${{ matrix.build-type }} - go-version: "1.25.x" + go-version: "1.27.x" tag-suffix: ${{ matrix.tag-suffix }} lang: ${{ matrix.lang || 'python' }} use-pip: ${{ matrix.backend == 'diffusers' }} diff --git a/.github/workflows/gallery_publish.yml b/.github/workflows/gallery_publish.yml new file mode 100644 index 000000000..c27b3d282 --- /dev/null +++ b/.github/workflows/gallery_publish.yml @@ -0,0 +1,78 @@ +name: Publish official OCI galleries + +on: + push: + branches: [master] + paths: + - 'gallery/**' + - 'backend/index.yaml' + - 'scripts/build/gallery/**' + - '.github/workflows/gallery_publish.yml' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: publish-official-galleries + cancel-in-progress: false + +jobs: + publish: + if: github.repository == 'mudler/LocalAI' && github.ref == 'refs/heads/master' + runs-on: ubuntu-latest + permissions: + contents: read + id-token: write + env: + COSIGN_EXPERIMENTAL: '1' + GALLERY_REPOSITORY: quay.io/go-skynet/local-ai-backends + strategy: + matrix: + include: + - source: gallery + tag: gallery-models + - source: backend + tag: gallery-backends + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-go@v6 + with: + go-version-file: go.mod + - name: Test and package gallery + env: + GALLERY_SOURCE: ${{ matrix.source }} + run: | + go test ./scripts/build/gallery -count=1 + go run ./scripts/build/gallery . "$GALLERY_SOURCE" "$RUNNER_TEMP/gallery" + - uses: oras-project/setup-oras@v1 + with: + version: '1.3.0' + - uses: sigstore/cosign-installer@v3 + with: + cosign-release: 'v2.6.5' + - name: Login to Quay.io + uses: docker/login-action@v4 + with: + registry: quay.io + username: ${{ secrets.LOCALAI_REGISTRY_USERNAME }} + password: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }} + - name: Publish and sign gallery + shell: bash + env: + GALLERY_TAG: ${{ matrix.tag }} + run: | + set -euo pipefail + cd "$RUNNER_TEMP/gallery" + files=() + while IFS= read -r -d '' file; do + files+=("${file#./}:application/yaml") + done < <(find . -type f -print0 | sort -z) + # Publish an immutable revision, then expose latest only after signing. + ref="$GALLERY_REPOSITORY:$GALLERY_TAG-$GITHUB_SHA" + oras push --artifact-type application/vnd.localai.gallery.v1 \ + --format json "$ref" "${files[@]}" > "$RUNNER_TEMP/push.json" + digest=$(jq -er '.digest' "$RUNNER_TEMP/push.json") + cosign sign --yes --new-bundle-format \ + --registry-referrers-mode=oci-1-1 "$GALLERY_REPOSITORY@$digest" + oras tag "$GALLERY_REPOSITORY@$digest" "$GALLERY_TAG" diff --git a/.github/workflows/gh-pages.yml b/.github/workflows/gh-pages.yml index 746ed923d..7de870401 100644 --- a/.github/workflows/gh-pages.yml +++ b/.github/workflows/gh-pages.yml @@ -40,7 +40,7 @@ jobs: # fetch their own toolchains, and no step uses sudo, apt, make or unzip. runs-on: ${{ github.repository == 'mudler/LocalAI' && 'arc-runner-set' || 'ubuntu-latest' }} env: - HUGO_VERSION: "0.146.3" + HUGO_VERSION: "0.166.0" steps: - name: Checkout uses: actions/checkout@v7 diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index 85da233bd..e8d27eb52 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=e79205f3e0083d04e812e1a4a376f71be97e9a22 +AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index 0f2a84e7d..d6bfd7490 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=1aaf7105be6e55a97fa4a9fd6f5bd362b08436dc +IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662 LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index b672f2d31..8bf0ff5c0 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=84e76d8a23162eca70490da131945ebec1f09bf4 +LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/turboquant/Makefile b/backend/cpp/turboquant/Makefile index e3482d8db..7a8024ebf 100644 --- a/backend/cpp/turboquant/Makefile +++ b/backend/cpp/turboquant/Makefile @@ -1,7 +1,7 @@ # Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant. # Auto-bumped nightly by .github/workflows/bump_deps.yaml. -TURBOQUANT_VERSION?=4deec5587b2963af00bdf80884f3337e02eb7d64 +TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant CMAKE_ARGS?= diff --git a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch b/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch deleted file mode 100644 index 7dfb385c3..000000000 --- a/backend/cpp/turboquant/patches/0002-remove-d512-turbo-shared-mem-overflow.patch +++ /dev/null @@ -1,52 +0,0 @@ -diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh -index 680fd12..ffd6604 100644 ---- a/ggml/src/ggml-cuda/fattn-vec.cuh -+++ b/ggml/src/ggml-cuda/fattn-vec.cuh -@@ -980,6 +980,3 @@ extern DECL_FATTN_VEC_CASE(256, GGML_TYPE_TURBO2_0, GGML_TYPE_TURBO4_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0); - extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --extern DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); -diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu -index 5c614a9..d765cfc 100644 ---- a/ggml/src/ggml-cuda/fattn.cu -+++ b/ggml/src/ggml-cuda/fattn.cu -@@ -507,9 +507,6 @@ static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_t - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_F16) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_BF16) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0) -- FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0) - - #ifdef GGML_CUDA_FA_ALL_QUANTS - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16) -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -index a93be56..3630d87 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo2_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO2_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -index 3c806c2..c8a4d9f 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo3_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO3_0); -diff --git a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -index 180902f..1646ef0 100644 ---- a/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -+++ b/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-turbo4_0.cu -@@ -5,4 +5,3 @@ - DECL_FATTN_VEC_CASE( 64, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(128, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); - DECL_FATTN_VEC_CASE(256, GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); --DECL_FATTN_VEC_CASE_D512(GGML_TYPE_Q8_0, GGML_TYPE_TURBO4_0); diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index 65a4a784d..f2155ffb9 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=6b78932d09765406ba0e0154d95bc6289246ceee +CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index 8fc14bcb8..e288f6fcc 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=e75de9b6b9b688fd293aa22f7e27aa724ea286f8 +PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go diff --git a/backend/go/stablediffusion-ggml/Makefile b/backend/go/stablediffusion-ggml/Makefile index 15ffb569e..6c5e70484 100644 --- a/backend/go/stablediffusion-ggml/Makefile +++ b/backend/go/stablediffusion-ggml/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # stablediffusion.cpp (ggml) STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp -STABLEDIFFUSION_GGML_VERSION?=b167b942f77ecb17e7f78e163a8c32ff7ac95c10 +STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551 CMAKE_ARGS+=-DGGML_MAX_NAME=128 diff --git a/backend/go/stablediffusion-ggml/cpp/gosd.cpp b/backend/go/stablediffusion-ggml/cpp/gosd.cpp index 4a1911015..74b2a0387 100644 --- a/backend/go/stablediffusion-ggml/cpp/gosd.cpp +++ b/backend/go/stablediffusion-ggml/cpp/gosd.cpp @@ -710,14 +710,14 @@ void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled) { params->enabled = enabled; } -void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y) { - params->tile_size_x = tile_size_x; - params->tile_size_y = tile_size_y; +void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h) { + params->tile_size_w = tile_size_w; + params->tile_size_h = tile_size_h; } -void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y) { - params->rel_size_x = rel_size_x; - params->rel_size_y = rel_size_y; +void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h) { + params->rel_size_w = rel_size_w; + params->rel_size_h = rel_size_h; } void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap) { diff --git a/backend/go/stablediffusion-ggml/cpp/gosd.h b/backend/go/stablediffusion-ggml/cpp/gosd.h index 31ce72ab7..c6613d4ed 100644 --- a/backend/go/stablediffusion-ggml/cpp/gosd.h +++ b/backend/go/stablediffusion-ggml/cpp/gosd.h @@ -6,8 +6,8 @@ extern "C" { #endif void sd_tiling_params_set_enabled(sd_tiling_params_t *params, bool enabled); -void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_x, int tile_size_y); -void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_x, float rel_size_y); +void sd_tiling_params_set_tile_sizes(sd_tiling_params_t *params, int tile_size_w, int tile_size_h); +void sd_tiling_params_set_rel_sizes(sd_tiling_params_t *params, float rel_size_w, float rel_size_h); void sd_tiling_params_set_target_overlap(sd_tiling_params_t *params, float target_overlap); sd_tiling_params_t* sd_img_gen_params_get_vae_tiling_params(sd_img_gen_params_t *params); diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index d19d103b1..56a32dbaf 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=e28ec46c6fe2d35f2b234270915421a49c72bbcb +VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun diff --git a/backend/python/transformers/requirements-cpu.txt b/backend/python/transformers/requirements-cpu.txt index 3e3206912..8c5a025e2 100644 --- a/backend/python/transformers/requirements-cpu.txt +++ b/backend/python/transformers/requirements-cpu.txt @@ -2,9 +2,9 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas12.txt b/backend/python/transformers/requirements-cublas12.txt index 40bf331d4..388a2334c 100644 --- a/backend/python/transformers/requirements-cublas12.txt +++ b/backend/python/transformers/requirements-cublas12.txt @@ -2,9 +2,9 @@ torch==2.7.1 accelerate llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-cublas13.txt b/backend/python/transformers/requirements-cublas13.txt index f394f98b1..aa49676b1 100644 --- a/backend/python/transformers/requirements-cublas13.txt +++ b/backend/python/transformers/requirements-cublas13.txt @@ -2,9 +2,9 @@ torch==2.9.0 llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-hipblas.txt b/backend/python/transformers/requirements-hipblas.txt index e4b1bba11..84b62042b 100644 --- a/backend/python/transformers/requirements-hipblas.txt +++ b/backend/python/transformers/requirements-hipblas.txt @@ -1,11 +1,11 @@ --extra-index-url https://download.pytorch.org/whl/rocm7.0 torch==2.10.0+rocm7.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 llvmlite==0.49.0 numba==0.67.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-intel.txt b/backend/python/transformers/requirements-intel.txt index 54ee6ce67..b0b565bf1 100644 --- a/backend/python/transformers/requirements-intel.txt +++ b/backend/python/transformers/requirements-intel.txt @@ -3,9 +3,9 @@ torch optimum[openvino] llvmlite==0.49.0 numba==0.67.0 -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 \ No newline at end of file diff --git a/backend/python/transformers/requirements-mps.txt b/backend/python/transformers/requirements-mps.txt index ea8ba5ab0..9659c943f 100644 --- a/backend/python/transformers/requirements-mps.txt +++ b/backend/python/transformers/requirements-mps.txt @@ -2,9 +2,9 @@ torch==2.7.1 llvmlite==0.49.0 numba==0.67.0 accelerate -transformers>=5.15.1 +transformers>=5.17.0 bitsandbytes -sentence-transformers==5.7.0 +sentence-transformers==6.1.0 diffusers soundfile protobuf==7.36.1 diff --git a/backend/python/transformers/requirements.txt b/backend/python/transformers/requirements.txt index d85aca02a..af5027f10 100644 --- a/backend/python/transformers/requirements.txt +++ b/backend/python/transformers/requirements.txt @@ -1,6 +1,6 @@ -grpcio==1.83.0 +grpcio==1.84.0 protobuf==7.36.1 certifi setuptools scipy==1.18.0 -numpy>=2.5.2 \ No newline at end of file +numpy>=2.5.3 \ No newline at end of file diff --git a/backend/rust/kokoros/src/service.rs b/backend/rust/kokoros/src/service.rs index aeddbf107..a97cc4bf5 100644 --- a/backend/rust/kokoros/src/service.rs +++ b/backend/rust/kokoros/src/service.rs @@ -132,6 +132,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Kokoros TTS model loaded".into(), + ..Default::default() })) } @@ -180,11 +181,13 @@ impl Backend for KokorosService { return Ok(Response::new(backend::Result { success: false, message: format!("Failed to write WAV: {}", e), + ..Default::default() })); } Ok(Response::new(backend::Result { success: true, message: String::new(), + ..Default::default() })) } Err(e) => { @@ -192,6 +195,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: false, message: format!("TTS error: {}", e), + ..Default::default() })) } } @@ -292,6 +296,7 @@ impl Backend for KokorosService { Ok(Response::new(backend::Result { success: true, message: "Model freed".into(), + ..Default::default() })) } @@ -348,6 +353,13 @@ impl Backend for KokorosService { Err(Status::unimplemented("Not supported")) } + async fn animate3_d( + &self, + _: Request, + ) -> Result, Status> { + Err(Status::unimplemented("Not supported")) + } + async fn audio_transcription( &self, _: Request, diff --git a/core/config/gallery.go b/core/config/gallery.go index e22cbc94f..3f6b31ab9 100644 --- a/core/config/gallery.go +++ b/core/config/gallery.go @@ -47,10 +47,13 @@ type Gallery struct { // fallback for availability, not a load-balancing pool: the primary is // always preferred, and a mirror is only consulted after the one before // it fails. Any URI the gallery loader understands works here - // (https://, github:, file://). + // (https://, github:, file://, oci://). Mirrors []string `json:"mirrors,omitempty" yaml:"mirrors,omitempty"` Name string `json:"name" yaml:"name"` Verification *GalleryVerification `json:"verification,omitempty" yaml:"verification,omitempty"` + // ArtifactVerification overrides Verification only for the gallery OCI artifact. + // Backend images keep their separate Verification policy. + ArtifactVerification *GalleryVerification `json:"artifact_verification,omitempty" yaml:"artifact_verification,omitempty"` } // Equal reports whether two gallery entries describe the same gallery. @@ -68,6 +71,13 @@ func (g Gallery) Equal(other Gallery) bool { if !slices.Equal(g.Mirrors, other.Mirrors) { return false } + if g.ArtifactVerification == nil || other.ArtifactVerification == nil { + if g.ArtifactVerification != other.ArtifactVerification { + return false + } + } else if *g.ArtifactVerification != *other.ArtifactVerification { + return false + } if g.Verification == nil || other.Verification == nil { return g.Verification == other.Verification } diff --git a/core/config/gallery_test.go b/core/config/gallery_test.go index 71f4f3a36..7e1848a36 100644 --- a/core/config/gallery_test.go +++ b/core/config/gallery_test.go @@ -179,3 +179,24 @@ var _ = Describe("GalleryVerification", func() { Expect(g[0].Verification.SourceRepository).To(Equal("https://github.com/acme/gallery")) }) }) + +var _ = Describe("Gallery artifact verification", func() { + It("compares artifact policies by value and preserves them in JSON and YAML", func() { + a := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + b := config.Gallery{Name: "gallery", ArtifactVerification: &config.GalleryVerification{Identity: "gallery-workflow"}} + Expect(a.Equal(b)).To(BeTrue()) + b.ArtifactVerification.Identity = "another-workflow" + Expect(a.Equal(b)).To(BeFalse()) + b.ArtifactVerification = nil + Expect(a.Equal(b)).To(BeFalse()) + raw, err := json.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + Expect(json.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + raw, err = yaml.Marshal(a) + Expect(err).ToNot(HaveOccurred()) + b = config.Gallery{} + Expect(yaml.Unmarshal(raw, &b)).To(Succeed()) + Expect(a.Equal(b)).To(BeTrue()) + }) +}) diff --git a/core/config/runtime_settings_startup.go b/core/config/runtime_settings_startup.go index 9808c877b..e5d7e2a46 100644 --- a/core/config/runtime_settings_startup.go +++ b/core/config/runtime_settings_startup.go @@ -17,8 +17,8 @@ import ( // a caching mirror of the files below. The GitHub URI stays as a mirror so an // install still resolves its gallery unchanged whenever the primary is // unreachable - see the fallback chain in core/gallery/gallery_mirrors.go. -const DefaultGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/models", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}]` -const DefaultBackendGalleriesJSON = `[{"name":"localai", "url":"https://index.localai.io/backends", "mirrors":["github:mudler/LocalAI/backend/index.yaml@master"]}]` +const DefaultGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/models","mirrors":["github:mudler/LocalAI/gallery/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-models"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` +const DefaultBackendGalleriesJSON = `[{"name":"localai","url":"https://index.localai.io/backends","mirrors":["github:mudler/LocalAI/backend/index.yaml@master","oci://quay.io/go-skynet/local-ai-backends:gallery-backends"],"artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity":"https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master"}}]` func mustGalleries(jsonList string) []Gallery { var g []Gallery diff --git a/core/config/runtime_settings_startup_test.go b/core/config/runtime_settings_startup_test.go index 410e457d6..d49c53ac4 100644 --- a/core/config/runtime_settings_startup_test.go +++ b/core/config/runtime_settings_startup_test.go @@ -10,22 +10,22 @@ import ( ) var _ = Describe("default galleries", func() { - It("serves the model gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the model gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/models")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/gallery/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-models"})) }) - It("serves the backend gallery from index.localai.io with GitHub as a mirror", func() { + It("serves the backend gallery from index.localai.io with GitHub then OCI as mirrors", func() { var galleries []config.Gallery Expect(json.Unmarshal([]byte(config.DefaultBackendGalleriesJSON), &galleries)).To(Succeed()) Expect(galleries).To(HaveLen(1)) Expect(galleries[0].Name).To(Equal("localai")) Expect(galleries[0].URL).To(Equal("https://index.localai.io/backends")) - Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master"})) + Expect(galleries[0].Mirrors).To(Equal([]string{"github:mudler/LocalAI/backend/index.yaml@master", "oci://quay.io/go-skynet/local-ai-backends:gallery-backends"})) }) // The mirror is the whole reason this default is safe to ship: if @@ -37,6 +37,10 @@ var _ = Describe("default galleries", func() { Expect(json.Unmarshal([]byte(raw), &galleries)).To(Succeed()) for _, g := range galleries { Expect(g.Mirrors).ToNot(BeEmpty(), "default %q has no mirror", g.Name) + Expect(g.ArtifactVerification).ToNot(BeNil()) + Expect(g.ArtifactVerification.Identity).To(Equal("https://github.com/mudler/LocalAI/.github/workflows/gallery_publish.yml@refs/heads/master")) + Expect(g.ArtifactVerification.Issuer).To(Equal("https://token.actions.githubusercontent.com")) + Expect(g.Verification).To(BeNil(), "gallery policy must not change backend image trust") } } }) diff --git a/core/gallery/entry_url.go b/core/gallery/entry_url.go index 184ce2dd9..49f4f720d 100644 --- a/core/gallery/entry_url.go +++ b/core/gallery/entry_url.go @@ -42,7 +42,7 @@ func ociGalleryRoot(g config.Gallery, basePath string) string { if !looksLikeOCIGallery(candidate) { continue } - dir := ociGalleryCacheDir(basePath, candidate, g.Verification) + dir := ociGalleryCacheDir(basePath, candidate, galleryArtifactPolicy(g)) if dir == "" { continue } diff --git a/core/gallery/gallery.go b/core/gallery/gallery.go index bd054b0ff..d0c0f43e0 100644 --- a/core/gallery/gallery.go +++ b/core/gallery/gallery.go @@ -4,7 +4,6 @@ import ( "context" "fmt" "os" - "path/filepath" "slices" "strings" "sync" @@ -19,6 +18,7 @@ import ( "github.com/mudler/LocalAI/pkg/vram" "github.com/mudler/LocalAI/pkg/xsync" "github.com/mudler/xlog" + "golang.org/x/sync/singleflight" "gopkg.in/yaml.v3" ) @@ -276,13 +276,12 @@ func FindGalleryElement[T GalleryElement](models []T, name string) T { func AvailableGalleryModels(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryModel], error) { var models []*GalleryModel + isInstalled := installedConfigs(systemState.Model.ModelsPath) + // Get models from galleries for _, gallery := range galleries { galleryModels, err := getGalleryElements(gallery, systemState.Model.ModelsPath, systemState.RequireBackendIntegrity, func(model *GalleryModel) bool { - if _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", model.GetName()))); err == nil { - return true - } - return false + return isInstalled(model.GetName()) }) if err != nil { return nil, err @@ -351,6 +350,7 @@ var ( // same cache-defeating loop the refresh interval exists to stop. availableModelsLoaded bool refreshing atomic.Bool + coldLoad singleflight.Group galleryGeneration atomic.Uint64 lastRefreshUnixNano atomic.Int64 ) @@ -429,12 +429,15 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste availableModelsMu.RUnlock() if loaded { + // The directory is read before taking the lock. Held across the + // filesystem work, the lock serialized every caller behind it, and a + // page view is dozens of concurrent callers. + isInstalled := installedConfigs(systemState.Model.ModelsPath) // Refresh installed status under write lock to avoid races with // concurrent readers and the background refresh goroutine. availableModelsMu.Lock() for _, m := range cached { - _, err := os.Stat(filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", m.GetName()))) - m.SetInstalled(err == nil) + m.SetInstalled(isInstalled(m.GetName())) } availableModelsMu.Unlock() // Trigger a background refresh if one is not already running. @@ -442,20 +445,29 @@ func AvailableGalleryModelsCached(galleries []config.Gallery, systemState *syste return cached, nil } - // No cache yet — must do a blocking load. - models, err := AvailableGalleryModels(galleries, systemState) + // No cache yet, so the load blocks. Callers arriving while it runs wait + // for it instead of each starting their own: a page view on a fresh + // server is the listing plus one estimate per row at once, and each load + // fetches the gallery index and every config it references. + v, err, _ := coldLoad.Do("gallery", func() (any, error) { + models, err := AvailableGalleryModels(galleries, systemState) + if err != nil { + return nil, err + } + + availableModelsMu.Lock() + availableModelsCache = models + availableModelsLoaded = true + galleryGeneration.Add(1) + availableModelsMu.Unlock() + lastRefreshUnixNano.Store(time.Now().UnixNano()) + + return models, nil + }) if err != nil { return nil, err } - - availableModelsMu.Lock() - availableModelsCache = models - availableModelsLoaded = true - galleryGeneration.Add(1) - availableModelsMu.Unlock() - lastRefreshUnixNano.Store(time.Now().UnixNano()) - - return models, nil + return v.(GalleryElements[*GalleryModel]), nil } // triggerGalleryRefresh starts a background goroutine that refreshes the @@ -634,7 +646,7 @@ var galleryCache = xsync.NewSyncedMap[string, galleryCacheEntry]() // would also point relative entry urls at an unpacked tree the new policy has // not produced yet, so they could not be installed. func galleryIndexCacheKey(g config.Gallery) string { - return g.Name + "-" + galleryCacheName(g.URL, g.Verification) + return g.Name + "-" + galleryCacheName(g.URL, galleryArtifactPolicy(g)) } func getGalleryElements[T GalleryElement](gallery config.Gallery, basePath string, requireIntegrity bool, isInstalledCallback func(T) bool) ([]T, error) { diff --git a/core/gallery/gallery_installed_scan_test.go b/core/gallery/gallery_installed_scan_test.go new file mode 100644 index 000000000..7bcf22a1f --- /dev/null +++ b/core/gallery/gallery_installed_scan_test.go @@ -0,0 +1,138 @@ +package gallery_test + +import ( + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "sync" + "sync/atomic" + "time" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/system" +) + +// The models directory is often network storage (SMB, NFS), where every +// filesystem call is a round trip. The cached listing is read by the gallery +// page and by one VRAM estimate per row, so whatever it costs is paid dozens +// of times per page view. +var _ = Describe("Gallery cache installed status", func() { + const index = ` +- name: plain + backend: llama-cpp +- name: linked + backend: llama-cpp +- name: dangling + backend: llama-cpp +- name: later + backend: llama-cpp +- name: absent + backend: llama-cpp +` + + var ( + modelsDir string + state *system.SystemState + galleries []config.Gallery + hits atomic.Int32 + delay time.Duration + ) + + BeforeEach(func() { + var err error + modelsDir, err = os.MkdirTemp("", "gallery-installed") + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(func() { _ = os.RemoveAll(modelsDir) }) + state, err = system.GetSystemState(system.WithModelPath(modelsDir)) + Expect(err).ToNot(HaveOccurred()) + + hits.Store(0) + delay = 0 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + hits.Add(1) + time.Sleep(delay) + _, _ = w.Write([]byte(index)) + })) + DeferCleanup(server.Close) + galleries = []config.Gallery{{Name: "test", URL: server.URL + "/index.yaml"}} + + gallery.ResetGalleryModelCache() + DeferCleanup(gallery.ResetGalleryModelCache) + }) + + installed := func(models gallery.GalleryElements[*gallery.GalleryModel]) map[string]bool { + out := map[string]bool{} + for _, m := range models { + out[m.Name] = m.Installed + } + return out + } + + It("reports what os.Stat would, for files, symlinks and dangling symlinks", func() { + Expect(os.WriteFile(filepath.Join(modelsDir, "plain.yaml"), []byte("name: plain\n"), 0o644)).To(Succeed()) + target := filepath.Join(modelsDir, "target.txt") + Expect(os.WriteFile(target, []byte("name: linked\n"), 0o644)).To(Succeed()) + Expect(os.Symlink(target, filepath.Join(modelsDir, "linked.yaml"))).To(Succeed()) + Expect(os.Symlink(filepath.Join(modelsDir, "missing"), filepath.Join(modelsDir, "dangling.yaml"))).To(Succeed()) + + // Both the blocking first load and the cached path set the flag, and + // they must agree. + for range 2 { + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(Equal(map[string]bool{ + "plain": true, + "linked": true, + "dangling": false, + "later": false, + "absent": false, + })) + } + }) + + It("picks up a config written after the gallery was cached", func() { + _, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + + Expect(os.WriteFile(filepath.Join(modelsDir, "later.yaml"), []byte("name: later\n"), 0o644)).To(Succeed()) + + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(HaveKeyWithValue("later", true)) + }) + + It("reports nothing installed when the models directory is gone", func() { + _, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(os.RemoveAll(modelsDir)).To(Succeed()) + + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(installed(models)).To(HaveEach(BeFalse())) + }) + + It("shares one upstream load between concurrent callers on a cold cache", func() { + // Slow enough that every caller arrives while the first load is still + // in flight, which is what a page view does to a freshly started + // server: the listing and every row's estimate at once. + delay = 300 * time.Millisecond + + var wg sync.WaitGroup + for range 8 { + wg.Go(func() { + defer GinkgoRecover() + models, err := gallery.AvailableGalleryModelsCached(galleries, state) + Expect(err).ToNot(HaveOccurred()) + Expect(models).To(HaveLen(5)) + }) + } + wg.Wait() + + Expect(hits.Load()).To(Equal(int32(1))) + }) +}) diff --git a/core/gallery/gallery_mirrors.go b/core/gallery/gallery_mirrors.go index 4058c0a45..083512417 100644 --- a/core/gallery/gallery_mirrors.go +++ b/core/gallery/gallery_mirrors.go @@ -144,7 +144,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { if !looksLikeOCIGallery(g.URL) { return nil } - return g.Verification + return galleryArtifactPolicy(g) } // verifiableCandidates drops the candidates that cannot answer for a signed @@ -156,7 +156,7 @@ func indexCachePolicy(g config.Gallery) *config.GalleryVerification { // at, and after a refusal it would turn "this artifact is not trusted" into // "use this other, unchecked copy instead". func verifiableCandidates(g config.Gallery, candidates []string, requireIntegrity bool) []string { - if !looksLikeOCIGallery(g.URL) || (g.Verification == nil && !requireIntegrity) { + if !looksLikeOCIGallery(g.URL) || (galleryArtifactPolicy(g) == nil && !requireIntegrity) { return candidates } out := make([]string, 0, len(candidates)) diff --git a/core/gallery/gallery_oci.go b/core/gallery/gallery_oci.go index 19d6d16c0..cbd649229 100644 --- a/core/gallery/gallery_oci.go +++ b/core/gallery/gallery_oci.go @@ -151,17 +151,18 @@ func readCachedOCIGallery(cacheDir string) ([]byte, bool) { // later fetch served would hand the user a truncated gallery with no sign that // anything went wrong. func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, basePath string, requireIntegrity bool) ([]byte, error) { + policy := galleryArtifactPolicy(g) // Checked before the cache: a copy unpacked while strict integrity was // off was never verified, and turning strict integrity on must not keep // serving it for the rest of its TTL. - if g.Verification == nil && requireIntegrity { + if policy == nil && requireIntegrity { return nil, &galleryVerificationError{ strict: true, - err: fmt.Errorf("no verification policy is set for %q (set verification: in the gallery configuration or disable --require-backend-integrity)", candidate), + err: fmt.Errorf("no verification policy is set for %q (set artifact_verification: in the gallery configuration or disable --require-backend-integrity)", candidate), } } - cacheDir := ociGalleryCacheDir(basePath, candidate, g.Verification) + cacheDir := ociGalleryCacheDir(basePath, candidate, policy) if cacheDir == "" { return nil, fmt.Errorf("gallery %q needs an absolute models directory to cache %q", g.Name, candidate) } @@ -171,7 +172,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base pullRef := downloader.URI(candidate).OCIReference() - if g.Verification != nil { + if policy != nil { // Resolve first, verify the digest, then pull that same digest. // Nothing has been fetched at this point beyond the manifest, so a // policy failure leaves no content anywhere. @@ -179,7 +180,7 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base if err != nil { return nil, err } - if err := verifyGalleryArtifact(ctx, g.Verification, digestRef); err != nil { + if err := verifyGalleryArtifact(ctx, policy, digestRef); err != nil { // Only a decision about the artifact is a refusal. The // verifier also reaches the Sigstore TUF mirror and the // registry, and a timeout or a 5xx there says nothing about @@ -239,3 +240,10 @@ func fetchOCIGalleryIndex(ctx context.Context, g config.Gallery, candidate, base return body, nil } + +func galleryArtifactPolicy(g config.Gallery) *config.GalleryVerification { + if g.ArtifactVerification != nil { + return g.ArtifactVerification + } + return g.Verification +} diff --git a/core/gallery/gallery_oci_test.go b/core/gallery/gallery_oci_test.go index 5672da123..4c038b5a2 100644 --- a/core/gallery/gallery_oci_test.go +++ b/core/gallery/gallery_oci_test.go @@ -229,6 +229,21 @@ var _ = Describe("oci:// galleries", func() { }) }) + It("uses the artifact policy without replacing backend image verification", func() { + srv, _, _ := ociRegistry() + url := pushGalleryArtifact(srv.URL, "galleries/separate-policy", galleryArtifactType, []ociGalleryFile{{title: "index.yaml", body: "- name: demo\n"}}) + backendPolicy := &config.GalleryVerification{Identity: "backend-workflow"} + artifactPolicy := &config.GalleryVerification{Identity: "gallery-workflow"} + var seen *config.GalleryVerification + stubGalleryVerifier(func(_ context.Context, policy *config.GalleryVerification, _ string) error { seen = policy; return nil }) + g := config.Gallery{URL: srv.URL + "/unavailable", Mirrors: []string{srv.URL + "/also-unavailable", url}, Name: "separate", Verification: backendPolicy, ArtifactVerification: artifactPolicy} + _, source, err := fetchGalleryIndex(context.Background(), g, tempModelsDir(), true) + Expect(source).To(Equal(url)) + Expect(err).ToNot(HaveOccurred()) + Expect(seen).To(Equal(artifactPolicy)) + Expect(g.Verification).To(Equal(backendPolicy)) + }) + It("refuses an unsigned gallery in strict integrity mode", func() { srv, _, blobs := ociRegistry() url := pushGalleryArtifact(srv.URL, "galleries/strict", galleryArtifactType, []ociGalleryFile{ diff --git a/core/gallery/installed_configs.go b/core/gallery/installed_configs.go new file mode 100644 index 000000000..fb36e5d29 --- /dev/null +++ b/core/gallery/installed_configs.go @@ -0,0 +1,61 @@ +package gallery + +import ( + "errors" + "io/fs" + "os" + "path/filepath" + "strings" +) + +const modelConfigExt = ".yaml" + +// installedConfigs answers "does /.yaml exist?" for every +// entry of a gallery from a single read of the models directory. +// +// The question used to be asked with one os.Stat per gallery entry. The gallery +// holds thousands of entries and the models directory is often network storage +// (SMB, NFS), where each Stat is a round trip, so one listing cost seconds. The +// listing is read by the gallery page and by one VRAM estimate per row, which +// turned a page view into minutes. +// +// Answers match os.Stat on the same path: a symlink counts only when its target +// exists, and anything else carrying the name counts, directories included. +// Names that are not a plain file name are checked with os.Stat directly, since +// they point outside the listed directory. +func installedConfigs(modelsPath string) func(name string) bool { + statInstalled := func(name string) bool { + _, err := os.Stat(filepath.Join(modelsPath, name+modelConfigExt)) + return err == nil + } + + entries, err := os.ReadDir(modelsPath) + if err != nil { + if errors.Is(err, fs.ErrNotExist) { + return func(string) bool { return false } + } + // A directory that exists but cannot be listed may still answer a + // Stat, so fall back rather than report everything as not installed. + return statInstalled + } + + present := make(map[string]struct{}, len(entries)) + for _, e := range entries { + base, ok := strings.CutSuffix(e.Name(), modelConfigExt) + if !ok { + continue + } + if e.Type()&fs.ModeSymlink != 0 && !statInstalled(base) { + continue + } + present[base] = struct{}{} + } + + return func(name string) bool { + if strings.ContainsRune(name, '/') || strings.ContainsRune(name, filepath.Separator) { + return statInstalled(name) + } + _, ok := present[name] + return ok + } +} diff --git a/core/http/endpoints/localai/system.go b/core/http/endpoints/localai/system.go index 996c9a781..9c4ba7503 100644 --- a/core/http/endpoints/localai/system.go +++ b/core/http/endpoints/localai/system.go @@ -8,6 +8,7 @@ import ( "github.com/mudler/LocalAI/core/schema" "github.com/mudler/LocalAI/core/services/monitoring" "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/xsysinfo" ) // SystemInformations returns the system informations @@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app entry.Process = proc } } + if pid, ok := localPID(m); ok { + if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok { + entry.SizeVRAM = &used + } + } sysmodels = append(sysmodels, entry) } if sampler != nil { diff --git a/core/http/endpoints/localai/system_info_test.go b/core/http/endpoints/localai/system_info_test.go new file mode 100644 index 000000000..83f7daa07 --- /dev/null +++ b/core/http/endpoints/localai/system_info_test.go @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: MIT +package localai_test + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/http/endpoints/localai" + "github.com/mudler/LocalAI/pkg/model" + "github.com/mudler/LocalAI/pkg/system" + process "github.com/mudler/go-processmanager" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SystemInformations memory", func() { + It("keeps model metadata and omits VRAM for remote or stopped backends", func() { + path, err := os.MkdirTemp("", "system-info-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, path) + configFile := filepath.Join(path, "remote.yaml") + Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed()) + cl := config.NewModelConfigLoader(path) + Expect(cl.ReadModelConfig(configFile)).To(Succeed()) + ml := model.NewModelLoader(&system.SystemState{}) + store := model.NewInMemoryModelStore() + store.Set("remote", model.NewModel("remote", "worker:50051", nil)) + store.Set("stopped", model.NewModel("stopped", "", &process.Process{})) + ml.SetModelStore(store) + app := echo.New() + app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil)) + rec := httptest.NewRecorder() + app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil)) + Expect(rec.Code).To(Equal(http.StatusOK)) + var response struct { + Models []map[string]any `json:"loaded_models"` + } + Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed()) + Expect(response.Models).To(ConsistOf( + map[string]any{"id": "remote", "backend": "llama-cpp"}, + map[string]any{"id": "stopped"}, + )) + }) +}) diff --git a/core/http/endpoints/openresponses/responses.go b/core/http/endpoints/openresponses/responses.go index 553c01558..d62fa7534 100644 --- a/core/http/endpoints/openresponses/responses.go +++ b/core/http/endpoints/openresponses/responses.go @@ -1873,49 +1873,37 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 return true } - // Try JSON parsing as fallback - jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true) - if jsonErr == nil && len(jsonResults) > lastEmittedToolCallCount { + // Only completed JSON calls can be emitted as completed SSE items. + jsonResults := parseStreamingJSONToolCalls(cleanedResult) + if len(jsonResults) > lastEmittedToolCallCount { for i := lastEmittedToolCallCount; i < len(jsonResults); i++ { - jsonObj := jsonResults[i] - if name, ok := jsonObj["name"].(string); ok && name != "" { - args := "{}" - if argsVal, ok := jsonObj["arguments"]; ok { - if argsStr, ok := argsVal.(string); ok { - args = argsStr - } else { - argsBytes, _ := json.Marshal(argsVal) - args = string(argsBytes) - } - } + tc := jsonResults[i] + toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) + outputIndex++ - toolCallID := fmt.Sprintf("fc_%s", uuid.New().String()) - outputIndex++ - - functionCallItem := &schema.ORItemField{ - Type: "function_call", - ID: toolCallID, - Status: "completed", - CallID: toolCallID, - Name: name, - Arguments: args, - } - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.added", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ - - sendSSEEvent(c, &schema.ORStreamEvent{ - Type: "response.output_item.done", - SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, - Item: functionCallItem, - }) - sequenceNumber++ + functionCallItem := &schema.ORItemField{ + Type: "function_call", + ID: toolCallID, + Status: "completed", + CallID: toolCallID, + Name: tc.Name, + Arguments: tc.Arguments, } + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.added", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ + + sendSSEEvent(c, &schema.ORStreamEvent{ + Type: "response.output_item.done", + SequenceNumber: sequenceNumber, + OutputIndex: &outputIndex, + Item: functionCallItem, + }) + sequenceNumber++ } lastEmittedToolCallCount = len(jsonResults) c.Response().Flush() @@ -2424,6 +2412,8 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Non-tool-call streaming path + messageOutputIndex := outputIndex + var reasoningOutputIndex int // Emit output_item.added for message currentMessageID = fmt.Sprintf("msg_%s", uuid.New().String()) messageItem := &schema.ORItemField{ @@ -2436,7 +2426,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2448,7 +2438,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &emptyTextPart, }) @@ -2471,10 +2461,11 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 } // Handle reasoning item - if extractor.Reasoning() != "" { + if extractor.Reasoning() != "" || reasoningDelta != "" { // Check if we need to create reasoning item if currentReasoningID == "" { outputIndex++ + reasoningOutputIndex = outputIndex currentReasoningID = fmt.Sprintf("reasoning_%s", uuid.New().String()) reasoningItem := &schema.ORItemField{ Type: "reasoning", @@ -2484,7 +2475,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.added", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2496,7 +2487,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.added", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &emptyPart, }) @@ -2509,7 +2500,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Delta: strPtr(reasoningDelta), Logprobs: emptyLogprobs(), @@ -2526,7 +2517,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.delta", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Delta: strPtr(contentDelta), Logprobs: emptyLogprobs(), @@ -2595,7 +2586,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Text: strPtr(finalReasoning), Logprobs: emptyLogprobs(), @@ -2608,7 +2599,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentReasoningID, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, ContentIndex: ¤tReasoningContentIndex, Part: &reasoningPart, }) @@ -2624,7 +2615,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &reasoningOutputIndex, Item: reasoningItem, }) sequenceNumber++ @@ -2658,7 +2649,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.output_text.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Text: strPtr(result), Logprobs: logprobsPtr(mcpStreamLogprobs), @@ -2671,7 +2662,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 Type: "response.content_part.done", SequenceNumber: sequenceNumber, ItemID: currentMessageID, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, ContentIndex: ¤tContentIndex, Part: &resultPart, }) @@ -2683,7 +2674,7 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 sendSSEEvent(c, &schema.ORStreamEvent{ Type: "response.output_item.done", SequenceNumber: sequenceNumber, - OutputIndex: &outputIndex, + OutputIndex: &messageOutputIndex, Item: messageItem, }) sequenceNumber++ @@ -2723,34 +2714,9 @@ func handleOpenResponsesStream(c echo.Context, responseID string, createdAt int6 // Emit response.completed now := time.Now().Unix() - // Collect final output items (reasoning first, then messages, then tool calls) - var finalOutputItems []schema.ORItemField - // Add reasoning item if it exists - if currentReasoningID != "" && finalReasoning != "" { - finalOutputItems = append(finalOutputItems, schema.ORItemField{ - Type: "reasoning", - ID: currentReasoningID, - Status: "completed", - Content: []schema.ORContentPart{makeOutputTextPart(finalReasoning)}, - }) - } - // Add message item - if len(collectedOutputItems) > 0 { - // Use collected items (may include reasoning already) - for _, item := range collectedOutputItems { - if item.Type == "message" { - finalOutputItems = append(finalOutputItems, item) - } - } - } else { - finalOutputItems = append(finalOutputItems, *messageItem) - } - // Add function_call items from fallback - for _, item := range collectedOutputItems { - if item.Type == "function_call" { - finalOutputItems = append(finalOutputItems, item) - } - } + // The final output array must use the indices announced in the stream. + // The message is opened first, followed by reasoning and fallback calls. + finalOutputItems := append([]schema.ORItemField{*messageItem}, collectedOutputItems...) responseCompleted := buildORResponse(responseID, createdAt, &now, "completed", input, finalOutputItems, &schema.ORUsage{ InputTokens: noToolTokenUsage.Prompt, OutputTokens: noToolTokenUsage.Completion, diff --git a/core/http/endpoints/openresponses/responses_stream_test.go b/core/http/endpoints/openresponses/responses_stream_test.go new file mode 100644 index 000000000..13ddb10e1 --- /dev/null +++ b/core/http/endpoints/openresponses/responses_stream_test.go @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: MIT +package openresponses + +import ( + "context" + "encoding/json" + "net/http/httptest" + "strings" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/schema" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/LocalAI/pkg/model" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Responses stream item consistency", func() { + DescribeTable("preserves every item and its announced output index", func(tokens []string, chatDeltas []*pb.ChatDelta, wantReasoning, wantAnswer string, fallback bool) { + originalInference := backend.ModelInferenceFunc + DeferCleanup(func() { backend.ModelInferenceFunc = originalInference }) + backend.ModelInferenceFunc = func( + ctx context.Context, prompt string, messages schema.Messages, + images, videos, audios []string, loader *model.ModelLoader, + cfg *config.ModelConfig, cl *config.ModelConfigLoader, app *config.ApplicationConfig, + tokenCallback func(string, backend.TokenUsage) bool, tools, toolChoice string, + logprobs, topLogprobs *int, logitBias map[string]float64, metadata map[string]string, + ) (func() (backend.LLMResponse, error), error) { + return func() (backend.LLMResponse, error) { + for i, token := range tokens { + usage := backend.TokenUsage{} + if len(chatDeltas) > 0 { + usage.ChatDeltas = []*pb.ChatDelta{chatDeltas[i]} + } + if !tokenCallback(token, usage) { + break + } + } + return backend.LLMResponse{Response: strings.Join(tokens, ""), ChatDeltas: chatDeltas, Usage: backend.TokenUsage{Prompt: 3, Completion: 8}}, nil + }, nil + } + cfg := &config.ModelConfig{} + cfg.FunctionsConfig.AutomaticToolParsingFallback = fallback + cfg.FunctionsConfig.JSONRegexMatch = []string{`(?s)(.*?)`} + recorder := httptest.NewRecorder() + request := httptest.NewRequest("POST", "/v1/responses", nil) + c := echo.New().NewContext(request, recorder) + input := &schema.OpenResponsesRequest{Model: "test-model", Input: "hello", Stream: true} + err := handleOpenResponsesStream(c, "resp_test", 1, input, cfg, nil, nil, config.NewApplicationConfig(), "hello", &schema.OpenAIRequest{Context: request.Context()}, nil, false, false, nil, nil) + Expect(err).NotTo(HaveOccurred()) + Expect(recorder.Body.String()).To(HaveSuffix("data: [DONE]\n\n")) + + var events []schema.ORStreamEvent + var completed *schema.ORResponseResource + for _, line := range strings.Split(recorder.Body.String(), "\n") { + if !strings.HasPrefix(line, "data: ") || line == "data: [DONE]" { + continue + } + var event schema.ORStreamEvent + Expect(json.Unmarshal([]byte(strings.TrimPrefix(line, "data: ")), &event)).To(Succeed()) + Expect(event.Type).NotTo(Equal("error")) + events = append(events, event) + if event.Type == "response.completed" { + completed = event.Response + } + } + Expect(completed).NotTo(BeNil()) + wantCount := 1 + if wantReasoning != "" { + wantCount++ + } + if fallback { + wantCount++ + } + Expect(completed.Output).To(HaveLen(wantCount), "final output must retain the answer alongside reasoning and fallback calls") + + indices := map[string]int{} + done := map[string]int{} + deltas := map[string]string{} + for i, event := range events { + Expect(event.SequenceNumber).To(Equal(i)) + if event.Type == "response.output_item.added" { + Expect(event.Item).NotTo(BeNil()) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(indices).NotTo(HaveKey(event.Item.ID)) + Expect(*event.OutputIndex).To(Equal(len(indices))) + indices[event.Item.ID] = *event.OutputIndex + } + id := event.ItemID + if event.Item != nil { + id = event.Item.ID + } + if id == "" { + continue + } + Expect(indices).To(HaveKey(id)) + Expect(event.OutputIndex).NotTo(BeNil()) + Expect(*event.OutputIndex).To(Equal(indices[id]), "event %s changes the index for %s", event.Type, id) + Expect(completed.Output[indices[id]].ID).To(Equal(id)) + if event.Type == "response.output_item.done" { + done[id]++ + Expect(event.Item.Status).To(Equal("completed")) + Expect(event.Item.Type).To(Equal(completed.Output[indices[id]].Type)) + if event.Item.Type == "function_call" { + Expect(event.Item.Name).To(Equal(completed.Output[indices[id]].Name)) + Expect(event.Item.Arguments).To(Equal(completed.Output[indices[id]].Arguments)) + } else { + Expect(event.Item.Content).To(Equal(completed.Output[indices[id]].Content)) + } + } + if event.Type == "response.output_text.delta" { + deltas[id] += *event.Delta + } + } + Expect(indices).To(HaveLen(wantCount)) + for _, item := range completed.Output { + Expect(done[item.ID]).To(Equal(1)) + switch item.Type { + case "message", "reasoning": + want := wantAnswer + if item.Type == "reasoning" { + want = wantReasoning + } + parts, ok := item.Content.([]any) + Expect(ok).To(BeTrue()) + Expect(parts).To(HaveLen(1)) + Expect(parts[0].(map[string]any)["text"]).To(Equal(want)) + if !fallback { + Expect(deltas[item.ID]).To(Equal(want)) + } + case "function_call": + Expect(item.Name).To(Equal("get_weather")) + Expect(item.Arguments).To(MatchJSON(`{"city":"Rome"}`)) + Expect(item.CallID).NotTo(BeEmpty()) + default: + Fail("unexpected output item type: " + item.Type) + } + } + }, + Entry("tagged reasoning and answer", []string{"", "Let me think.", "", "The answer is 42."}, nil, "Let me think.", "The answer is 42.", false), + Entry("backend reasoning and answer deltas", []string{"", ""}, []*pb.ChatDelta{{ReasoningContent: "Let me think."}, {Content: "The answer is 42."}}, "Let me think.", "The answer is 42.", false), + Entry("plain text", []string{"Hello", " world."}, nil, "", "Hello world.", false), + Entry("automatic fallback tool call", []string{`{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "", "", true), + Entry("reasoning and automatic fallback tool call", []string{"", "Let me think.", "", `{"name":"get_weather","arguments":{"city":"Rome"}}`}, nil, "Let me think.", "", true), + ) +}) diff --git a/core/http/endpoints/openresponses/stream_tool_calls.go b/core/http/endpoints/openresponses/stream_tool_calls.go new file mode 100644 index 000000000..b8f0f185d --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls.go @@ -0,0 +1,35 @@ +package openresponses + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/pkg/functions" +) + +func parseStreamingJSONToolCalls(text string) []functions.FuncCallResults { + // Partial parsing heals unfinished arguments. The caller emits terminal + // events and never revisits emitted calls, so only accept complete JSON. + // Keep completed objects returned before an unfinished trailing object. + objects, _ := functions.ParseJSONIterative(text, false) + var calls []functions.FuncCallResults + for _, object := range objects { + name, ok := object["name"].(string) + if !ok || name == "" { + continue + } + arguments := "{}" + if value, ok := object["arguments"]; ok { + if s, ok := value.(string); ok { + arguments = s + } else { + data, err := json.Marshal(value) + if err != nil { + continue + } + arguments = string(data) + } + } + calls = append(calls, functions.FuncCallResults{Name: name, Arguments: arguments}) + } + return calls +} diff --git a/core/http/endpoints/openresponses/stream_tool_calls_test.go b/core/http/endpoints/openresponses/stream_tool_calls_test.go new file mode 100644 index 000000000..1fa6bca2a --- /dev/null +++ b/core/http/endpoints/openresponses/stream_tool_calls_test.go @@ -0,0 +1,44 @@ +package openresponses + +import ( + "github.com/mudler/LocalAI/pkg/functions" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Streaming JSON tool calls", func() { + It("waits for the arguments before completing a split call", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash",`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls`)).To(BeEmpty()) + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("does not complete a call at any intermediate token boundary", func() { + text := `{"name":"Bash","arguments":{"command":"printf \"hello\"","options":[1,2]}}` + for end := 1; end < len(text); end++ { + Expect(parseStreamingJSONToolCalls(text[:end])).To(BeEmpty(), "prefix: %s", text[:end]) + } + Expect(parseStreamingJSONToolCalls(text)).To(HaveLen(1)) + }) + + It("keeps completed calls while the next call is incomplete", func() { + Expect(parseStreamingJSONToolCalls(`{"name":"Bash","arguments":{"command":"ls -la"}} {"name":"Read",`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + })) + }) + + It("preserves string arguments and calls that take no arguments", func() { + Expect(parseStreamingJSONToolCalls(`[{"name":"Bash","arguments":"{\"command\":\"ls -la\"}"},{"name":"status"}]`)).To(Equal([]functions.FuncCallResults{ + {Name: "Bash", Arguments: `{"command":"ls -la"}`}, + {Name: "status", Arguments: `{}`}, + })) + }) + + It("does not count unrelated JSON objects as emitted calls", func() { + Expect(parseStreamingJSONToolCalls(`{"message":"checking"} {"name":"status","arguments":{}}`)).To(Equal([]functions.FuncCallResults{ + {Name: "status", Arguments: `{}`}, + })) + }) +}) diff --git a/core/schema/localai.go b/core/schema/localai.go index 7e5d5e314..dc99a1dbe 100644 --- a/core/schema/localai.go +++ b/core/schema/localai.go @@ -208,6 +208,9 @@ type SysInfoModel struct { // when the model has no local process (a distributed worker holds it) or // the process could not be read. Process *SysInfoProcess `json:"process,omitempty"` + // SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + // the backend process tree has no complete supported reading. + SizeVRAM *uint64 `json:"size_vram,omitempty"` } // SysInfoProcess is a point-in-time reading of one backend process. diff --git a/core/schema/openai.go b/core/schema/openai.go index 2aa69969b..6f3717256 100644 --- a/core/schema/openai.go +++ b/core/schema/openai.go @@ -99,7 +99,7 @@ type OpenAIResponse struct { // OpenAI-SDK consumers that filter on a truthy `result.usage` // (continuedev/continue, Kilo Code, Roo Code, etc.). Usage *OpenAIUsage `json:"usage,omitempty"` - Metadata json.RawMessage `json:"metadata,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty" swaggertype:"object"` } // StreamOptions mirrors OpenAI's `stream_options` request field. The only diff --git a/core/schema/system_info_test.go b/core/schema/system_info_test.go new file mode 100644 index 000000000..79a1bdba1 --- /dev/null +++ b/core/schema/system_info_test.go @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: MIT +package schema_test + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/core/schema" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SysInfoModel memory", func() { + It("omits unavailable VRAM while preserving a measured zero", func() { + entry := schema.SysInfoModel{ID: "model"} + encoded, err := json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`)) + + zero := uint64(0) + entry.SizeVRAM = &zero + encoded, err = json.Marshal(entry) + Expect(err).NotTo(HaveOccurred()) + Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`)) + }) +}) diff --git a/docker-compose.yaml b/docker-compose.yaml index ee137e83c..82b3c18b6 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -59,6 +59,7 @@ services: # capabilities: [gpu, utility] # # For legacy NVIDIA driver (for older NVIDIA Container Toolkit): + # Request compute for CUDA libraries (libcuda.so.1) and utility for NVML. # environment: # NVIDIA_DRIVER_CAPABILITIES: "compute,utility" # init: true @@ -68,7 +69,7 @@ services: # devices: # - driver: nvidia # count: 1 - # capabilities: [gpu, utility] + # capabilities: [gpu, compute, utility] ## Uncomment for PostgreSQL-backed knowledge base (see Agents docs) # postgres: diff --git a/docs/content/advanced/advanced-usage.md b/docs/content/advanced/advanced-usage.md index f7d9546cf..8580aa499 100644 --- a/docs/content/advanced/advanced-usage.md +++ b/docs/content/advanced/advanced-usage.md @@ -38,7 +38,7 @@ For a complete reference of all available configuration options, see the [Model local-ai run github://mudler/LocalAI/examples/configurations/phi-2.yaml@master ``` -See also [chatbot-ui](https://github.com/mudler/LocalAI-examples/tree/main/chatbot-ui) as an example on how to use config files. +See also the [configuration examples](https://github.com/mudler/LocalAI-examples/tree/main/configurations) in the LocalAI-examples repository for more config files. ### Prompt templates diff --git a/docs/content/features/backends.md b/docs/content/features/backends.md index 6085cf536..c0818df85 100644 --- a/docs/content/features/backends.md +++ b/docs/content/features/backends.md @@ -82,6 +82,8 @@ tags: ### Verifying OCI Backends +The default backend gallery tries `https://index.localai.io/backends`, then `github:mudler/LocalAI/backend/index.yaml@master`, then `oci://quay.io/go-skynet/local-ai-backends:gallery-backends`. The OCI fallback is signed by `gallery_publish.yml`. Its `artifact_verification` policy applies only to the gallery artifact; `verification` continues to control backend image signatures. Existing custom gallery lists are not changed. See [gallery publishing]({{% relref "features/model-gallery#official-gallery-publishing" %}}) for details. + Backend galleries can require keyless Sigstore signatures for every OCI image they provide. Add a `verification` policy to the gallery configuration, then enable strict integrity mode: diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md index f8a06539a..86ad8b14e 100644 --- a/docs/content/features/distributed-mode.md +++ b/docs/content/features/distributed-mode.md @@ -417,8 +417,12 @@ usage is reported back to the frontend: NVML library (and therefore `nvidia-smi`) is not available inside the container. CUDA compute still works, but the worker cannot query free VRAM and the Nodes page will show the node as fully used. Set - `NVIDIA_DRIVER_CAPABILITIES=compute,utility` (or, with the NVIDIA CDI - runtime, list `capabilities: [gpu, utility]` on the device reservation). + `NVIDIA_DRIVER_CAPABILITIES=compute,utility` when using the NVIDIA runtime. + For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Docker derives driver capabilities from this reservation, so include `compute` + for CUDA libraries such as `libcuda.so.1`. The `utility` capability alone + enables monitoring but does not provide CUDA libraries. - **Run the container with `init: true` (or `docker run --init`).** The worker process becomes PID 1 in the container and cannot reap zombies on diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index 7baa18087..0bee9c890 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,99 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Cyber-Tiel-Coder + +Install `cyber-tiel-coder-35b-a3b-q4-mtp` for coding and image chat with llama.cpp. +The gallery groups UD-Q4_K_XL and UD-Q8_K_XL builds; both enable MTP speculative decoding and include a BF16 vision projector. +To select Q8 explicitly, run `local-ai models install cyber-tiel-coder-35b-a3b-q4-mtp --variant cyber-tiel-coder-35b-a3b-q8-mtp`. +Both configurations use the embedded chat template and default to 32,768 context tokens. +The [model card](https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP) describes its abliterated Ornith-1.5 base and MIT license. + +## Qwen3.8-27B Agention Precision + +The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of +Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image +input and use a 32,768-token context by default. + +Install with automatic variant selection: + +```bash +local-ai models install qwen3.8-27b-agention-iq4-xs +``` + +To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs` +or `--variant qwen3.8-27b-agention-q4-k-m` to the same command. +The files use standard llama.cpp quantization types and the Apache-2.0 license. +See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF) +for quantization details. These entries do not enable MTP speculative decoding. + +## Swift 1.5 Qwen3.8-27B GSQ-RCO + +Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp. +The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model. +To select IQ3_S explicitly, run: + +```bash +local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s +``` + +The configurations use the embedded chat template and default to 32,768 context tokens. +These builds support text chat only: the publisher has no verified vision projector for this release. +They use standard GGUF files without MTP decoding. +See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms. + +## Sharp-Spark-X2.5-4B + +Install `sharp-spark-x2.5-4b` for coding and text chat with llama.cpp. +The gallery groups Q4_K_XL, Q5_K_XL, and Q6_K_XL builds as variants. +To select the publisher's recommended Q6 build, run: + +```bash +local-ai models install sharp-spark-x2.5-4b --variant sharp-spark-x2.5-4b-q6 +``` + +All builds use a 32,768-token default context and the embedded Sharp-Spark chat template. +That template adds a terseness instruction to the system prompt. +See the [publisher's model card](https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF) for quantization and template details. + +## MiMo-V2.6-Distill-Qwen-9B + +Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. +This MIT-licensed 9B Qwen3.5 fine-tune targets coding, agent tasks, and visual coding. +The gallery groups Q4_K_M and Q8_0 builds as variants; both include the F16 vision projector. +To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9b --variant mimo-v2.6-distill-qwen-9b-q8`. +The configurations default to 32,768 context tokens and use the model's embedded chat template. +See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. + +## Qwopus3.8 Flash V2 + +Install `qwopus3.8-27b-flash-v2` for the Q4_K_M GGUF build, with Q8_0 available through variant selection: + +```bash +local-ai models install qwopus3.8-27b-flash-v2 +local-ai models install qwopus3.8-27b-flash-v2 --variant qwopus3.8-27b-flash-v2-q8 +``` + +Both builds use llama.cpp with the embedded chat template, MTP speculative decoding, and the F32 vision projector. +Weights and projector downloads are pinned to a Hugging Face revision and verified with SHA256. +This Apache-2.0 release is a further post-training of Qwopus3.8 Flash for reasoning and agent tasks. +See the [publisher's model card](https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF) for evaluation details and limitations. + +## ThinkingCap Qwen3.8-27B + +Install `thinkingcap-qwen3.8-27b` for a 27B reasoning model with text and image input. +The llama.cpp entries include Q4_K_M and Q8_0 weights, each paired with the F16 vision projector. +LocalAI selects between the builds using the gallery variant rules. To request Q8_0 explicitly: + +```bash +local-ai models install thinkingcap-qwen3.8-27b --variant thinkingcap-qwen3.8-27b-q8 +``` + +Both builds use the embedded chat template, a 32,768-token default context, and the publisher's sampled decoding settings. +MTP speculative decoding is not enabled by these entries. +The weights use [PolyForm Small Business 1.0.0 with a personal-use grant](https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/blob/main/LICENSE). +Review that license for permitted use. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. @@ -88,7 +181,7 @@ To use a gallery that needs authentication, such as a private GitHub repository A gallery entry can declare a `mirrors` list of alternative locations for the same index file. Mirrors exist for availability, not for load balancing: LocalAI always prefers the `url`, and only falls back to the mirrors, in the order you listed them, when the one before it cannot be fetched. If the primary works, the mirrors are never contacted. -Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), and `file://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. +Mirrors accept any URI the gallery loader understands — `https://`, `github:`, `huggingface://` (also `hf://` and `hf.co/`), `file://`, and `oci://` — and the same rules apply to them as to a primary URL, so a `file://` mirror must still live inside your models directory. ```json GALLERIES=[{"name":"localai", "url":"https://example.org/gallery/index.yaml", "mirrors":["github:mudler/LocalAI/gallery/index.yaml@master"]}] @@ -142,10 +235,10 @@ A relative `url` cannot leave the gallery root. An entry that tries to climb out ### Signature verification -An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add a `verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: +An `oci://` gallery can be signed, and LocalAI verifies the signature before it unpacks anything. Add an `artifact_verification` block with the Fulcio issuer and the signing identity, in the same form the [backend galleries]({{%relref "features/backends#verifying-oci-backends" %}}) use: ```json -GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] +GALLERIES=[{"name":"premium","url":"oci://quay.io/acme/gallery:latest","artifact_verification":{"issuer":"https://token.actions.githubusercontent.com","identity_regex":"^https://github\\.com/acme/gallery/\\.github/workflows/publish\\.yml@refs/tags/.+$"}}] ``` The tag is resolved to a digest, the signature is checked against that digest, and the same digest is then pulled. A gallery that fails verification is never written to the cache, so no unverified file reaches your disk. The optional `not_before` RFC3339 value revokes signatures logged before that time, exactly as it does for backends. @@ -164,9 +257,17 @@ With strict integrity on (`--require-backend-integrity` or `LOCALAI_REQUIRE_BACK The optional `source_repository` value works the same for `oci://` galleries as it does for backends: it pins the repository the signature was made for when a shared reusable workflow does the signing. See [Verifying OCI Backends]({{%relref "features/backends#verifying-oci-backends" %}}). {{% notice warning %}} -With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery that has no `verification` block is refused when the models are listed, not only when one is installed. Add a `verification` block to every `oci://` gallery before you turn strict integrity on, or the galleries without one stop listing. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. +`artifact_verification` applies only to the gallery artifact. Backend image signatures use `verification`. For compatibility, the artifact loader uses `verification` when `artifact_verification` is absent. Set both fields when the gallery and its backend images have different signing identities. + +With `--require-backend-integrity` (`LOCALAI_REQUIRE_BACKEND_INTEGRITY=1`), an `oci://` gallery with neither policy is refused when the models are listed, not only when one is installed. An `oci://` gallery without a policy still lists outside strict mode, with a warning in the log. {{% /notice %}} +### Official gallery publishing + +The `gallery_publish.yml` workflow publishes both official galleries on relevant changes to `master`, or through a manual dispatch on `master`. It uses the existing `LOCALAI_REGISTRY_USERNAME` and `LOCALAI_REGISTRY_PASSWORD` secrets. It reuses the public backend repository `go-skynet/local-ai-backends`. The `gallery-models` and `gallery-backends` tags move only after their artifact digest has been signed. Revision tags include the source commit SHA. + +To prepare the same files locally, run `go run ./scripts/build/gallery . gallery /tmp/model-gallery` or use `backend` as the source directory. The helper rewrites repository-local base configuration URLs to artifact-relative paths and copies the files. The published artifact type is `application/vnd.localai.gallery.v1`; each file is a separate layer with its relative path as its title. + ### Private registries A gallery in a private registry needs a credentials entry that matches the registry, the same entry an image pull from it would use: @@ -198,10 +299,10 @@ GALLERIES=[{"name":"", "url":"}} {{% tab title="Apple" %}} +To build pure-Go backend hosts that load Metal libraries, use Go 1.27 or later on macOS 13 or later. +Go 1.27 records macOS SDK 26.2 in internally linked executables, which enables modern Metal APIs in these hosts. +Rebuild the affected backend after upgrading Go. Rebuilding only `local-ai` does not update installed backend executables. + Install `xcode` from the App Store ```bash diff --git a/docs/content/operations/cloud-proxy.md b/docs/content/operations/cloud-proxy.md index 42d522bfc..312fa2327 100644 --- a/docs/content/operations/cloud-proxy.md +++ b/docs/content/operations/cloud-proxy.md @@ -63,9 +63,9 @@ against - and two modes: `proxy.provider` selects the auth scheme and (in translate mode) the wire format. Supported values: `openai`, `anthropic`. -API keys are loaded from either an environment variable (`api_key_env`) or a -file (`api_key_file`). The key never appears in the config file or the admin -UI; pick whichever fits your secret-management setup. +If the upstream requires an API key, configure either an environment variable +(`api_key_env`) or a file (`api_key_file`). The key never appears in the config +file or the admin UI. If the upstream requires no API key, omit both fields. ### OpenAI passthrough @@ -129,7 +129,7 @@ Anthropic clients hit `http://localhost:8080/v1/messages` with Most third-party providers (Together, Groq, DeepInfra, OpenRouter, …) speak the OpenAI chat-completions wire format. Use `provider: openai` with the -provider's URL and API key: +provider's URL and, if required, its API key: ```yaml name: llama-3-70b-via-together @@ -143,6 +143,37 @@ proxy: upstream_model: meta-llama/Llama-3-70b-chat-hf ``` +### Upstreams without an API key + +For an OpenAI-compatible upstream that accepts requests without authentication, +omit both `api_key_env` and `api_key_file`: + +```yaml +name: internal-chat-proxy +backend: cloud-proxy + +proxy: + mode: passthrough + provider: openai + upstream_url: http://inference.internal:8000/v1/chat/completions + upstream_model: my-model +``` + +Replace the example URL and model name with your upstream's values. LocalAI +loads this configuration without resolving a key and adds no upstream +`Authorization` header. This also applies to OpenAI-compatible upstreams in +translate mode. + +Omitting both fields differs from setting `api_key_env` to an empty or unset +environment variable: the latter causes a backend load error. + +LocalAI's client authentication is separate. Clients must still authenticate +to LocalAI when its authentication is enabled. LocalAI does not forward their +`Authorization` header to the upstream. + +An upstream without API keys can still require another authentication or +payment protocol. Omitting these fields does not implement that protocol. + ### Translate mode In translate mode the cloud-proxy backend converts LocalAI's internal proto diff --git a/docs/content/reference/nvidia-l4t.md b/docs/content/reference/nvidia-l4t.md index 2adac3a84..e3b54020a 100644 --- a/docs/content/reference/nvidia-l4t.md +++ b/docs/content/reference/nvidia-l4t.md @@ -88,8 +88,10 @@ page in the frontend shows the node as fully used, check two things: NVML work inside the container. With `--gpus all` alone (or `--runtime nvidia` without extra flags) only `compute` is wired in on some driver versions. Add `-e NVIDIA_DRIVER_CAPABILITIES=compute,utility` - to your `docker run`, or `capabilities: [gpu, utility]` in compose / - Kubernetes device reservations. + to your `docker run`. For Docker Compose with `driver: nvidia`, use + `capabilities: [gpu, compute, utility]` on the device reservation. + Include `compute` for CUDA libraries such as `libcuda.so.1`; `utility` + alone only provides monitoring libraries and tools. 2. Pass `--init` to `docker run` (or `init: true` in compose) so the container has a proper PID 1 reaper - otherwise short-lived child processes like `nvidia-smi` can intermittently fail with diff --git a/docs/content/reference/system-info.md b/docs/content/reference/system-info.md index b825e4e06..ed4de00ff 100644 --- a/docs/content/reference/system-info.md +++ b/docs/content/reference/system-info.md @@ -28,6 +28,29 @@ Returns available backends and currently loaded models. | `loaded_models[].process.memory_percent` | `number` | `rss_bytes` as a percentage of host RAM | | `loaded_models[].process.cpu_percent` | `number` | Share of the whole host's CPU used since the previous call, 0-100. Omitted on the first call that sees the process, because there is no earlier reading to compare against | | `loaded_models[].process.started_at` | `string` | When the process started (RFC 3339) | +| `loaded_models[].size_vram` | `integer` | Optional DRM-accounted resident device memory, in bytes | + +### Per-model VRAM + +On Linux, `size_vram` reports resident device memory for the local backend +process and its child processes. LocalAI reads `drm-resident-local*` and +`drm-resident-vram*` from `/proc` and counts each DRM client once per GPU. +Host-memory regions are excluded. The reading includes buffers attributed +to the backend, without separating weights, KV cache, and other allocations. +See the [kernel DRM accounting specification](https://docs.kernel.org/gpu/drm-usage-stats.html) +for these counters. + +The field is omitted when accounting is unavailable or incomplete. This +includes external and distributed backends, macOS, proprietary NVIDIA +drivers, primary DRM nodes (`/dev/dri/card*`), missing resident counters, +and unreadable process information. +A present value of `0` means the supported counters report zero bytes. +Treat an absent field as unknown. + +This is a snapshot of driver accounting, not a memory reservation. Shared +buffers can appear in different clients' counters, and allocations can change +during collection. Do not treat the sum across models as exclusive physical +GPU usage. These readings do not replace capacity checks when scheduling work. ### Usage @@ -49,6 +72,7 @@ curl http://localhost:8080/system { "id": "my-llama-model", "backend": "llama-cpp", + "size_vram": 5368709120, "process": { "pid": 48213, "rss_bytes": 5368709120, diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..f6fa9361c 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,96 @@ --- +- name: "ternary-bonsai-2-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + - https://github.com/PrismML-Eng/llama.cpp + description: | + Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary + transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits + per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a + Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's + llama.cpp fork) instead of stock llama.cpp. + license: "apache-2.0" + tags: + - llm + - gguf + - reasoning + - vision + - multimodal + icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg + overrides: + backend: bonsai + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf + - filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf +- name: "swift-qwen3.8-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF + description: | + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B. + The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss. + This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding. + The weights use the Swift Open License v1.0. + license: "swift-open-license-1.0" + tags: + - llm + - gguf + - reasoning + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + min_p: 0 + model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + presence_penalty: 1.5 + repeat_penalty: 1 + temperature: 0.7 + top_k: 20 + top_p: 0.8 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf + - filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: @@ -205,7 +297,101 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: 237123aeeea5ac31d3327650e4fadd7125c8e1b32717fe110117dcfb0903f2b7 + sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf +- name: "qwopus3.8-27b-flash-v2" + variants: + - model: qwopus3.8-27b-flash-v2-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q4_K_M GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M.gguf + sha256: 227bedb8ebf4a05e342c99f1f852be19cf0ed394f6cc5901823c07a735ea983e + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 +- name: "qwopus3.8-27b-flash-v2-q8" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash + - https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF + description: | + Qwopus3.8-27B-Flash-V2 is a new post-training release for reasoning and agent + workloads. This Q8_0 GGUF includes the F32 vision projector and uses + llama.cpp's embedded chat template with MTP speculative decoding. + license: "apache-2.0" + tags: + - llm + - gguf + - qwen + - qwen3 + - vision + - multimodal + - instruction-tuned + - reasoning + - mtp + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + model: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Qwopus3.8-27B-Flash-V2-MTP-Q8_0/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/Qwopus3.8-27B-Flash-V2-MTP-Q8_0.gguf + sha256: bc291a2ab2ac209d2cd97f0e0d25bfb98381d4cb4ee4f8baa4cd3c662db95f78 + - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-V2-MTP-Q4_K_M/mmproj-F32.gguf + uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-V2-GGUF/resolve/ecb87867b0977dfd1554d2fc54105a802b34345a/mmproj-F32.gguf + sha256: c9d201ea8a2a474ce55cfab6d1e1480d4b2e1574dda976db15aee267072ca4d6 - name: "qwopus3.8-27b-flash" variants: - model: qwopus3.8-27b-flash-q8 @@ -302,6 +488,188 @@ - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-MTP-Q4_K_M/mmproj-F32.gguf uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-GGUF/resolve/e146d61e88782677805b3b68ad3adf8674dde80d/mmproj-F32.gguf sha256: 52e6818e4d18eea010c50e5245eaa10a8cc3dcc30efea4ff60cbad8abf5669e1 +- name: mimo-v2.6-distill-qwen-9b + variants: + - model: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q4_K_M GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + sha256: 4bca6f18c73f72270c7a20c2ea2bea581de8246e318714277120369d34048c81 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q8_0 GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + sha256: 2fad0aa11bb9e7aa491ff12f768954f9dd0a6e7d4ce4a897ca73ec420f3b90ae + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: thinkingcap-qwen3.8-27b + variants: + - model: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q4_K_M GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + sha256: fafa890ce2ce8531b4ade225c7dbd5f5d72a92303ca9ef72890c6cf78f19f299 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q4_K_M.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf +- name: thinkingcap-qwen3.8-27b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B + - https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF + description: | + ThinkingCap is a 27B Qwen3.8 fine-tune trained to reduce reasoning tokens, with text and image input. + This Q8_0 GGUF build uses llama.cpp, the embedded chat template, and the F16 vision projector. + Licensed under PolyForm Small Business 1.0.0 with the publisher's personal-use grant; see the model license for permitted use. + license: polyform-small-business-1.0.0 + tags: + - llm + - gguf + - cpu + - gpu + - vision + - multimodal + - reasoning + last_checked: "2026-09-27" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + temperature: 1.0 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + files: + - filename: ThinkingCap-Qwen3.8-27B-Q8_0.gguf + sha256: 41070725606f4be781db804e8458f3346c699d0dac24f2b96d2a734556c6c0f7 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/ThinkingCap-Qwen3.8-27B-Q8_0.gguf + - filename: mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf + sha256: 98fa9aad59b42449786a16bbce96bcd92204d03cac0aee0cdccca711c2adefd1 + uri: https://huggingface.co/bottlecapai/ThinkingCap-Qwen3.8-27B-GGUF/resolve/108ff8f24ce8e9335fbf308844cd3c59c13380a4/mmproj-ThinkingCap-Qwen3.8-27B-f16.gguf - name: hemmingway-1 variants: - model: hemmingway-1-q8 @@ -4681,6 +5049,110 @@ - filename: llama-cpp/mmproj/thomson-1.0-small/mmproj-bf16.gguf uri: huggingface://bartowski/thomsonreuters_Thomson-1.0-Small-GGUF/mmproj-thomsonreuters_Thomson-1.0-Small-bf16.gguf sha256: 11634fcccd59c23f1b95e34e5cf479dec86290eeb3dda980324aabd8b0b48f41 +- name: cyber-tiel-coder-35b-a3b-q4-mtp + variants: + - model: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q4_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q4_K_XL.gguf + sha256: 0bbcf3cc9be4c976bad20e641baf629dad9c178d39ebdc9cd72129179943c06a + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 +- name: cyber-tiel-coder-35b-a3b-q8-mtp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + license: mit + urls: + - https://huggingface.co/huihui-ai/Huihui-Ornith-1.5-35B-A3B-abliterated + - https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP + description: | + Cyber-Tiel-Coder is a 35B mixture-of-experts coding model with 3B active parameters, + based on Huihui's abliterated Ornith-1.5. This UD-Q8_K_XL build includes + MTP speculative decoding, the embedded Sharp chat template, and a BF16 vision projector. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - moe + - coding + - tools + - vision + - multimodal + - mtp + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + parameters: + model: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + min_p: 0.0 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/cyber-tiel-coder-35b-a3b/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/Cyber-Tiel-Coder-35B-A3B-MTP-UD-Q8_K_XL.gguf + sha256: 601052bb18c97b40808a5d93992b25eeb64b9b0bc5e2de0681c15681adf19961 + - filename: llama-cpp/mmproj/cyber-tiel-coder-35b-a3b/mmproj-BF16.gguf + uri: https://huggingface.co/peculiar-ragdoll/Cyber-Tiel-Coder-35B-A3B-GGUF-MTP/resolve/fa19d4f33561dc0d107c2a2f8943f1ca2e288109/mmproj-BF16.gguf + sha256: d9ce31026d1cb1f3f8d5152e2e2a014d9d2b302b6c93a7dc07bb0a0487f52837 - &tiel-coder-35b-a3b name: "tiel-coder-35b-a3b-q4" variants: @@ -5161,6 +5633,104 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545 +- name: qwen3.8-27b-agention-iq4-xs + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: qwen3.8-27b-agention-q4-k-m + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf + sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67 + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 +- name: qwen3.8-27b-agention-q4-k-m + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf + sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 - &qwen3-8-27b name: "qwen3.8-27b-q4" variants: @@ -5447,6 +6017,178 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2 +- name: "swift-1.5-qwen3.8-27b-gsq-rco" + variants: + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf - !!merge <<: *qwen3-8-27b name: "qwen3.8-27b-gsq-rco-iq2-xs" variants: [] @@ -5639,6 +6381,120 @@ - filename: llama-cpp/models/spark-x2.5-1.7b/Spark-X2.5-1.7B-Q8_0.gguf uri: huggingface://XHToken/Spark-X2.5-1.7B-GGUF/Spark-X2.5-1.7B-Q8_0.gguf sha256: cd77c03185a834bb1162a4b7713520be5838058bfc54873645beff470bb24442 +- name: sharp-spark-x2.5-4b + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: sharp-spark-x2.5-4b-q5 + - model: sharp-spark-x2.5-4b-q6 + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q4_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf + sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837 + +- name: sharp-spark-x2.5-4b-q5 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q5_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf + sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26 + +- name: sharp-spark-x2.5-4b-q6 + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/XHToken/Spark-X2.5-4B + - https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF + description: | + Sharp-Spark is an imatrix quantization of XHToken's Spark-X2.5-4B text model + with an adjusted chat template for coding. This Q6_K_XL build uses the + embedded Sharp-Spark template and a 32K-token default context. + license: apache-2.0 + tags: + - llm + - gguf + - cpu + - gpu + - coding + - reasoning + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + options: + - use_jinja:true + parameters: + model: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + temperature: 0.6 + top_p: 0.95 + top_k: 20 + template: + use_tokenizer_template: true + files: + - filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf + sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc + - &spark-x2-5-4b name: "spark-x2.5-4b-q4" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" @@ -45126,6 +45982,12 @@ sha256: "" uri: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/clip_vision/clip_vision_h.safetensors - name: kimodo-soma-rp + variants: + - model: kimodo-soma-rp-bf16 + - model: kimodo-soma-rp-q4_k + - model: kimodo-soma-rp-q4_k_m + - model: kimodo-soma-rp-q5_k + - model: kimodo-soma-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -45313,6 +46175,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-soma-seed + variants: + - model: kimodo-soma-seed-bf16 + - model: kimodo-soma-seed-q4_k + - model: kimodo-soma-seed-q4_k_m + - model: kimodo-soma-seed-q5_k + - model: kimodo-soma-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -45500,6 +46368,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-rp + variants: + - model: kimodo-g1-rp-bf16 + - model: kimodo-g1-rp-q4_k + - model: kimodo-g1-rp-q4_k_m + - model: kimodo-g1-rp-q5_k + - model: kimodo-g1-rp-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: @@ -45687,6 +46561,12 @@ uri: https://huggingface.co/LocalAI-io/Llama-3-Kimodo-GGML/resolve/3e8d958803beaddb6011ac534f2be972e2710c7d/Llama-3-Kimodo-BF16.gguf sha256: d9a60017b3981bac874c4d118fc7e34f05b41763a12f0c0c7ee1e3b84eebb20f - name: kimodo-g1-seed + variants: + - model: kimodo-g1-seed-bf16 + - model: kimodo-g1-seed-q4_k + - model: kimodo-g1-seed-q4_k_m + - model: kimodo-g1-seed-q5_k + - model: kimodo-g1-seed-q6_k url: github:mudler/LocalAI/gallery/kimodocpp.yaml@master backend: kimodocpp urls: diff --git a/go.mod b/go.mod index 1d4025b89..c81fbd82e 100644 --- a/go.mod +++ b/go.mod @@ -259,7 +259,7 @@ require ( github.com/kevinburke/ssh_config v1.2.0 // indirect github.com/labstack/gommon v0.4.2 // indirect github.com/mschoch/smat v0.2.0 // indirect - github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 + github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e github.com/mudler/localrecall v0.6.5 // indirect github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 github.com/olekukonko/tablewriter v0.0.5 // indirect @@ -535,7 +535,7 @@ require ( golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f // indirect golang.org/x/mod v0.36.0 // indirect golang.org/x/sync v0.20.0 - golang.org/x/sys v0.45.0 // indirect + golang.org/x/sys v0.45.0 golang.org/x/term v0.43.0 golang.org/x/text v0.37.0 golang.org/x/tools v0.45.0 // indirect diff --git a/go.sum b/go.sum index 891a60d05..863118db1 100644 --- a/go.sum +++ b/go.sum @@ -1032,6 +1032,8 @@ github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336 h1:iKBkSnpisOvMVxF github.com/mudler/LocalAGI v0.0.0-20260606071251-14aed1ae4336/go.mod h1:U+g6u8mF2wQxhkdBl3dr8G4db1cv3n7KTKmraoJ7D0c= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1 h1:AqQJhjUIMFvpJ+8ShSpzEp8ClaW5vNqJKq+/9bKTNpc= github.com/mudler/LocalAGI v0.0.0-20260911225740-d93d478e42f1/go.mod h1:Z97IpFdxmKaigCCpIzfo2Jz6wLwwbnaQrcBTLxyrF+o= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e h1:ZaKo7Pp44STT196mJS0OUYSnN2TU62KQmSXKvQ8HS0Q= +github.com/mudler/LocalAGI v0.0.0-20260912140006-8253de99163e/go.mod h1:Wo2UItZdZZd2PkMvhDT19a9MPyiwC+8gnk2nLZniVcY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= diff --git a/pkg/modelartifacts/materializer.go b/pkg/modelartifacts/materializer.go index a3e2a9e8e..87036f756 100644 --- a/pkg/modelartifacts/materializer.go +++ b/pkg/modelartifacts/materializer.go @@ -414,6 +414,9 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedFiles := 0 skippedBytes := int64(0) tasks := make([]downloader.FileTask, 0, len(snapshot.Files)) + // Sibling manifests are read once, before the staging loop, so the + // per-file reuse lookups below never re-read or re-parse them. + siblings := loadSiblingCandidates(modelsPath, spec, layout) for index, file := range snapshot.Files { if err := ctx.Err(); err != nil { return Result{}, err @@ -437,6 +440,21 @@ func (m *Manager) materializeLocked(ctx context.Context, modelsPath string, spec skippedBytes += file.Size continue } + // Before reaching for the network, consult committed sibling trees for the + // same Source (type+endpoint+repo+revision). A narrower allow_patterns + // request gets a different CacheKey, so committedResult misses even though a + // broader sibling already holds this exact file; reusing it avoids a + // redundant re-download of tens of gigabytes. The match is re-verified + // through verifyDownloadedFile (full SHA-256), never size-only, and a broader + // request can never inherit a narrower sibling's gaps because each file is + // matched individually against the sibling's manifest. + if entry, ok := reuseFromCommittedSibling(siblings, file, layout, root); ok { + manifest.Files[taskIndex] = entry + completedBytes.Add(file.Size) + skippedFiles++ + skippedBytes += file.Size + continue + } nameSum := sha256.Sum256([]byte(file.Path)) blobRel := path.Join(".downloads", hex.EncodeToString(nameSum[:])) blobAbs := filepath.Join(layout.Partial, filepath.FromSlash(blobRel)) @@ -588,6 +606,129 @@ func reuseMaterializedFile(fileName string, source hfapi.SnapshotFile) (Manifest return entry, true } +// siblingCandidate is one committed sibling artifact tree that shares this +// request's Source (type+endpoint+repo+revision), with its manifest files +// indexed by path. +type siblingCandidate struct { + final string + filesByPath map[string][]ManifestFile +} + +// loadSiblingCandidates reads the committed sibling manifest set once, before +// the staging loop. Doing it per file instead would re-read and re-parse every +// sibling manifest for every file — 20 committed siblings and a 300-file +// snapshot means 6000 manifest reads before the first byte is fetched. +// +// The current artifact's own committed tree is excluded: it is either absent +// (the reason materializeLocked is running) or already handled by +// committedResult's exact-key fast path. +func loadSiblingCandidates(modelsPath string, spec Spec, layout Layout) []siblingCandidate { + if spec.Resolved == nil || layout.Final == "" { + return nil + } + siblingsRoot := filepath.Join(modelsPath, ".artifacts", "huggingface") + entries, err := os.ReadDir(siblingsRoot) + if err != nil { + return nil + } + var candidates []siblingCandidate + for _, entry := range entries { + if !entry.IsDir() { + continue + } + siblingFinal := filepath.Join(siblingsRoot, entry.Name()) + if siblingFinal == layout.Final { + continue + } + siblingManifest, err := ReadManifest(filepath.Join(siblingFinal, "manifest.json")) + if err != nil { + continue + } + siblingArtifact := siblingManifest.Artifact + if siblingArtifact.Resolved == nil || + siblingArtifact.Source.Type != spec.Source.Type || + siblingArtifact.Resolved.Endpoint != spec.Resolved.Endpoint || + siblingArtifact.Source.Repo != spec.Source.Repo || + siblingArtifact.Resolved.Revision != spec.Resolved.Revision { + continue + } + byPath := make(map[string][]ManifestFile, len(siblingManifest.Files)) + for _, f := range siblingManifest.Files { + byPath[f.Path] = append(byPath[f.Path], f) + } + candidates = append(candidates, siblingCandidate{final: siblingFinal, filesByPath: byPath}) + } + return candidates +} + +// reuseFromCommittedSibling looks for a file already committed under a sibling +// artifact tree — same Source (type+endpoint+repo+revision), different +// allow/ignore patterns — and stages it for this writer instead of fetching. +// A narrower allow_patterns request gets a different CacheKey (path.go:62), so +// committedResult misses and materializeLocked would otherwise re-download +// files an already-committed broader sibling already holds. +// +// The match is never size-only: the sibling file is re-hashed through the +// shared verifyDownloadedFile against the current request's SnapshotFile (its +// LFS or git blob OID), so the staged entry is byte-for-byte identical to a +// fresh download. A broader request can never stand in for files a narrower +// sibling lacks, because each requested file is matched individually against +// the sibling's manifest file set. Hard-link keeps the shared models volume +// disk-neutral; a byte copy is the fallback only for EXDEV, the one case the +// kernel cannot hard-link. +func reuseFromCommittedSibling(candidates []siblingCandidate, file hfapi.SnapshotFile, layout Layout, root *os.Root) (ManifestFile, bool) { + snapshotRel := path.Join("snapshot", file.Path) + snapshotAbs := filepath.Join(layout.Partial, filepath.FromSlash(snapshotRel)) + for _, sibling := range candidates { + for _, siblingFile := range sibling.filesByPath[file.Path] { + if siblingFile.Size != file.Size { + continue + } + siblingPath := filepath.Join(sibling.final, "snapshot", filepath.FromSlash(file.Path)) + verified, err := verifyDownloadedFile(siblingPath, file) + if err != nil { + continue + } + if err := root.MkdirAll(path.Dir(snapshotRel), 0o750); err != nil { + return ManifestFile{}, false + } + _ = root.Remove(snapshotRel) + if err := linkOrCopy(siblingPath, snapshotAbs); err != nil { + return ManifestFile{}, false + } + return verified, true + } + } + return ManifestFile{}, false +} + +// linkOrCopy hard-links src to dst, falling back to a byte-for-byte copy only +// when the kernel refuses a hard link across filesystems (EXDEV). Hard-linking +// keeps the shared models volume neutral — a narrowed request does not double +// the storage of a broad sibling's files. +func linkOrCopy(src, dst string) error { + if err := os.Link(src, dst); err == nil { + return nil + } else if !errors.Is(err, syscall.EXDEV) { + return err + } + in, err := os.Open(src) + if err != nil { + return err + } + defer func() { _ = in.Close() }() + out, err := os.Create(dst) + if err != nil { + return err + } + if _, err := io.Copy(out, in); err != nil { + _ = out.Close() + _ = os.Remove(dst) + return err + } + return out.Close() +} + func verifyDownloadedFile(fileName string, source hfapi.SnapshotFile) (ManifestFile, error) { file, err := os.Open(fileName) if err != nil { diff --git a/pkg/modelartifacts/materializer_sibling_reuse_test.go b/pkg/modelartifacts/materializer_sibling_reuse_test.go new file mode 100644 index 000000000..572b255d8 --- /dev/null +++ b/pkg/modelartifacts/materializer_sibling_reuse_test.go @@ -0,0 +1,230 @@ +package modelartifacts_test + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strconv" + "strings" + "sync" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + hfapi "github.com/mudler/LocalAI/pkg/huggingface-api" + "github.com/mudler/LocalAI/pkg/modelartifacts" +) + +const siblingReuseRevision = "0123456789abcdef0123456789abcdef01234567" + +// recordingResolver serves a fixed full file set filtered by each request's +// allow/ignore patterns, so a narrower request genuinely resolves to a strict +// subset of a broader sibling's files. The HTTP server behind it records every +// fetch, which is the signal the sibling-reuse fix is verified through. The +// function under fix is never mocked: a real Manager drives the real staging + +// commit path against this stub collaborator. +type recordingResolver struct { + endpoint string + repo string + files []hfapi.SnapshotFile + server *httptest.Server + + mu sync.Mutex + fetched map[string]int +} + +func newRecordingResolver(files []hfapi.SnapshotFile, contents map[string][]byte) *recordingResolver { + r := &recordingResolver{ + endpoint: "https://huggingface.co", + repo: "owner/repo", + files: files, + fetched: map[string]int{}, + } + r.server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, req *http.Request) { + name := strings.TrimPrefix(req.URL.Path, "/file/") + body, ok := contents[name] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + r.mu.Lock() + r.fetched[name]++ + r.mu.Unlock() + w.Header().Set("Content-Length", strconv.Itoa(len(body))) + _, _ = w.Write(body) + })) + return r +} + +func (r *recordingResolver) ResolveSnapshot(_ context.Context, req hfapi.SnapshotRequest) (hfapi.Snapshot, error) { + files, err := hfapi.FilterSnapshotFiles(r.files, req.AllowPatterns, req.IgnorePatterns) + if err != nil { + return hfapi.Snapshot{}, err + } + out := make([]hfapi.SnapshotFile, len(files)) + for i, f := range files { + f.URL = r.server.URL + "/file/" + f.Path + out[i] = f + } + return hfapi.Snapshot{ + Endpoint: r.endpoint, Repo: r.repo, + RequestedRevision: req.Revision, ResolvedRevision: siblingReuseRevision, Files: out, + }, nil +} + +func (r *recordingResolver) fetchCount(path string) int { + r.mu.Lock() + defer r.mu.Unlock() + return r.fetched[path] +} + +func (r *recordingResolver) resetFetches() { + r.mu.Lock() + defer r.mu.Unlock() + r.fetched = map[string]int{} +} + +func siblingReuseFiles(contents map[string][]byte) []hfapi.SnapshotFile { + paths := []string{"a/first.bin", "b/second.bin", "c/third.bin"} + files := make([]hfapi.SnapshotFile, 0, len(paths)) + for _, p := range paths { + sum := sha256.Sum256(contents[p]) + files = append(files, hfapi.SnapshotFile{ + Path: p, Size: int64(len(contents[p])), LFSOID: hex.EncodeToString(sum[:]), + }) + } + return files +} + +// The narrow-request case proves the fix for #11047: +// a request with narrower allow_patterns (a strict subset) reuses files an +// already-committed broader sibling holds, hard-linking instead of re-fetching. +// +// On master this is RED: a narrower allow_patterns set hashes to a different +// CacheKey (path.go:62), so committedResult misses and materializeLocked +// re-fetches the file (fetches > 0) into a separate copy (no os.SameFile). On +// the branch it is GREEN: reuseFromCommittedSibling hits the broad sibling, +// verifies the file via verifyDownloadedFile, and hard-links it (fetches == 0, +// os.SameFile true). +var _ = Describe("committed sibling reuse", func() { + It("reuses files from a broader committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit the broad sibling: all three files, fetched from the resolver. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(broad.CacheHit).To(BeFalse()) + Expect(resolver.fetchCount("a/first.bin")).To(BeNumerically(">", 0), + "the broad sibling must have fetched a/first.bin to commit it") + + resolver.resetFetches() + + // Narrowed request: a strict subset of the broad sibling's file set. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + Expect(narrow.CacheHit).To(BeFalse()) + + // (b) The sibling-present file must NOT be re-fetched: zero fetches. This is + // the assertion that is RED on master (one fetch) and GREEN on the branch. + Expect(resolver.fetchCount("a/first.bin")).To(Equal(0), + "a/first.bin must be reused from the committed broad sibling, not re-fetched") + + // (a) The narrowed tree's staged file is the same inode as the broad + // sibling's file (hard-link), not a freshly downloaded second copy. RED on + // master (separate file), GREEN on the branch (hard-link). + broadFile := filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), "a", "first.bin") + narrowFile := filepath.Join(modelsPath, filepath.FromSlash(narrow.RelativePath), "a", "first.bin") + broadInfo, err := os.Stat(broadFile) + Expect(err).NotTo(HaveOccurred()) + narrowInfo, err := os.Stat(narrowFile) + Expect(err).NotTo(HaveOccurred()) + Expect(os.SameFile(broadInfo, narrowInfo)).To(BeTrue(), + "the narrowed request must hard-link the broad sibling's file rather than store a second copy") + + // The reused bytes are intact end to end. + Expect(os.ReadFile(narrowFile)).To(Equal(contents["a/first.bin"])) + }) + + // The broader-request case is the manifest file-set guard: a broader request + // against a narrower committed sibling must still fetch the files the sibling + // lacks and commit a complete tree. Sibling-reuse can never serve an incomplete + // model as complete, because each requested file is matched individually against + // the sibling's manifest. + It("fetches files missing from a narrower committed sibling", func() { + contents := map[string][]byte{ + "a/first.bin": []byte("first-file-bytes"), + "b/second.bin": []byte("second-file-bytes-longer"), + "c/third.bin": []byte("third-file"), + } + resolver := newRecordingResolver(siblingReuseFiles(contents), contents) + defer resolver.server.Close() + + modelsPath := GinkgoT().TempDir() + manager := modelartifacts.NewManager(resolver, + modelartifacts.WithLocker(func(string) modelartifacts.Locker { return bypassedLocker{} })) + + // Commit a NARROW sibling first: only a/first.bin and b/second.bin. + narrowSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + AllowPatterns: []string{"a/first.bin", "b/second.bin"}, + }} + narrow, err := manager.Ensure(context.Background(), modelsPath, narrowSpec) + Expect(err).NotTo(HaveOccurred()) + narrowPaths := make([]string, 0, len(narrow.Manifest.Files)) + for _, f := range narrow.Manifest.Files { + narrowPaths = append(narrowPaths, f.Path) + } + Expect(narrowPaths).To(Equal([]string{"a/first.bin", "b/second.bin"})) + + resolver.resetFetches() + + // A BROADER request asks for all three files, including c/third.bin which the + // narrow sibling does not hold. + broadSpec := modelartifacts.Spec{Source: modelartifacts.Source{ + Type: modelartifacts.SourceTypeHuggingFace, Repo: "owner/repo", + }} + broad, err := manager.Ensure(context.Background(), modelsPath, broadSpec) + Expect(err).NotTo(HaveOccurred()) + + // The file the narrow sibling lacks MUST be fetched: sibling-reuse must not + // inherit a narrower tree's gaps as if the broad request were complete. + Expect(resolver.fetchCount("c/third.bin")).To(BeNumerically(">", 0), + "c/third.bin is absent from the narrow sibling and must be fetched, not served as complete") + + // The broad tree's manifest file set is exactly the full set — never the + // narrow sibling's subset. This file-set comparison proves no incomplete model + // is ever served as complete via sibling-reuse. + broadPaths := make([]string, 0, len(broad.Manifest.Files)) + for _, f := range broad.Manifest.Files { + broadPaths = append(broadPaths, f.Path) + } + Expect(broadPaths).To(Equal([]string{"a/first.bin", "b/second.bin", "c/third.bin"})) + + // Every file is present on disk with the right bytes after commit. + for _, p := range []string{"a/first.bin", "b/second.bin", "c/third.bin"} { + Expect(os.ReadFile(filepath.Join(modelsPath, filepath.FromSlash(broad.RelativePath), filepath.FromSlash(p)))). + To(Equal(contents[p])) + } + }) +}) diff --git a/pkg/xsysinfo/process_vram_linux.go b/pkg/xsysinfo/process_vram_linux.go new file mode 100644 index 000000000..3dd0a59a9 --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux.go @@ -0,0 +1,165 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "bufio" + "bytes" + "math" + "os" + "path/filepath" + "strconv" + "strings" +) + +// ProcessVRAM reports device-local resident bytes accounted to a process tree +// by DRM. Unsupported or incomplete accounting returns false, not a measured zero. +func ProcessVRAM(pid int) (uint64, bool) { + return processVRAM("/proc", pid) +} + +func processVRAM(procRoot string, pid int) (uint64, bool) { + if pid <= 0 { + return 0, false + } + clients := map[string]uint64{} + seen := map[int]bool{} + pending := []int{pid} + for len(pending) > 0 { + current := pending[len(pending)-1] + pending = pending[:len(pending)-1] + if seen[current] { + continue + } + seen[current] = true + base := filepath.Join(procRoot, strconv.Itoa(current)) + fds, err := os.ReadDir(filepath.Join(base, "fd")) + if err != nil { + return 0, false + } + for _, fd := range fds { + target, err := os.Readlink(filepath.Join(base, "fd", fd.Name())) + if err != nil { + return 0, false + } + // A mixed DRM/NVIDIA tree cannot provide a complete DRM reading. + if strings.HasPrefix(target, "/dev/nvidia") { + return 0, false + } + if !strings.HasPrefix(target, "/dev/dri/render") { + // Primary nodes can also own allocations. Until their device + // identity is resolved, omitting them would undercount the tree. + if strings.HasPrefix(target, "/dev/dri/") { + return 0, false + } + continue + } + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and fd.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "fdinfo", fd.Name())) + if err != nil { + return 0, false + } + client, used, ok := drmResidentClient(data) + if !ok { + return 0, false + } + key := target + ":" + client + // dup() and fork() can expose the same client more than once. The + // snapshot is not atomic; retain its largest observed reading. + clients[key] = max(clients[key], used) + } + + // A worker may be spawned by any thread, not just the thread leader. + tasks, err := os.ReadDir(filepath.Join(base, "task")) + if err != nil || len(tasks) == 0 { + return 0, false + } + for _, task := range tasks { + // #nosec G304 -- procRoot is /proc in production (a temp dir in tests); + // base adds an integer PID, and task.Name comes from os.ReadDir. + // The kernel supplies these path components, not request input. + data, err := os.ReadFile(filepath.Join(base, "task", task.Name(), "children")) + if err != nil { + return 0, false + } + for _, raw := range strings.Fields(string(data)) { + child, err := strconv.Atoi(raw) + if err != nil || child <= 0 { + return 0, false + } + pending = append(pending, child) + } + } + } + var total uint64 + for _, used := range clients { + if used > math.MaxUint64-total { + return 0, false + } + total += used + } + return total, len(clients) > 0 +} + +func drmResidentClient(data []byte) (string, uint64, bool) { + var client string + var total uint64 + found := false + scanner := bufio.NewScanner(bytes.NewReader(data)) + for scanner.Scan() { + key, value, ok := strings.Cut(scanner.Text(), ":") + if !ok { + continue + } + if key == "drm-client-id" { + id, err := strconv.ParseUint(strings.TrimSpace(value), 10, 64) + if err != nil { + return "", 0, false + } + client = strconv.FormatUint(id, 10) + } + region, resident := strings.CutPrefix(key, "drm-resident-") + if !resident || !isVRAMRegion(region) { + continue + } + used, ok := drmResidentBytes(value) + if !ok || used > math.MaxUint64-total { + return "", 0, false + } + total += used + found = true + } + return client, total, scanner.Err() == nil && client != "" && found +} + +func drmResidentBytes(value string) (uint64, bool) { + fields := strings.Fields(value) + if len(fields) == 0 || len(fields) > 2 { + return 0, false + } + n, err := strconv.ParseUint(fields[0], 10, 64) + if err != nil { + return 0, false + } + unit := uint64(1) + if len(fields) == 2 { + switch strings.ToLower(fields[1]) { + case "b": + case "kib": + unit = 1 << 10 + case "mib": + unit = 1 << 20 + case "gib": + unit = 1 << 30 + default: + return 0, false + } + } + if n > math.MaxUint64/unit { + return 0, false + } + return n * unit, true +} diff --git a/pkg/xsysinfo/process_vram_linux_test.go b/pkg/xsysinfo/process_vram_linux_test.go new file mode 100644 index 000000000..4de1f6cdd --- /dev/null +++ b/pkg/xsysinfo/process_vram_linux_test.go @@ -0,0 +1,105 @@ +//go:build linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +import ( + "os" + "path/filepath" + "strconv" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("ProcessVRAM", func() { + var root string + write := func(path, contents string) { + Expect(os.MkdirAll(filepath.Dir(path), 0750)).To(Succeed()) + Expect(os.WriteFile(path, []byte(contents), 0600)).To(Succeed()) + } + addProcess := func(pid int, children string) { + base := filepath.Join(root, strconv.Itoa(pid)) + Expect(os.MkdirAll(filepath.Join(base, "fd"), 0750)).To(Succeed()) + write(filepath.Join(base, "task", strconv.Itoa(pid), "children"), children) + } + addFD := func(pid, fd int, render, info string) { + base := filepath.Join(root, strconv.Itoa(pid)) + name := strconv.Itoa(fd) + Expect(os.Symlink("/dev/dri/"+render, filepath.Join(base, "fd", name))).To(Succeed()) + write(filepath.Join(base, "fdinfo", name), info) + } + BeforeEach(func() { + var err error + root, err = os.MkdirTemp("", "process-vram-") + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(os.RemoveAll, root) + addProcess(100, "") + }) + + It("sums resident device memory across GPUs and child processes without duplicate clients", func() { + write(filepath.Join(root, "100/task/101/children"), "200") + addProcess(200, "") + info := "drm-client-id: 7\ndrm-total-local0: 900 MiB\ndrm-resident-local0: 128 MiB\ndrm-resident-system0: 4 GiB\n" + addFD(100, 3, "renderD128", info) + addFD(100, 4, "renderD128", info) + addFD(200, 3, "renderD128", info) + addFD(200, 4, "renderD129", "drm-client-id: 7\ndrm-resident-vram0: 256 MiB\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(Equal(uint64(384 * 1024 * 1024))) + }) + + It("distinguishes a measured zero from unavailable accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-local0: 0 B\n") + used, ok := processVRAM(root, 100) + Expect(ok).To(BeTrue()) + Expect(used).To(BeZero()) + }) + + DescribeTable("does not invent readings from unsupported or invalid accounting", + func(info string) { + addFD(100, 3, "renderD128", info) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("no resident keys", "drm-client-id: 7\ndrm-total-vram0: 128 MiB\n"), + Entry("host memory only", "drm-client-id: 7\ndrm-resident-system0: 128 MiB\n"), + Entry("no client identity", "drm-resident-vram0: 128 MiB\n"), + Entry("malformed size", "drm-client-id: 7\ndrm-resident-vram0: unknown KiB\n"), + Entry("unknown unit", "drm-client-id: 7\ndrm-resident-vram0: 128 widgets\n"), + Entry("overflow", "drm-client-id: 7\ndrm-resident-vram0: 18446744073709551615 GiB\n"), + ) + + It("omits a partial reading if a child cannot be inspected", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + write(filepath.Join(root, "100/task/100/children"), "200") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + It("omits a partial reading if another DRM client lacks accounting", func() { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + addFD(100, 4, "renderD129", "drm-client-id: 8\n") + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }) + + DescribeTable("omits mixed readings with unsupported GPU descriptors", + func(target string) { + addFD(100, 3, "renderD128", "drm-client-id: 7\ndrm-resident-vram0: 128 MiB\n") + Expect(os.Symlink(target, filepath.Join(root, "100/fd/4"))).To(Succeed()) + _, ok := processVRAM(root, 100) + Expect(ok).To(BeFalse()) + }, + Entry("primary DRM node", "/dev/dri/card0"), + Entry("NVIDIA device", "/dev/nvidia0"), + ) + + It("returns unavailable for missing processes or no DRM descriptors", func() { + for _, pid := range []int{-1, 0, 100, 999} { + _, ok := processVRAM(root, pid) + Expect(ok).To(BeFalse()) + } + }) +}) diff --git a/pkg/xsysinfo/process_vram_other.go b/pkg/xsysinfo/process_vram_other.go new file mode 100644 index 000000000..06cc49186 --- /dev/null +++ b/pkg/xsysinfo/process_vram_other.go @@ -0,0 +1,9 @@ +//go:build !linux + +// SPDX-License-Identifier: MIT +package xsysinfo + +// ProcessVRAM is unavailable on platforms without Linux DRM fdinfo accounting. +func ProcessVRAM(pid int) (uint64, bool) { + return 0, false +} diff --git a/scripts/build/gallery/main.go b/scripts/build/gallery/main.go new file mode 100644 index 000000000..5e749eaaa --- /dev/null +++ b/scripts/build/gallery/main.go @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: MIT +// Package the official index and its repository-local base configurations. +package main + +import ( + "fmt" + "os" + "path/filepath" + "strings" + + "gopkg.in/yaml.v3" +) + +func main() { + if len(os.Args) != 4 { + fmt.Fprintln(os.Stderr, "usage: gallery REPOSITORY {gallery|backend} OUTPUT") + os.Exit(1) + } + if err := packageGallery(os.Args[1], os.Args[2], os.Args[3]); err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } +} + +func packageGallery(root, source, output string) error { + if source != "gallery" && source != "backend" { + return fmt.Errorf("unsupported gallery directory %q", source) + } + repository, err := os.OpenRoot(root) + if err != nil { + return err + } + defer func() { _ = repository.Close() }() + body, err := repository.ReadFile(filepath.Join(source, "index.yaml")) + if err != nil { + return err + } + var doc yaml.Node + if err := yaml.Unmarshal(body, &doc); err != nil { + return err + } + // The build operator explicitly selects the output directory via the CLI. + if err := os.MkdirAll(output, 0700); err != nil { // #nosec G703 -- caller-selected output root + return err + } + destination, err := os.OpenRoot(output) + if err != nil { + return err + } + defer func() { _ = destination.Close() }() + // Keep the tree relative to the repository root so repeated base configs + // share a layer, even when an index refers outside its own directory. + const prefix = "github:mudler/LocalAI/" + var walk func(*yaml.Node) error + walk = func(n *yaml.Node) error { + if n.Kind == yaml.MappingNode { + for i := 0; i < len(n.Content); i += 2 { + value := n.Content[i+1] + if n.Content[i].Value != "url" || value.Kind != yaml.ScalarNode || !strings.HasPrefix(value.Value, prefix) || !strings.HasSuffix(value.Value, "@master") { + continue + } + path := strings.TrimSuffix(strings.TrimPrefix(value.Value, prefix), "@master") + if !filepath.IsLocal(path) { + return fmt.Errorf("base config escapes repository: %q", path) + } + config, err := repository.ReadFile(path) + if err != nil { + return err + } + if err := destination.MkdirAll(filepath.Dir(path), 0700); err != nil { + return err + } + if err := destination.WriteFile(path, config, 0600); err != nil { + return err + } + value.Value = filepath.ToSlash(path) + } + } + for _, child := range n.Content { + if err := walk(child); err != nil { + return err + } + } + return nil + } + if err := walk(&doc); err != nil { + return err + } + body, err = yaml.Marshal(&doc) + if err != nil { + return err + } + return destination.WriteFile("index.yaml", body, 0600) +} diff --git a/scripts/build/gallery/main_test.go b/scripts/build/gallery/main_test.go new file mode 100644 index 000000000..646499b4b --- /dev/null +++ b/scripts/build/gallery/main_test.go @@ -0,0 +1,83 @@ +// SPDX-License-Identifier: MIT +package main + +import ( + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "gopkg.in/yaml.v3" + "os" + "path/filepath" + "strings" + "testing" +) + +func TestGalleryPackage(t *testing.T) { RegisterFailHandler(Fail); RunSpecs(t, "Gallery packaging") } + +var _ = Describe("Gallery packaging", func() { + It("packages both official indexes with every repository-local base available offline", func() { + for _, source := range []string{"gallery", "backend"} { + out := GinkgoT().TempDir() + Expect(packageGallery("../../..", source, out)).To(Succeed()) + body, err := os.ReadFile(filepath.Join(out, "index.yaml")) + Expect(err).ToNot(HaveOccurred()) + var entries []map[string]any + Expect(yaml.Unmarshal(body, &entries)).To(Succeed()) + Expect(entries).ToNot(BeEmpty()) + for _, entry := range entries { + url, _ := entry["url"].(string) + Expect(url).ToNot(HavePrefix("github:mudler/LocalAI/")) + if strings.HasPrefix(url, "gallery/") { + Expect(filepath.Join(out, url)).To(BeAnExistingFile()) + } + } + } + }) + It("bundles local base configs and preserves external URLs and YAML aliases", func() { + root := GinkgoT().TempDir() + Expect(os.MkdirAll(filepath.Join(root, "gallery"), 0755)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(root, "gallery/base.yaml"), []byte("backend: llama-cpp\n"), 0644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- &base\n name: first\n url: github:mudler/LocalAI/gallery/base.yaml@master\n- <<: *base\n name: second\n- name: external\n url: https://example.com/config.yaml\n"), 0644)).To(Succeed()) + out := filepath.Join(root, "out") + Expect(packageGallery(root, "gallery", out)).To(Succeed()) + data, err := os.ReadFile(filepath.Join(out, "index.yaml")) + Expect(err).ToNot(HaveOccurred()) + var entries []map[string]any + Expect(yaml.Unmarshal(data, &entries)).To(Succeed()) + Expect(entries[0]["url"]).To(Equal("gallery/base.yaml")) + Expect(entries[1]["url"]).To(Equal("gallery/base.yaml")) + Expect(entries[2]["url"]).To(Equal("https://example.com/config.yaml")) + body, err := os.ReadFile(filepath.Join(out, "gallery/base.yaml")) + Expect(err).ToNot(HaveOccurred()) + Expect(string(body)).To(Equal("backend: llama-cpp\n")) + }) + It("fails if a referenced config is missing or escapes the repository", func() { + for _, ref := range []string{"missing.yaml", "../../outside.yaml"} { + root := GinkgoT().TempDir() + Expect(os.Mkdir(filepath.Join(root, "gallery"), 0755)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), []byte("- name: broken\n url: github:mudler/LocalAI/gallery/"+ref+"@master\n"), 0644)).To(Succeed()) + Expect(packageGallery(root, "gallery", filepath.Join(root, "out"))).ToNot(Succeed()) + } + }) + It("rejects symlink escapes when reading configs or writing the bundle", func() { + for _, location := range []string{"source", "output"} { + root, out, outside := GinkgoT().TempDir(), GinkgoT().TempDir(), GinkgoT().TempDir() + for _, dir := range []string{filepath.Join(root, "gallery"), filepath.Join(out, "gallery")} { + Expect(os.Mkdir(dir, 0700)).To(Succeed()) + } + index := []byte("- name: test\n url: github:mudler/LocalAI/gallery/base.yaml@master\n") + Expect(os.WriteFile(filepath.Join(root, "gallery/index.yaml"), index, 0600)).To(Succeed()) + outsideFile := filepath.Join(outside, "base.yaml") + Expect(os.WriteFile(outsideFile, []byte("outside"), 0600)).To(Succeed()) + link := filepath.Join(root, "gallery/base.yaml") + if location == "output" { + Expect(os.WriteFile(link, []byte("inside"), 0600)).To(Succeed()) + link = filepath.Join(out, "gallery/base.yaml") + } + Expect(os.Symlink(outsideFile, link)).To(Succeed()) + Expect(packageGallery(root, "gallery", out)).ToNot(Succeed(), location) + data, err := os.ReadFile(outsideFile) + Expect(err).ToNot(HaveOccurred()) + Expect(string(data)).To(Equal("outside")) + } + }) +}) diff --git a/swagger/docs.go b/swagger/docs.go index 6ae74b94a..c447043ae 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -7313,6 +7313,9 @@ const docTemplate = `{ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, @@ -7807,6 +7810,10 @@ const docTemplate = `{ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.json b/swagger/swagger.json index c19487f6c..b4e49b347 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -7310,6 +7310,9 @@ "id": { "type": "string" }, + "metadata": { + "type": "object" + }, "model": { "type": "string" }, @@ -7804,6 +7807,10 @@ }, "id": { "type": "string" + }, + "size_vram": { + "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", + "type": "integer" } } }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 6f9f6b1b7..34fcfc8d7 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2266,6 +2266,8 @@ definitions: type: array id: type: string + metadata: + type: object model: type: string object: @@ -2652,6 +2654,11 @@ definitions: type: string id: type: string + size_vram: + description: |- + SizeVRAM is DRM-accounted resident device memory in bytes. Nil means + the backend process tree has no complete supported reading. + type: integer type: object schema.SystemInformationResponse: properties: diff --git a/website/data/stats.yaml b/website/data/stats.yaml index a54e240e8..253b5444d 100644 --- a/website/data/stats.yaml +++ b/website/data/stats.yaml @@ -3,10 +3,10 @@ # The four GitHub fields are rewritten by .github/ci/refresh-site-counters.sh, # which runs weekly from .github/workflows/refresh-site-counters.yml. Editing # them by hand works but will be overwritten on the next run. -stars: 48949 -forks: 4430 -contributors: 237 -releases: 135 +stars: 49204 +forks: 4459 +contributors: 245 +releases: 136 # The GitHub API cannot answer for this one, so it is maintained by hand and # the refresh script carries it through untouched.