chore: drop neutts for l4t (#8101 )

Builds exhausts CI currently, and there are better backends at this point in time. We will probably deprecate it in the future. Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
chore(model gallery): add qwen3-coder-30b-a3b-instruct based on model request (#8082 )
2026-02-03 11:13:31 -05:00 · 2026-01-18 21:55:56 +01:00 · 2026-01-18 09:23:07 +01:00 · 2026-01-17 22:12:21 +01:00 · 2026-01-17 22:11:47 +01:00 · 2026-01-17 21:10:18 +00:00
159 changed files with 18725 additions and 1681 deletions
--- a/.github/workflows/backend.yml
+++ b/.github/workflows/backend.yml
--- a/.github/workflows/dependabot_auto.yml
+++ b/.github/workflows/dependabot_auto.yml
@@ -14,7 +14,7 @@ jobs:
    steps:
      - name: Dependabot metadata
        id: metadata
-        uses: dependabot/fetch-metadata@v2.4.0
+        uses: dependabot/fetch-metadata@v2.5.0
        with:
          github-token: "${{ secrets.GITHUB_TOKEN }}"
          skip-commit-verification: true
--- a/.github/workflows/generate_grpc_cache.yaml
+++ b/.github/workflows/generate_grpc_cache.yaml
@@ -16,7 +16,7 @@ jobs:
    strategy:
      matrix:
        include:
-          - grpc-base-image: ubuntu:22.04
+          - grpc-base-image: ubuntu:24.04
            runs-on: 'ubuntu-latest'
            platforms: 'linux/amd64,linux/arm64'
    runs-on: ${{matrix.runs-on}}
--- a/.github/workflows/generate_intel_image.yaml
+++ b/.github/workflows/generate_intel_image.yaml
@@ -15,7 +15,7 @@ jobs:
    strategy:
      matrix:
        include:
-          - base-image: intel/oneapi-basekit:2025.2.0-0-devel-ubuntu22.04
+          - base-image: intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04
            runs-on: 'arc-runner-set'
            platforms: 'linux/amd64'
    runs-on: ${{matrix.runs-on}}
@@ -53,7 +53,7 @@ jobs:
            BASE_IMAGE=${{ matrix.base-image }}
          context: .
          file: ./Dockerfile
-          tags: quay.io/go-skynet/intel-oneapi-base:latest
+          tags: quay.io/go-skynet/intel-oneapi-base:24.04
          push: true
          target: intel
          platforms: ${{ matrix.platforms }}
--- a/.github/workflows/image-pr.yml
+++ b/.github/workflows/image-pr.yml
@@ -1,94 +1,95 @@
 ---
-name: 'build container images tests'
-
-on:
-  pull_request:
-
-concurrency:
-  group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
-  cancel-in-progress: true
-
-jobs:
-  image-build:
-    uses: ./.github/workflows/image_build.yml
-    with:
-      tag-latest: ${{ matrix.tag-latest }}
-      tag-suffix: ${{ matrix.tag-suffix }}
-      build-type: ${{ matrix.build-type }}
-      cuda-major-version: ${{ matrix.cuda-major-version }}
-      cuda-minor-version: ${{ matrix.cuda-minor-version }}
-      platforms: ${{ matrix.platforms }}
-      runs-on: ${{ matrix.runs-on }}
-      base-image: ${{ matrix.base-image }}
-      grpc-base-image: ${{ matrix.grpc-base-image }}
-      makeflags: ${{ matrix.makeflags }}
-      ubuntu-version: ${{ matrix.ubuntu-version }}
-    secrets:
-      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-    strategy:
-      # Pushing with all jobs in parallel
-      # eats the bandwidth of all the nodes
-      max-parallel: ${{ github.event_name != 'pull_request' && 4 || 8 }}
-      fail-fast: false
-      matrix:
-        include:
-          - build-type: 'cublas'
-            cuda-major-version: "12"
-            cuda-minor-version: "0"
-            platforms: 'linux/amd64'
-            tag-latest: 'false'
-            tag-suffix: '-gpu-nvidia-cuda-12'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            makeflags: "--jobs=3 --output-sync=target"
-            ubuntu-version: '2204'
-          - build-type: 'cublas'
-            cuda-major-version: "13"
-            cuda-minor-version: "0"
-            platforms: 'linux/amd64'
-            tag-latest: 'false'
-            tag-suffix: '-gpu-nvidia-cuda-13'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            makeflags: "--jobs=3 --output-sync=target"
-            ubuntu-version: '2204'
-          - build-type: 'hipblas'
-            platforms: 'linux/amd64'
-            tag-latest: 'false'
-            tag-suffix: '-hipblas'
-            base-image: "rocm/dev-ubuntu-22.04:6.4.3"
-            grpc-base-image: "ubuntu:22.04"
-            runs-on: 'ubuntu-latest'
-            makeflags: "--jobs=3 --output-sync=target"
-            ubuntu-version: '2204'
-          - build-type: 'sycl'
-            platforms: 'linux/amd64'
-            tag-latest: 'false'
-            base-image: "quay.io/go-skynet/intel-oneapi-base:latest"
-            grpc-base-image: "ubuntu:22.04"
-            tag-suffix: 'sycl'
-            runs-on: 'ubuntu-latest'
-            makeflags: "--jobs=3 --output-sync=target"
-            ubuntu-version: '2204'
-          - build-type: 'vulkan'
-            platforms: 'linux/amd64'
-            tag-latest: 'false'
-            tag-suffix: '-vulkan-core'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            makeflags: "--jobs=4 --output-sync=target"
-            ubuntu-version: '2204'
-          - build-type: 'cublas'
-            cuda-major-version: "13"
-            cuda-minor-version: "0"
-            platforms: 'linux/arm64'
-            tag-latest: 'false'
-            tag-suffix: '-nvidia-l4t-arm64-cuda-13'
-            base-image: "ubuntu:24.04"
-            runs-on: 'ubuntu-24.04-arm'
-            makeflags: "--jobs=4 --output-sync=target"
-            skip-drivers: 'false'
-            ubuntu-version: '2404'
+  name: 'build container images tests'
+  
+  on:
+    pull_request:
+  
+  concurrency:
+    group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
+    cancel-in-progress: true
+  
+  jobs:
+    image-build:
+      uses: ./.github/workflows/image_build.yml
+      with:
+        tag-latest: ${{ matrix.tag-latest }}
+        tag-suffix: ${{ matrix.tag-suffix }}
+        build-type: ${{ matrix.build-type }}
+        cuda-major-version: ${{ matrix.cuda-major-version }}
+        cuda-minor-version: ${{ matrix.cuda-minor-version }}
+        platforms: ${{ matrix.platforms }}
+        runs-on: ${{ matrix.runs-on }}
+        base-image: ${{ matrix.base-image }}
+        grpc-base-image: ${{ matrix.grpc-base-image }}
+        makeflags: ${{ matrix.makeflags }}
+        ubuntu-version: ${{ matrix.ubuntu-version }}
+      secrets:
+        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+      strategy:
+        # Pushing with all jobs in parallel
+        # eats the bandwidth of all the nodes
+        max-parallel: ${{ github.event_name != 'pull_request' && 4 || 8 }}
+        fail-fast: false
+        matrix:
+          include:
+            - build-type: 'cublas'
+              cuda-major-version: "12"
+              cuda-minor-version: "9"
+              platforms: 'linux/amd64'
+              tag-latest: 'false'
+              tag-suffix: '-gpu-nvidia-cuda-12'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:24.04"
+              makeflags: "--jobs=3 --output-sync=target"
+              ubuntu-version: '2404'
+            - build-type: 'cublas'
+              cuda-major-version: "13"
+              cuda-minor-version: "0"
+              platforms: 'linux/amd64'
+              tag-latest: 'false'
+              tag-suffix: '-gpu-nvidia-cuda-13'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:22.04"
+              makeflags: "--jobs=3 --output-sync=target"
+              ubuntu-version: '2404'
+            - build-type: 'hipblas'
+              platforms: 'linux/amd64'
+              tag-latest: 'false'
+              tag-suffix: '-hipblas'
+              base-image: "rocm/dev-ubuntu-24.04:6.4.4"
+              grpc-base-image: "ubuntu:24.04"
+              runs-on: 'ubuntu-latest'
+              makeflags: "--jobs=3 --output-sync=target"
+              ubuntu-version: '2404'
+            - build-type: 'sycl'
+              platforms: 'linux/amd64'
+              tag-latest: 'false'
+              base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
+              grpc-base-image: "ubuntu:24.04"
+              tag-suffix: 'sycl'
+              runs-on: 'ubuntu-latest'
+              makeflags: "--jobs=3 --output-sync=target"
+              ubuntu-version: '2404'
+            - build-type: 'vulkan'
+              platforms: 'linux/amd64,linux/arm64'
+              tag-latest: 'false'
+              tag-suffix: '-vulkan-core'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:24.04"
+              makeflags: "--jobs=4 --output-sync=target"
+              ubuntu-version: '2404'
+            - build-type: 'cublas'
+              cuda-major-version: "13"
+              cuda-minor-version: "0"
+              platforms: 'linux/arm64'
+              tag-latest: 'false'
+              tag-suffix: '-nvidia-l4t-arm64-cuda-13'
+              base-image: "ubuntu:24.04"
+              runs-on: 'ubuntu-24.04-arm'
+              makeflags: "--jobs=4 --output-sync=target"
+              skip-drivers: 'false'
+              ubuntu-version: '2404'
+  
--- a/.github/workflows/image.yml
+++ b/.github/workflows/image.yml
@@ -1,187 +1,187 @@
 ---
-name: 'build container images'
-
-on:
-  push:
-    branches:
-      - master
-    tags:
-      - '*'
-
-concurrency:
-  group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
-  cancel-in-progress: true
-
-jobs:
-  hipblas-jobs:
-    uses: ./.github/workflows/image_build.yml
-    with:
-      tag-latest: ${{ matrix.tag-latest }}
-      tag-suffix: ${{ matrix.tag-suffix }}
-      build-type: ${{ matrix.build-type }}
-      cuda-major-version: ${{ matrix.cuda-major-version }}
-      cuda-minor-version: ${{ matrix.cuda-minor-version }}
-      platforms: ${{ matrix.platforms }}
-      runs-on: ${{ matrix.runs-on }}
-      base-image: ${{ matrix.base-image }}
-      grpc-base-image: ${{ matrix.grpc-base-image }}
-      aio: ${{ matrix.aio }}
-      makeflags: ${{ matrix.makeflags }}
-      ubuntu-version: ${{ matrix.ubuntu-version }}
-    secrets:
-      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-    strategy:
-      matrix:
-        include:
-          - build-type: 'hipblas'
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            tag-suffix: '-gpu-hipblas'
-            base-image: "rocm/dev-ubuntu-22.04:6.4.3"
-            grpc-base-image: "ubuntu:22.04"
-            runs-on: 'ubuntu-latest'
-            makeflags: "--jobs=3 --output-sync=target"
-            aio: "-aio-gpu-hipblas"
-            ubuntu-version: '2204'
-
-  core-image-build:
-    uses: ./.github/workflows/image_build.yml
-    with:
-      tag-latest: ${{ matrix.tag-latest }}
-      tag-suffix: ${{ matrix.tag-suffix }}
-      build-type: ${{ matrix.build-type }}
-      cuda-major-version: ${{ matrix.cuda-major-version }}
-      cuda-minor-version: ${{ matrix.cuda-minor-version }}
-      platforms: ${{ matrix.platforms }}
-      runs-on: ${{ matrix.runs-on }}
-      aio: ${{ matrix.aio }}
-      base-image: ${{ matrix.base-image }}
-      grpc-base-image: ${{ matrix.grpc-base-image }}
-      makeflags: ${{ matrix.makeflags }}
-      skip-drivers: ${{ matrix.skip-drivers }}
-      ubuntu-version: ${{ matrix.ubuntu-version }}
-    secrets:
-      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-    strategy:
-      #max-parallel: ${{ github.event_name != 'pull_request' && 2 || 4 }}
-      matrix:
-        include:
-          - build-type: ''
-            platforms: 'linux/amd64,linux/arm64'
-            tag-latest: 'auto'
-            tag-suffix: ''
-            base-image: "ubuntu:22.04"
-            runs-on: 'ubuntu-latest'
-            aio: "-aio-cpu"
-            makeflags: "--jobs=4 --output-sync=target"
-            skip-drivers: 'false'
-            ubuntu-version: '2204'
-          - build-type: 'cublas'
-            cuda-major-version: "11"
-            cuda-minor-version: "7"
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            tag-suffix: '-gpu-nvidia-cuda-11'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            makeflags: "--jobs=4 --output-sync=target"
-            skip-drivers: 'false'
-            aio: "-aio-gpu-nvidia-cuda-11"
-            ubuntu-version: '2204'
-          - build-type: 'cublas'
-            cuda-major-version: "12"
-            cuda-minor-version: "0"
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            tag-suffix: '-gpu-nvidia-cuda-12'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            skip-drivers: 'false'
-            makeflags: "--jobs=4 --output-sync=target"
-            aio: "-aio-gpu-nvidia-cuda-12"
-            ubuntu-version: '2204'
-          - build-type: 'cublas'
-            cuda-major-version: "13"
-            cuda-minor-version: "0"
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            tag-suffix: '-gpu-nvidia-cuda-13'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            skip-drivers: 'false'
-            makeflags: "--jobs=4 --output-sync=target"
-            aio: "-aio-gpu-nvidia-cuda-13"
-            ubuntu-version: '2204'
-          - build-type: 'vulkan'
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            tag-suffix: '-gpu-vulkan'
-            runs-on: 'ubuntu-latest'
-            base-image: "ubuntu:22.04"
-            skip-drivers: 'false'
-            makeflags: "--jobs=4 --output-sync=target"
-            aio: "-aio-gpu-vulkan"
-            ubuntu-version: '2204'
-          - build-type: 'intel'
-            platforms: 'linux/amd64'
-            tag-latest: 'auto'
-            base-image: "quay.io/go-skynet/intel-oneapi-base:latest"
-            grpc-base-image: "ubuntu:22.04"
-            tag-suffix: '-gpu-intel'
-            runs-on: 'ubuntu-latest'
-            makeflags: "--jobs=3 --output-sync=target"
-            aio: "-aio-gpu-intel"
-            ubuntu-version: '2204'
-
-  gh-runner:
-    uses: ./.github/workflows/image_build.yml
-    with:
-      tag-latest: ${{ matrix.tag-latest }}
-      tag-suffix: ${{ matrix.tag-suffix }}
-      build-type: ${{ matrix.build-type }}
-      cuda-major-version: ${{ matrix.cuda-major-version }}
-      cuda-minor-version: ${{ matrix.cuda-minor-version }}
-      platforms: ${{ matrix.platforms }}
-      runs-on: ${{ matrix.runs-on }}
-      aio: ${{ matrix.aio }}
-      base-image: ${{ matrix.base-image }}
-      grpc-base-image: ${{ matrix.grpc-base-image }}
-      makeflags: ${{ matrix.makeflags }}
-      skip-drivers: ${{ matrix.skip-drivers }}
-      ubuntu-version: ${{ matrix.ubuntu-version }}
-    secrets:
-      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-    strategy:
-      matrix:
-        include:
-          - build-type: 'cublas'
-            cuda-major-version: "12"
-            cuda-minor-version: "0"
-            platforms: 'linux/arm64'
-            tag-latest: 'auto'
-            tag-suffix: '-nvidia-l4t-arm64'
-            base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
-            runs-on: 'ubuntu-24.04-arm'
-            makeflags: "--jobs=4 --output-sync=target"
-            skip-drivers: 'true'
-            ubuntu-version: "2204"
-          - build-type: 'cublas'
-            cuda-major-version: "13"
-            cuda-minor-version: "0"
-            platforms: 'linux/arm64'
-            tag-latest: 'auto'
-            tag-suffix: '-nvidia-l4t-arm64-cuda-13'
-            base-image: "ubuntu:24.04"
-            runs-on: 'ubuntu-24.04-arm'
-            makeflags: "--jobs=4 --output-sync=target"
-            skip-drivers: 'false'
-            ubuntu-version: '2404'
+  name: 'build container images'
+  
+  on:
+    push:
+      branches:
+        - master
+      tags:
+        - '*'
+  
+  concurrency:
+    group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
+    cancel-in-progress: true
+  
+  jobs:
+    hipblas-jobs:
+      uses: ./.github/workflows/image_build.yml
+      with:
+        tag-latest: ${{ matrix.tag-latest }}
+        tag-suffix: ${{ matrix.tag-suffix }}
+        build-type: ${{ matrix.build-type }}
+        cuda-major-version: ${{ matrix.cuda-major-version }}
+        cuda-minor-version: ${{ matrix.cuda-minor-version }}
+        platforms: ${{ matrix.platforms }}
+        runs-on: ${{ matrix.runs-on }}
+        base-image: ${{ matrix.base-image }}
+        grpc-base-image: ${{ matrix.grpc-base-image }}
+        aio: ${{ matrix.aio }}
+        makeflags: ${{ matrix.makeflags }}
+        ubuntu-version: ${{ matrix.ubuntu-version }}
+        ubuntu-codename: ${{ matrix.ubuntu-codename }}
+      secrets:
+        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+      strategy:
+        matrix:
+          include:
+            - build-type: 'hipblas'
+              platforms: 'linux/amd64'
+              tag-latest: 'auto'
+              tag-suffix: '-gpu-hipblas'
+              base-image: "rocm/dev-ubuntu-24.04:6.4.4"
+              grpc-base-image: "ubuntu:24.04"
+              runs-on: 'ubuntu-latest'
+              makeflags: "--jobs=3 --output-sync=target"
+              aio: "-aio-gpu-hipblas"
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+  
+    core-image-build:
+      uses: ./.github/workflows/image_build.yml
+      with:
+        tag-latest: ${{ matrix.tag-latest }}
+        tag-suffix: ${{ matrix.tag-suffix }}
+        build-type: ${{ matrix.build-type }}
+        cuda-major-version: ${{ matrix.cuda-major-version }}
+        cuda-minor-version: ${{ matrix.cuda-minor-version }}
+        platforms: ${{ matrix.platforms }}
+        runs-on: ${{ matrix.runs-on }}
+        aio: ${{ matrix.aio }}
+        base-image: ${{ matrix.base-image }}
+        grpc-base-image: ${{ matrix.grpc-base-image }}
+        makeflags: ${{ matrix.makeflags }}
+        skip-drivers: ${{ matrix.skip-drivers }}
+        ubuntu-version: ${{ matrix.ubuntu-version }}
+        ubuntu-codename: ${{ matrix.ubuntu-codename }}
+      secrets:
+        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+      strategy:
+        #max-parallel: ${{ github.event_name != 'pull_request' && 2 || 4 }}
+        matrix:
+          include:
+            - build-type: ''
+              platforms: 'linux/amd64,linux/arm64'
+              tag-latest: 'auto'
+              tag-suffix: ''
+              base-image: "ubuntu:24.04"
+              runs-on: 'ubuntu-latest'
+              aio: "-aio-cpu"
+              makeflags: "--jobs=4 --output-sync=target"
+              skip-drivers: 'false'
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+            - build-type: 'cublas'
+              cuda-major-version: "12"
+              cuda-minor-version: "9"
+              platforms: 'linux/amd64'
+              tag-latest: 'auto'
+              tag-suffix: '-gpu-nvidia-cuda-12'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:24.04"
+              skip-drivers: 'false'
+              makeflags: "--jobs=4 --output-sync=target"
+              aio: "-aio-gpu-nvidia-cuda-12"
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+            - build-type: 'cublas'
+              cuda-major-version: "13"
+              cuda-minor-version: "0"
+              platforms: 'linux/amd64'
+              tag-latest: 'auto'
+              tag-suffix: '-gpu-nvidia-cuda-13'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:22.04"
+              skip-drivers: 'false'
+              makeflags: "--jobs=4 --output-sync=target"
+              aio: "-aio-gpu-nvidia-cuda-13"
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+            - build-type: 'vulkan'
+              platforms: 'linux/amd64,linux/arm64'
+              tag-latest: 'auto'
+              tag-suffix: '-gpu-vulkan'
+              runs-on: 'ubuntu-latest'
+              base-image: "ubuntu:24.04"
+              skip-drivers: 'false'
+              makeflags: "--jobs=4 --output-sync=target"
+              aio: "-aio-gpu-vulkan"
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+            - build-type: 'intel'
+              platforms: 'linux/amd64'
+              tag-latest: 'auto'
+              base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
+              grpc-base-image: "ubuntu:24.04"
+              tag-suffix: '-gpu-intel'
+              runs-on: 'ubuntu-latest'
+              makeflags: "--jobs=3 --output-sync=target"
+              aio: "-aio-gpu-intel"
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+  
+    gh-runner:
+      uses: ./.github/workflows/image_build.yml
+      with:
+        tag-latest: ${{ matrix.tag-latest }}
+        tag-suffix: ${{ matrix.tag-suffix }}
+        build-type: ${{ matrix.build-type }}
+        cuda-major-version: ${{ matrix.cuda-major-version }}
+        cuda-minor-version: ${{ matrix.cuda-minor-version }}
+        platforms: ${{ matrix.platforms }}
+        runs-on: ${{ matrix.runs-on }}
+        aio: ${{ matrix.aio }}
+        base-image: ${{ matrix.base-image }}
+        grpc-base-image: ${{ matrix.grpc-base-image }}
+        makeflags: ${{ matrix.makeflags }}
+        skip-drivers: ${{ matrix.skip-drivers }}
+        ubuntu-version: ${{ matrix.ubuntu-version }}
+        ubuntu-codename: ${{ matrix.ubuntu-codename }}
+      secrets:
+        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+      strategy:
+        matrix:
+          include:
+            - build-type: 'cublas'
+              cuda-major-version: "12"
+              cuda-minor-version: "0"
+              platforms: 'linux/arm64'
+              tag-latest: 'auto'
+              tag-suffix: '-nvidia-l4t-arm64'
+              base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
+              runs-on: 'ubuntu-24.04-arm'
+              makeflags: "--jobs=4 --output-sync=target"
+              skip-drivers: 'true'
+              ubuntu-version: "2204"
+              ubuntu-codename: 'jammy'
+            - build-type: 'cublas'
+              cuda-major-version: "13"
+              cuda-minor-version: "0"
+              platforms: 'linux/arm64'
+              tag-latest: 'auto'
+              tag-suffix: '-nvidia-l4t-arm64-cuda-13'
+              base-image: "ubuntu:24.04"
+              runs-on: 'ubuntu-24.04-arm'
+              makeflags: "--jobs=4 --output-sync=target"
+              skip-drivers: 'false'
+              ubuntu-version: '2404'
+              ubuntu-codename: 'noble'
+  
--- a/.github/workflows/image_build.yml
+++ b/.github/workflows/image_build.yml
@@ -23,7 +23,7 @@ on:
        type: string
      cuda-minor-version:
        description: 'CUDA minor version'
-        default: "4"
+        default: "9"
        type: string
      platforms:
        description: 'Platforms'
@@ -61,6 +61,11 @@ on:
        required: false
        default: '2204'
        type: string
+      ubuntu-codename:
+        description: 'Ubuntu codename'
+        required: false
+        default: 'noble'
+        type: string
    secrets:
      dockerUsername:
        required: true
@@ -244,6 +249,7 @@ jobs:
            MAKEFLAGS=${{ inputs.makeflags }}
            SKIP_DRIVERS=${{ inputs.skip-drivers }}
            UBUNTU_VERSION=${{ inputs.ubuntu-version }}
+            UBUNTU_CODENAME=${{ inputs.ubuntu-codename }}
          context: .
          file: ./Dockerfile
          cache-from: type=gha
@@ -272,6 +278,7 @@ jobs:
            MAKEFLAGS=${{ inputs.makeflags }}
            SKIP_DRIVERS=${{ inputs.skip-drivers }}
            UBUNTU_VERSION=${{ inputs.ubuntu-version }}
+            UBUNTU_CODENAME=${{ inputs.ubuntu-codename }}
          context: .
          file: ./Dockerfile
          cache-from: type=gha
--- a/.github/workflows/test-extra.yml
+++ b/.github/workflows/test-extra.yml
@@ -247,3 +247,41 @@ jobs:
        run: |
          make --jobs=5 --output-sync=target -C backend/python/coqui
          make --jobs=5 --output-sync=target -C backend/python/coqui test
+  tests-moonshine:
+    runs-on: ubuntu-latest
+    steps:
+      - name: Clone
+        uses: actions/checkout@v6
+        with:
+          submodules: true
+      - name: Dependencies
+        run: |
+          sudo apt-get update
+          sudo apt-get install build-essential ffmpeg
+          sudo apt-get install -y ca-certificates cmake curl patch python3-pip
+          # Install UV
+          curl -LsSf https://astral.sh/uv/install.sh | sh
+          pip install --user --no-cache-dir grpcio-tools==1.64.1
+      - name: Test moonshine
+        run: |
+          make --jobs=5 --output-sync=target -C backend/python/moonshine
+          make --jobs=5 --output-sync=target -C backend/python/moonshine test
+  tests-pocket-tts:
+    runs-on: ubuntu-latest
+    steps:
+      - name: Clone
+        uses: actions/checkout@v6
+        with:
+          submodules: true
+      - name: Dependencies
+        run: |
+          sudo apt-get update
+          sudo apt-get install build-essential ffmpeg
+          sudo apt-get install -y ca-certificates cmake curl patch python3-pip
+          # Install UV
+          curl -LsSf https://astral.sh/uv/install.sh | sh
+          pip install --user --no-cache-dir grpcio-tools==1.64.1
+      - name: Test pocket-tts
+        run: |
+          make --jobs=5 --output-sync=target -C backend/python/pocket-tts
+          make --jobs=5 --output-sync=target -C backend/python/pocket-tts test
--- a/.gitignore
+++ b/.gitignore
@@ -25,6 +25,7 @@ go-bert
 # LocalAI build binary
 LocalAI
 /local-ai
+/local-ai-launcher
 # prevent above rules from omitting the helm chart
 !charts/*
 # prevent above rules from omitting the api/localai folder
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -2,6 +2,163 @@

 Building and testing the project depends on the components involved and the platform where development is taking place. Due to the amount of context required it's usually best not to try building or testing the project unless the user requests it. If you must build the project then inspect the Makefile in the project root and the Makefiles of any backends that are effected by changes you are making. In addition the workflows in .github/workflows can be used as a reference when it is unclear how to build or test a component. The primary Makefile contains targets for building inside or outside Docker, if the user has not previously specified a preference then ask which they would like to use.

+## Building a specified backend
+
+Let's say the user wants to build a particular backend for a given platform. For example let's say they want to build bark for ROCM/hipblas
+
+- The Makefile has targets like `docker-build-bark` created with `generate-docker-build-target` at the time of writing. Recently added backends may require a new target.
+- At a minimum we need to set the BUILD_TYPE, BASE_IMAGE build-args
+  - Use .github/workflows/backend.yml as a reference it lists the needed args in the `include` job strategy matrix
+  - l4t and cublas also requires the CUDA major and minor version
+- You can pretty print a command like `DOCKER_MAKEFLAGS=-j$(nproc --ignore=1) BUILD_TYPE=hipblas BASE_IMAGE=rocm/dev-ubuntu-24.04:6.4.4 make docker-build-bark`
+- Unless the user specifies that they want you to run the command, then just print it because not all agent frontends handle long running jobs well and the output may overflow your context
+- The user may say they want to build AMD or ROCM instead of hipblas, or Intel instead of SYCL or NVIDIA insted of l4t or cublas. Ask for confirmation if there is ambiguity.
+- Sometimes the user may need extra parameters to be added to `docker build` (e.g. `--platform` for cross-platform builds or `--progress` to view the full logs), in which case you can generate the `docker build` command directly.
+
+## Adding a New Backend
+
+When adding a new backend to LocalAI, you need to update several files to ensure the backend is properly built, tested, and registered. Here's a step-by-step guide based on the pattern used for adding backends like `moonshine`:
+
+### 1. Create Backend Directory Structure
+
+Create the backend directory under the appropriate location:
+- **Python backends**: `backend/python/<backend-name>/`
+- **Go backends**: `backend/go/<backend-name>/`
+- **C++ backends**: `backend/cpp/<backend-name>/`
+
+For Python backends, you'll typically need:
+- `backend.py` - Main gRPC server implementation
+- `Makefile` - Build configuration
+- `install.sh` - Installation script for dependencies
+- `protogen.sh` - Protocol buffer generation script
+- `requirements.txt` - Python dependencies
+- `run.sh` - Runtime script
+- `test.py` / `test.sh` - Test files
+
+### 2. Add Build Configurations to `.github/workflows/backend.yml`
+
+Add build matrix entries for each platform/GPU type you want to support. Look at similar backends (e.g., `chatterbox`, `faster-whisper`) for reference.
+
+**Placement in file:**
+- CPU builds: Add after other CPU builds (e.g., after `cpu-chatterbox`)
+- CUDA 12 builds: Add after other CUDA 12 builds (e.g., after `gpu-nvidia-cuda-12-chatterbox`)
+- CUDA 13 builds: Add after other CUDA 13 builds (e.g., after `gpu-nvidia-cuda-13-chatterbox`)
+
+**Additional build types you may need:**
+- ROCm/HIP: Use `build-type: 'hipblas'` with `base-image: "rocm/dev-ubuntu-24.04:6.4.4"`
+- Intel/SYCL: Use `build-type: 'intel'` or `build-type: 'sycl_f16'`/`sycl_f32` with `base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"`
+- L4T (ARM): Use `build-type: 'l4t'` with `platforms: 'linux/arm64'` and `runs-on: 'ubuntu-24.04-arm'`
+
+### 3. Add Backend Metadata to `backend/index.yaml`
+
+**Step 3a: Add Meta Definition**
+
+Add a YAML anchor definition in the `## metas` section (around line 2-300). Look for similar backends to use as a template such as `diffusers` or `chatterbox`
+
+**Step 3b: Add Image Entries**
+
+Add image entries at the end of the file, following the pattern of similar backends such as `diffusers` or `chatterbox`. Include both `latest` (production) and `master` (development) tags.
+
+### 4. Update the Makefile
+
+The Makefile needs to be updated in several places to support building and testing the new backend:
+
+**Step 4a: Add to `.NOTPARALLEL`**
+
+Add `backends/<backend-name>` to the `.NOTPARALLEL` line (around line 2) to prevent parallel execution conflicts:
+
+```makefile
+.NOTPARALLEL: ... backends/<backend-name>
+```
+
+**Step 4b: Add to `prepare-test-extra`**
+
+Add the backend to the `prepare-test-extra` target (around line 312) to prepare it for testing:
+
+```makefile
+prepare-test-extra: protogen-python
+	...
+	$(MAKE) -C backend/python/<backend-name>
+```
+
+**Step 4c: Add to `test-extra`**
+
+Add the backend to the `test-extra` target (around line 319) to run its tests:
+
+```makefile
+test-extra: prepare-test-extra
+	...
+	$(MAKE) -C backend/python/<backend-name> test
+```
+
+**Step 4d: Add Backend Definition**
+
+Add a backend definition variable in the backend definitions section (around line 428-457). The format depends on the backend type:
+
+**For Python backends with root context** (like `faster-whisper`, `bark`):
+```makefile
+BACKEND_<BACKEND_NAME> = <backend-name>|python|.|false|true
+```
+
+**For Python backends with `./backend` context** (like `chatterbox`, `moonshine`):
+```makefile
+BACKEND_<BACKEND_NAME> = <backend-name>|python|./backend|false|true
+```
+
+**For Go backends**:
+```makefile
+BACKEND_<BACKEND_NAME> = <backend-name>|golang|.|false|true
+```
+
+**Step 4e: Generate Docker Build Target**
+
+Add an eval call to generate the docker-build target (around line 480-501):
+
+```makefile
+$(eval $(call generate-docker-build-target,$(BACKEND_<BACKEND_NAME>)))
+```
+
+**Step 4f: Add to `docker-build-backends`**
+
+Add `docker-build-<backend-name>` to the `docker-build-backends` target (around line 507):
+
+```makefile
+docker-build-backends: ... docker-build-<backend-name>
+```
+
+**Determining the Context:**
+
+- If the backend is in `backend/python/<backend-name>/` and uses `./backend` as context in the workflow file, use `./backend` context
+- If the backend is in `backend/python/<backend-name>/` but uses `.` as context in the workflow file, use `.` context
+- Check similar backends to determine the correct context
+
+### 5. Verification Checklist
+
+After adding a new backend, verify:
+
+- [ ] Backend directory structure is complete with all necessary files
+- [ ] Build configurations added to `.github/workflows/backend.yml` for all desired platforms
+- [ ] Meta definition added to `backend/index.yaml` in the `## metas` section
+- [ ] Image entries added to `backend/index.yaml` for all build variants (latest + development)
+- [ ] Tag suffixes match between workflow file and index.yaml
+- [ ] Makefile updated with all 6 required changes (`.NOTPARALLEL`, `prepare-test-extra`, `test-extra`, backend definition, docker-build target eval, `docker-build-backends`)
+- [ ] No YAML syntax errors (check with linter)
+- [ ] No Makefile syntax errors (check with linter)
+- [ ] Follows the same pattern as similar backends (e.g., if it's a transcription backend, follow `faster-whisper` pattern)
+
+### 6. Example: Adding a Python Backend
+
+For reference, when `moonshine` was added:
+- **Files created**: `backend/python/moonshine/{backend.py, Makefile, install.sh, protogen.sh, requirements.txt, run.sh, test.py, test.sh}`
+- **Workflow entries**: 3 build configurations (CPU, CUDA 12, CUDA 13)
+- **Index entries**: 1 meta definition + 6 image entries (cpu, cuda12, cuda13 × latest/development)
+- **Makefile updates**: 
+  - Added to `.NOTPARALLEL` line
+  - Added to `prepare-test-extra` and `test-extra` targets
+  - Added `BACKEND_MOONSHINE = moonshine|python|./backend|false|true`
+  - Added eval for docker-build target generation
+  - Added `docker-build-moonshine` to `docker-build-backends`
+
 # Coding style

 - The project has the following .editorconfig
@@ -77,3 +234,49 @@ When fixing compilation errors after upstream changes:
 - HTTP server uses `server_routes` with HTTP handlers
 - Both use the same `server_context` and task queue infrastructure
 - gRPC methods: `LoadModel`, `Predict`, `PredictStream`, `Embedding`, `Rerank`, `TokenizeString`, `GetMetrics`, `Health`
+
+## Tool Call Parsing Maintenance
+
+When working on JSON/XML tool call parsing functionality, always check llama.cpp for reference implementation and updates:
+
+### Checking for XML Parsing Changes
+
+1. **Review XML Format Definitions**: Check `llama.cpp/common/chat-parser-xml-toolcall.h` for `xml_tool_call_format` struct changes
+2. **Review Parsing Logic**: Check `llama.cpp/common/chat-parser-xml-toolcall.cpp` for parsing algorithm updates
+3. **Review Format Presets**: Check `llama.cpp/common/chat-parser.cpp` for new XML format presets (search for `xml_tool_call_format form`)
+4. **Review Model Lists**: Check `llama.cpp/common/chat.h` for `COMMON_CHAT_FORMAT_*` enum values that use XML parsing:
+   - `COMMON_CHAT_FORMAT_GLM_4_5`
+   - `COMMON_CHAT_FORMAT_MINIMAX_M2`
+   - `COMMON_CHAT_FORMAT_KIMI_K2`
+   - `COMMON_CHAT_FORMAT_QWEN3_CODER_XML`
+   - `COMMON_CHAT_FORMAT_APRIEL_1_5`
+   - `COMMON_CHAT_FORMAT_XIAOMI_MIMO`
+   - Any new formats added
+
+### Model Configuration Options
+
+Always check `llama.cpp` for new model configuration options that should be supported in LocalAI:
+
+1. **Check Server Context**: Review `llama.cpp/tools/server/server-context.cpp` for new parameters
+2. **Check Chat Params**: Review `llama.cpp/common/chat.h` for `common_chat_params` struct changes
+3. **Check Server Options**: Review `llama.cpp/tools/server/server.cpp` for command-line argument changes
+4. **Examples of options to check**:
+   - `ctx_shift` - Context shifting support
+   - `parallel_tool_calls` - Parallel tool calling
+   - `reasoning_format` - Reasoning format options
+   - Any new flags or parameters
+
+### Implementation Guidelines
+
+1. **Feature Parity**: Always aim for feature parity with llama.cpp's implementation
+2. **Test Coverage**: Add tests for new features matching llama.cpp's behavior
+3. **Documentation**: Update relevant documentation when adding new formats or options
+4. **Backward Compatibility**: Ensure changes don't break existing functionality
+
+### Files to Monitor
+
+- `llama.cpp/common/chat-parser-xml-toolcall.h` - Format definitions
+- `llama.cpp/common/chat-parser-xml-toolcall.cpp` - Parsing logic
+- `llama.cpp/common/chat-parser.cpp` - Format presets and model-specific handlers
+- `llama.cpp/common/chat.h` - Format enums and parameter structures
+- `llama.cpp/tools/server/server-context.cpp` - Server configuration options
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -78,6 +78,20 @@ LOCALAI_IMAGE_TAG=test LOCALAI_IMAGE=local-ai-aio make run-e2e-aio

 We are welcome the contribution of the documents, please open new PR or create a new issue. The documentation is available under `docs/` https://github.com/mudler/LocalAI/tree/master/docs

+### Gallery YAML Schema
+
+LocalAI provides a JSON Schema for gallery model YAML files at:
+
+`core/schema/gallery-model.schema.json`
+
+This schema mirrors the internal gallery model configuration and can be used by editors (such as VS Code) to enable autocomplete, validation, and inline documentation when creating or modifying gallery files.
+
+To use it with the YAML language server, add the following comment at the top of a gallery YAML file:
+
+```yaml
+# yaml-language-server: $schema=../core/schema/gallery-model.schema.json
+```
+
 ## Community and Communication

 - You can reach out via the Github issue tracker.
--- a/59
+++ b/59
@@ -1,6 +1,7 @@
-ARG BASE_IMAGE=ubuntu:22.04
+ARG BASE_IMAGE=ubuntu:24.04
 ARG GRPC_BASE_IMAGE=${BASE_IMAGE}
 ARG INTEL_BASE_IMAGE=${BASE_IMAGE}
+ARG UBUNTU_CODENAME=noble

 FROM ${BASE_IMAGE} AS requirements

@@ -9,7 +10,7 @@ ENV DEBIAN_FRONTEND=noninteractive
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        ca-certificates curl wget espeak-ng libgomp1 \
-        ffmpeg && \
+        ffmpeg libopenblas0 libopenblas-dev && \
    apt-get clean && \
    rm -rf /var/lib/apt/lists/*

@@ -23,7 +24,7 @@ ARG SKIP_DRIVERS=false
 ARG TARGETARCH
 ARG TARGETVARIANT
 ENV BUILD_TYPE=${BUILD_TYPE}
-ARG UBUNTU_VERSION=2204
+ARG UBUNTU_VERSION=2404

 RUN mkdir -p /run/localai
 RUN echo "default" > /run/localai/capability
@@ -34,11 +35,45 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
-        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
-        apt-get update && \
-        apt-get install -y \
-            vulkan-sdk && \
+        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
+            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
+            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
+            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
+            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
+            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils mesa-vulkan-drivers
+        if [ "amd64" = "$TARGETARCH" ]; then
+            wget "https://sdk.lunarg.com/sdk/download/1.4.335.0/linux/vulkansdk-linux-x86_64-1.4.335.0.tar.xz" && \
+            tar -xf vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            rm vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            mkdir -p /opt/vulkan-sdk && \
+            mv 1.4.335.0 /opt/vulkan-sdk/ && \
+            cd /opt/vulkan-sdk/1.4.335.0 && \
+            ./vulkansdk --no-deps --maxjobs \
+                vulkan-loader \
+                vulkan-validationlayers \
+                vulkan-extensionlayer \
+                vulkan-tools \
+                shaderc && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/bin/* /usr/bin/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/include/* /usr/include/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/share/* /usr/share/ && \
+            rm -rf /opt/vulkan-sdk
+        fi
+        if [ "arm64" = "$TARGETARCH" ]; then
+            mkdir vulkan && cd vulkan && \
+            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
+            tar -xvf vulkan-sdk.tar.xz && \
+            rm vulkan-sdk.tar.xz && \
+            cd 1.4.335.0 && \
+            cp -rfv aarch64/bin/* /usr/bin/ && \
+            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
+            cp -rfv aarch64/include/* /usr/include/ && \
+            cp -rfv aarch64/share/* /usr/share/ && \
+            cd ../.. && \
+            rm -rf vulkan
+        fi
+        ldconfig && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/* && \
        echo "vulkan" > /run/localai/capability
@@ -141,13 +176,12 @@ ENV PATH=/opt/rocm/bin:${PATH}
 # The requirements-core target is common to all images.  It should not be placed in requirements-core unless every single build will use it.
 FROM requirements-drivers AS build-requirements

-ARG GO_VERSION=1.22.6
-ARG CMAKE_VERSION=3.26.4
+ARG GO_VERSION=1.25.4
+ARG CMAKE_VERSION=3.31.10
 ARG CMAKE_FROM_SOURCE=false
 ARG TARGETARCH
 ARG TARGETVARIANT

-
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        build-essential \
@@ -204,9 +238,10 @@ WORKDIR /build
 # https://community.intel.com/t5/Intel-oneAPI-Math-Kernel-Library/APT-Repository-not-working-signatures-invalid/m-p/1599436/highlight/true#M36143
 # This is a temporary workaround until Intel fixes their repository
 FROM ${INTEL_BASE_IMAGE} AS intel
+ARG UBUNTU_CODENAME=noble
 RUN wget -qO - https://repositories.intel.com/gpu/intel-graphics.key | \
 gpg --yes --dearmor --output /usr/share/keyrings/intel-graphics.gpg
-RUN echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] https://repositories.intel.com/gpu/ubuntu jammy/lts/2350 unified" > /etc/apt/sources.list.d/intel-graphics.list
+RUN echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] https://repositories.intel.com/gpu/ubuntu ${UBUNTU_CODENAME}/lts/2350 unified" > /etc/apt/sources.list.d/intel-graphics.list
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        intel-oneapi-runtime-libs && \
--- a/Dockerfile.aio
+++ b/Dockerfile.aio
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:22.04
+ARG BASE_IMAGE=ubuntu:24.04

 FROM ${BASE_IMAGE} 

--- a/55
+++ b/55
@@ -1,3 +1,6 @@
+# Disable parallel execution for backend builds
+.NOTPARALLEL: backends/diffusers backends/llama-cpp backends/piper backends/stablediffusion-ggml backends/whisper backends/faster-whisper backends/silero-vad backends/local-store backends/huggingface backends/rfdetr backends/kitten-tts backends/kokoro backends/chatterbox backends/llama-cpp-darwin backends/neutts build-darwin-python-backend build-darwin-go-backend backends/mlx backends/diffuser-darwin backends/mlx-vlm backends/mlx-audio backends/stablediffusion-ggml-darwin backends/vllm backends/moonshine backends/pocket-tts
+
 GOCMD=go
 GOTEST=$(GOCMD) test
 GOVET=$(GOCMD) vet
@@ -6,11 +9,14 @@ LAUNCHER_BINARY_NAME=local-ai-launcher

 CUDA_MAJOR_VERSION?=13
 CUDA_MINOR_VERSION?=0
-UBUNTU_VERSION?=2204
+UBUNTU_VERSION?=2404
+UBUNTU_CODENAME?=noble

 GORELEASER?=

 export BUILD_TYPE?=
+export CUDA_MAJOR_VERSION?=12
+export CUDA_MINOR_VERSION?=9

 GO_TAGS?=
 BUILD_ID?=
@@ -164,6 +170,7 @@ docker-build-aio:
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		-t local-ai:tests -f Dockerfile .
 	BASE_IMAGE=local-ai:tests DOCKER_AIO_IMAGE=local-ai-aio:test $(MAKE) docker-aio
@@ -194,6 +201,7 @@ prepare-e2e:
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
 		-t localai-tests .
@@ -307,6 +315,8 @@ prepare-test-extra: protogen-python
 	$(MAKE) -C backend/python/chatterbox
 	$(MAKE) -C backend/python/vllm
 	$(MAKE) -C backend/python/vibevoice
+	$(MAKE) -C backend/python/moonshine
+	$(MAKE) -C backend/python/pocket-tts

 test-extra: prepare-test-extra
 	$(MAKE) -C backend/python/transformers test
@@ -314,11 +324,13 @@ test-extra: prepare-test-extra
 	$(MAKE) -C backend/python/chatterbox test
 	$(MAKE) -C backend/python/vllm test
 	$(MAKE) -C backend/python/vibevoice test
+	$(MAKE) -C backend/python/moonshine test
+	$(MAKE) -C backend/python/pocket-tts test

 DOCKER_IMAGE?=local-ai
 DOCKER_AIO_IMAGE?=local-ai-aio
 IMAGE_TYPE?=core
-BASE_IMAGE?=ubuntu:22.04
+BASE_IMAGE?=ubuntu:24.04

 docker:
 	docker build \
@@ -330,19 +342,21 @@ docker:
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		-t $(DOCKER_IMAGE) .

-docker-cuda11:
+docker-cuda12:
 	docker build \
-		--build-arg CUDA_MAJOR_VERSION=11 \
-		--build-arg CUDA_MINOR_VERSION=8 \
+		--build-arg CUDA_MAJOR_VERSION=${CUDA_MAJOR_VERSION} \
+		--build-arg CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION} \
 		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
 		--build-arg IMAGE_TYPE=$(IMAGE_TYPE) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
 		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		-t $(DOCKER_IMAGE)-cuda-11 .
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
+		-t $(DOCKER_IMAGE)-cuda-12 .

 docker-aio:
 	@echo "Building AIO image with base $(BASE_IMAGE) as $(DOCKER_AIO_IMAGE)"
@@ -352,6 +366,7 @@ docker-aio:
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		-t $(DOCKER_AIO_IMAGE) -f Dockerfile.aio .

 docker-aio-all:
@@ -360,7 +375,7 @@ docker-aio-all:

 docker-image-intel:
 	docker build \
-		--build-arg BASE_IMAGE=quay.io/go-skynet/intel-oneapi-base:latest \
+		--build-arg BASE_IMAGE=intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04 \
 		--build-arg IMAGE_TYPE=$(IMAGE_TYPE) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
@@ -368,6 +383,7 @@ docker-image-intel:
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		-t $(DOCKER_IMAGE) .

 ########################################################
@@ -433,16 +449,16 @@ BACKEND_FASTER_WHISPER = faster-whisper|python|.|false|true
 BACKEND_COQUI = coqui|python|.|false|true
 BACKEND_BARK = bark|python|.|false|true
 BACKEND_EXLLAMA2 = exllama2|python|.|false|true
-
-# Python backends with ./backend context
-BACKEND_RFDETR = rfdetr|python|./backend|false|true
-BACKEND_KITTEN_TTS = kitten-tts|python|./backend|false|true
-BACKEND_NEUTTS = neutts|python|./backend|false|true
-BACKEND_KOKORO = kokoro|python|./backend|false|true
-BACKEND_VLLM = vllm|python|./backend|false|true
-BACKEND_DIFFUSERS = diffusers|python|./backend|--progress=plain|true
-BACKEND_CHATTERBOX = chatterbox|python|./backend|false|true
-BACKEND_VIBEVOICE = vibevoice|python|./backend|--progress=plain|true
+BACKEND_RFDETR = rfdetr|python|.|false|true
+BACKEND_KITTEN_TTS = kitten-tts|python|.|false|true
+BACKEND_NEUTTS = neutts|python|.|false|true
+BACKEND_KOKORO = kokoro|python|.|false|true
+BACKEND_VLLM = vllm|python|.|false|true
+BACKEND_DIFFUSERS = diffusers|python|.|--progress=plain|true
+BACKEND_CHATTERBOX = chatterbox|python|.|false|true
+BACKEND_VIBEVOICE = vibevoice|python|.|--progress=plain|true
+BACKEND_MOONSHINE = moonshine|python|.|false|true
+BACKEND_POCKET_TTS = pocket-tts|python|.|false|true

 # Helper function to build docker image for a backend
 # Usage: $(call docker-build-backend,BACKEND_NAME,DOCKERFILE_TYPE,BUILD_CONTEXT,PROGRESS_FLAG,NEEDS_BACKEND_ARG)
@@ -453,6 +469,7 @@ define docker-build-backend
 		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
 		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
 		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
+		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		$(if $(filter true,$(5)),--build-arg BACKEND=$(1)) \
 		-t local-ai-backend:$(1) -f backend/Dockerfile.$(2) $(3)
 endef
@@ -486,12 +503,14 @@ $(eval $(call generate-docker-build-target,$(BACKEND_VLLM)))
 $(eval $(call generate-docker-build-target,$(BACKEND_DIFFUSERS)))
 $(eval $(call generate-docker-build-target,$(BACKEND_CHATTERBOX)))
 $(eval $(call generate-docker-build-target,$(BACKEND_VIBEVOICE)))
+$(eval $(call generate-docker-build-target,$(BACKEND_MOONSHINE)))
+$(eval $(call generate-docker-build-target,$(BACKEND_POCKET_TTS)))

 # Pattern rule for docker-save targets
 docker-save-%: backend-images
 	docker save local-ai-backend:$* -o backend-images/$*.tar

-docker-build-backends: docker-build-llama-cpp docker-build-rerankers docker-build-vllm docker-build-transformers docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-coqui docker-build-bark docker-build-chatterbox docker-build-vibevoice docker-build-exllama2
+docker-build-backends: docker-build-llama-cpp docker-build-rerankers docker-build-vllm docker-build-transformers docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-coqui docker-build-bark docker-build-chatterbox docker-build-vibevoice docker-build-exllama2 docker-build-moonshine docker-build-pocket-tts

 ########################################################
 ### END Backends
--- a/README.md
+++ b/README.md
@@ -111,6 +111,8 @@

 ## 💻 Quickstart

+> ⚠️ **Note:** The `install.sh` script is currently experiencing issues due to the heavy changes currently undergoing in LocalAI and may produce broken or misconfigured installations. Please use Docker installation (see below) or manual binary installation until [issue #8032](https://github.com/mudler/LocalAI/issues/8032) is resolved.
+
 Run the installer script:

 ```bash
@@ -128,7 +130,7 @@ For more installation options, see [Installer Options](https://localai.io/instal

 > Note: the DMGs are not signed by Apple as quarantined. See https://github.com/mudler/LocalAI/issues/6268 for a workaround, fix is tracked here: https://github.com/mudler/LocalAI/issues/6244

-Or run with docker:
+### Containers (Docker, podman, ...)

 > **💡 Docker Run vs Docker Start**
 > 
@@ -137,13 +139,13 @@ Or run with docker:
 > 
 > If you've already run LocalAI before and want to start it again, use: `docker start -i local-ai`

-### CPU only image:
+#### CPU only image:

 ```bash
 docker run -ti --name local-ai -p 8080:8080 localai/localai:latest
 ```

-### NVIDIA GPU Images:
+#### NVIDIA GPU Images:

 ```bash
 # CUDA 13.0
@@ -152,9 +154,6 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gp
 # CUDA 12.0
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-12

-# CUDA 11.7
-docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-11
-
 # NVIDIA Jetson (L4T) ARM64
 # CUDA 12 (for Nvidia AGX Orin and similar platforms)
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-nvidia-l4t-arm64
@@ -163,25 +162,25 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-nv
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-nvidia-l4t-arm64-cuda-13
 ```

-### AMD GPU Images (ROCm):
+#### AMD GPU Images (ROCm):

 ```bash
 docker run -ti --name local-ai -p 8080:8080 --device=/dev/kfd --device=/dev/dri --group-add=video localai/localai:latest-gpu-hipblas
 ```

-### Intel GPU Images (oneAPI):
+#### Intel GPU Images (oneAPI):

 ```bash
 docker run -ti --name local-ai -p 8080:8080 --device=/dev/dri/card1 --device=/dev/dri/renderD128 localai/localai:latest-gpu-intel
 ```

-### Vulkan GPU Images:
+#### Vulkan GPU Images:

 ```bash
 docker run -ti --name local-ai -p 8080:8080 localai/localai:latest-gpu-vulkan
 ```

-### AIO Images (pre-downloaded models):
+#### AIO Images (pre-downloaded models):

 ```bash
 # CPU version
@@ -193,9 +192,6 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-ai
 # NVIDIA CUDA 12 version
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-12

-# NVIDIA CUDA 11 version
-docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-11
-
 # Intel GPU version
 docker run -ti --name local-ai -p 8080:8080 localai/localai:latest-aio-gpu-intel

@@ -279,9 +275,9 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration
 ### Text Generation & Language Models
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
-| **llama.cpp** | LLM inference in C/C++ | CUDA 11/12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
+| **llama.cpp** | LLM inference in C/C++ | CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
 | **vLLM** | Fast LLM inference with PagedAttention | CUDA 12/13, ROCm, Intel |
-| **transformers** | HuggingFace transformers framework | CUDA 11/12/13, ROCm, Intel, CPU |
+| **transformers** | HuggingFace transformers framework | CUDA 12/13, ROCm, Intel, CPU |
 | **exllama2** | GPTQ inference library | CUDA 12/13 |
 | **MLX** | Apple Silicon LLM inference | Metal (M1/M2/M3+) |
 | **MLX-VLM** | Apple Silicon Vision-Language Models | Metal (M1/M2/M3+) |
@@ -295,24 +291,25 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration
 | **bark-cpp** | C++ implementation of Bark | CUDA, Metal, CPU |
 | **coqui** | Advanced TTS with 1100+ languages | CUDA 12/13, ROCm, Intel, CPU |
 | **kokoro** | Lightweight TTS model | CUDA 12/13, ROCm, Intel, CPU |
-| **chatterbox** | Production-grade TTS | CUDA 11/12/13, CPU |
+| **chatterbox** | Production-grade TTS | CUDA 12/13, CPU |
 | **piper** | Fast neural TTS system | CPU |
 | **kitten-tts** | Kitten TTS models | CPU |
 | **silero-vad** | Voice Activity Detection | CPU |
 | **neutts** | Text-to-speech with voice cloning | CUDA 12/13, ROCm, CPU |
 | **vibevoice** | Real-time TTS with voice cloning | CUDA 12/13, ROCm, Intel, CPU |
+| **pocket-tts** | Lightweight CPU-based TTS | CUDA 12/13, ROCm, Intel, CPU |

 ### Image & Video Generation
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
 | **stablediffusion.cpp** | Stable Diffusion in C/C++ | CUDA 12/13, Intel SYCL, Vulkan, CPU |
-| **diffusers** | HuggingFace diffusion models | CUDA 11/12/13, ROCm, Intel, Metal, CPU |
+| **diffusers** | HuggingFace diffusion models | CUDA 12/13, ROCm, Intel, Metal, CPU |

 ### Specialized AI Tasks
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
 | **rfdetr** | Real-time object detection | CUDA 12/13, Intel, CPU |
-| **rerankers** | Document reranking API | CUDA 11/12/13, ROCm, Intel, CPU |
+| **rerankers** | Document reranking API | CUDA 12/13, ROCm, Intel, CPU |
 | **local-store** | Vector database | CPU |
 | **huggingface** | HuggingFace API integration | API-based |

@@ -320,11 +317,10 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration

 | Acceleration Type | Supported Backends | Hardware Support |
 |-------------------|-------------------|------------------|
-| **NVIDIA CUDA 11** | llama.cpp, whisper, stablediffusion, diffusers, rerankers, bark, chatterbox | Nvidia hardware |
 | **NVIDIA CUDA 12** | All CUDA-compatible backends | Nvidia hardware |
 | **NVIDIA CUDA 13** | All CUDA-compatible backends | Nvidia hardware |
-| **AMD ROCm** | llama.cpp, whisper, vllm, transformers, diffusers, rerankers, coqui, kokoro, bark, neutts, vibevoice | AMD Graphics |
-| **Intel oneAPI** | llama.cpp, whisper, stablediffusion, vllm, transformers, diffusers, rfdetr, rerankers, exllama2, coqui, kokoro, bark, vibevoice | Intel Arc, Intel iGPUs |
+| **AMD ROCm** | llama.cpp, whisper, vllm, transformers, diffusers, rerankers, coqui, kokoro, bark, neutts, vibevoice, pocket-tts | AMD Graphics |
+| **Intel oneAPI** | llama.cpp, whisper, stablediffusion, vllm, transformers, diffusers, rfdetr, rerankers, exllama2, coqui, kokoro, bark, vibevoice, pocket-tts | Intel Arc, Intel iGPUs |
 | **Apple Metal** | llama.cpp, whisper, diffusers, MLX, MLX-VLM, bark-cpp | Apple M1/M2/M3+ |
 | **Vulkan** | llama.cpp, whisper, stablediffusion | Cross-platform GPUs |
 | **NVIDIA Jetson (CUDA 12)** | llama.cpp, whisper, stablediffusion, diffusers, rfdetr | ARM64 embedded AI (AGX Orin, etc.) |
--- a/backend/Dockerfile.golang
+++ b/backend/Dockerfile.golang
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:22.04
+ARG BASE_IMAGE=ubuntu:24.04

 FROM ${BASE_IMAGE} AS builder
 ARG BACKEND=rerankers
@@ -12,8 +12,8 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG GO_VERSION=1.22.6
-ARG UBUNTU_VERSION=2204
+ARG GO_VERSION=1.25.4
+ARG UBUNTU_VERSION=2404

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -40,11 +40,45 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
-        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
-        apt-get update && \
-        apt-get install -y \
-            vulkan-sdk && \
+        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
+            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
+            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
+            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
+            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
+            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
+        if [ "amd64" = "$TARGETARCH" ]; then
+            wget "https://sdk.lunarg.com/sdk/download/1.4.335.0/linux/vulkansdk-linux-x86_64-1.4.335.0.tar.xz" && \
+            tar -xf vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            rm vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            mkdir -p /opt/vulkan-sdk && \
+            mv 1.4.335.0 /opt/vulkan-sdk/ && \
+            cd /opt/vulkan-sdk/1.4.335.0 && \
+            ./vulkansdk --no-deps --maxjobs \
+                vulkan-loader \
+                vulkan-validationlayers \
+                vulkan-extensionlayer \
+                vulkan-tools \
+                shaderc && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/bin/* /usr/bin/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/include/* /usr/include/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/share/* /usr/share/ && \
+            rm -rf /opt/vulkan-sdk
+        fi
+        if [ "arm64" = "$TARGETARCH" ]; then
+            mkdir vulkan && cd vulkan && \
+            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
+            tar -xvf vulkan-sdk.tar.xz && \
+            rm vulkan-sdk.tar.xz && \
+            cd 1.4.335.0 && \
+            cp -rfv aarch64/bin/* /usr/bin/ && \
+            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
+            cp -rfv aarch64/include/* /usr/include/ && \
+            cp -rfv aarch64/share/* /usr/share/ && \
+            cd ../.. && \
+            rm -rf vulkan
+        fi
+        ldconfig && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -148,6 +182,8 @@ EOT

 COPY . /LocalAI

+RUN git config --global --add safe.directory /LocalAI
+
 RUN cd /LocalAI && make protogen-go && make -C /LocalAI/backend/go/${BACKEND} build

 FROM scratch
--- a/backend/Dockerfile.llama-cpp
+++ b/backend/Dockerfile.llama-cpp
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:22.04
+ARG BASE_IMAGE=ubuntu:24.04
 ARG GRPC_BASE_IMAGE=${BASE_IMAGE}


@@ -10,7 +10,8 @@ FROM ${GRPC_BASE_IMAGE} AS grpc
 ARG GRPC_MAKEFLAGS="-j4 -Otarget"
 ARG GRPC_VERSION=v1.65.0
 ARG CMAKE_FROM_SOURCE=false
-ARG CMAKE_VERSION=3.26.4
+# CUDA Toolkit 13.x compatibility: CMake 3.31.9+ fixes toolchain detection/arch table issues
+ARG CMAKE_VERSION=3.31.10

 ENV MAKEFLAGS=${GRPC_MAKEFLAGS}

@@ -26,7 +27,7 @@ RUN apt-get update && \

 # Install CMake (the version in 22.04 is too old)
 RUN <<EOT bash
-    if [ "${CMAKE_FROM_SOURCE}}" = "true" ]; then
+    if [ "${CMAKE_FROM_SOURCE}" = "true" ]; then
        curl -L -s https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}.tar.gz -o cmake.tar.gz && tar xvf cmake.tar.gz && cd cmake-${CMAKE_VERSION} && ./configure && make && make install
    else
        apt-get update && \
@@ -50,6 +51,13 @@ RUN git clone --recurse-submodules --jobs 4 -b ${GRPC_VERSION} --depth 1 --shall
    rm -rf /build

 FROM ${BASE_IMAGE} AS builder
+ARG CMAKE_FROM_SOURCE=false
+ARG CMAKE_VERSION=3.31.10
+# We can target specific CUDA ARCHITECTURES like --build-arg CUDA_DOCKER_ARCH='75;86;89;120'
+ARG CUDA_DOCKER_ARCH
+ENV CUDA_DOCKER_ARCH=${CUDA_DOCKER_ARCH}
+ARG CMAKE_ARGS
+ENV CMAKE_ARGS=${CMAKE_ARGS}
 ARG BACKEND=rerankers
 ARG BUILD_TYPE
 ENV BUILD_TYPE=${BUILD_TYPE}
@@ -61,8 +69,8 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG GO_VERSION=1.22.6
-ARG UBUNTU_VERSION=2204
+ARG GO_VERSION=1.25.4
+ARG UBUNTU_VERSION=2404

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -70,6 +78,7 @@ RUN apt-get update && \
        ccache git \
        ca-certificates \
        make \
+        pkg-config libcurl4-openssl-dev \
        curl unzip \
        libssl-dev wget && \
    apt-get clean && \
@@ -88,11 +97,45 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
-        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
-        apt-get update && \
-        apt-get install -y \
-            vulkan-sdk && \
+        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
+            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
+            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
+            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
+            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
+            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
+        if [ "amd64" = "$TARGETARCH" ]; then
+            wget "https://sdk.lunarg.com/sdk/download/1.4.335.0/linux/vulkansdk-linux-x86_64-1.4.335.0.tar.xz" && \
+            tar -xf vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            rm vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            mkdir -p /opt/vulkan-sdk && \
+            mv 1.4.335.0 /opt/vulkan-sdk/ && \
+            cd /opt/vulkan-sdk/1.4.335.0 && \
+            ./vulkansdk --no-deps --maxjobs \
+                vulkan-loader \
+                vulkan-validationlayers \
+                vulkan-extensionlayer \
+                vulkan-tools \
+                shaderc && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/bin/* /usr/bin/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/include/* /usr/include/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/share/* /usr/share/ && \
+            rm -rf /opt/vulkan-sdk
+        fi
+        if [ "arm64" = "$TARGETARCH" ]; then
+            mkdir vulkan && cd vulkan && \
+            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
+            tar -xvf vulkan-sdk.tar.xz && \
+            rm vulkan-sdk.tar.xz && \
+            cd 1.4.335.0 && \
+            cp -rfv aarch64/bin/* /usr/bin/ && \
+            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
+            cp -rfv aarch64/include/* /usr/include/ && \
+            cp -rfv aarch64/share/* /usr/share/ && \
+            cd ../.. && \
+            rm -rf vulkan
+        fi
+        ldconfig && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -189,7 +232,7 @@ EOT

 # Install CMake (the version in 22.04 is too old)
 RUN <<EOT bash
-    if [ "${CMAKE_FROM_SOURCE}}" = "true" ]; then
+    if [ "${CMAKE_FROM_SOURCE}" = "true" ]; then
        curl -L -s https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}.tar.gz -o cmake.tar.gz && tar xvf cmake.tar.gz && cd cmake-${CMAKE_VERSION} && ./configure && make && make install
    else
        apt-get update && \
@@ -205,19 +248,30 @@ COPY --from=grpc /opt/grpc /usr/local

 COPY . /LocalAI

-## Otherwise just run the normal build
-RUN <<EOT bash
-if [ "${TARGETARCH}" = "arm64" ] || [ "${BUILD_TYPE}" = "hipblas" ]; then \
-        cd /LocalAI/backend/cpp/llama-cpp && make llama-cpp-fallback && \
-        make llama-cpp-grpc && make llama-cpp-rpc-server; \
-    else \
-        cd /LocalAI/backend/cpp/llama-cpp && make llama-cpp-avx && \
-        make llama-cpp-avx2 && \
-        make llama-cpp-avx512 && \
-        make llama-cpp-fallback && \
-        make llama-cpp-grpc && \
-        make llama-cpp-rpc-server; \
-    fi  
+RUN <<'EOT' bash
+set -euxo pipefail
+
+if [[ -n "${CUDA_DOCKER_ARCH:-}" ]]; then
+  CUDA_ARCH_ESC="${CUDA_DOCKER_ARCH//;/\\;}"
+  export CMAKE_ARGS="${CMAKE_ARGS:-} -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCH_ESC}"
+  echo "CMAKE_ARGS(env) = ${CMAKE_ARGS}"
+  rm -rf /LocalAI/backend/cpp/llama-cpp-*-build
+fi
+
+if [ "${TARGETARCH}" = "arm64" ] || [ "${BUILD_TYPE}" = "hipblas" ]; then
+  cd /LocalAI/backend/cpp/llama-cpp
+  make llama-cpp-fallback
+  make llama-cpp-grpc
+  make llama-cpp-rpc-server
+else
+  cd /LocalAI/backend/cpp/llama-cpp
+  make llama-cpp-avx
+  make llama-cpp-avx2
+  make llama-cpp-avx512
+  make llama-cpp-fallback
+  make llama-cpp-grpc
+  make llama-cpp-rpc-server
+fi
 EOT


--- a/backend/Dockerfile.python
+++ b/backend/Dockerfile.python
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:22.04
+ARG BASE_IMAGE=ubuntu:24.04

 FROM ${BASE_IMAGE} AS builder
 ARG BACKEND=rerankers
@@ -12,7 +12,7 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG UBUNTU_VERSION=2204
+ARG UBUNTU_VERSION=2404

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -54,11 +54,45 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
-        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
-        apt-get update && \
-        apt-get install -y \
-            vulkan-sdk && \
+        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
+            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
+            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
+            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
+            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
+            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
+        if [ "amd64" = "$TARGETARCH" ]; then
+            wget "https://sdk.lunarg.com/sdk/download/1.4.335.0/linux/vulkansdk-linux-x86_64-1.4.335.0.tar.xz" && \
+            tar -xf vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            rm vulkansdk-linux-x86_64-1.4.335.0.tar.xz && \
+            mkdir -p /opt/vulkan-sdk && \
+            mv 1.4.335.0 /opt/vulkan-sdk/ && \
+            cd /opt/vulkan-sdk/1.4.335.0 && \
+            ./vulkansdk --no-deps --maxjobs \
+                vulkan-loader \
+                vulkan-validationlayers \
+                vulkan-extensionlayer \
+                vulkan-tools \
+                shaderc && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/bin/* /usr/bin/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/include/* /usr/include/ && \
+            cp -rfv /opt/vulkan-sdk/1.4.335.0/x86_64/share/* /usr/share/ && \
+            rm -rf /opt/vulkan-sdk
+        fi
+        if [ "arm64" = "$TARGETARCH" ]; then
+            mkdir vulkan && cd vulkan && \
+            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
+            tar -xvf vulkan-sdk.tar.xz && \
+            rm vulkan-sdk.tar.xz && \
+            cd 1.4.335.0 && \
+            cp -rfv aarch64/bin/* /usr/bin/ && \
+            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
+            cp -rfv aarch64/include/* /usr/include/ && \
+            cp -rfv aarch64/share/* /usr/share/ && \
+            cd ../.. && \
+            rm -rf vulkan
+        fi
+        ldconfig && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -142,7 +176,8 @@ RUN if [ "${BUILD_TYPE}" = "hipblas" ]; then \
 # Install uv as a system package
 RUN curl -LsSf https://astral.sh/uv/install.sh | UV_INSTALL_DIR=/usr/bin sh
 ENV PATH="/root/.cargo/bin:${PATH}"
-
+# Increase timeout for uv installs behind slow networks
+ENV UV_HTTP_TIMEOUT=180
 RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y

 # Install grpcio-tools (the version in 22.04 is too old)
@@ -155,12 +190,18 @@ RUN <<EOT bash
 EOT


-COPY python/${BACKEND} /${BACKEND}
-COPY backend.proto /${BACKEND}/backend.proto
-COPY python/common/ /${BACKEND}/common
+COPY backend/python/${BACKEND} /${BACKEND}
+COPY backend/backend.proto /${BACKEND}/backend.proto
+COPY backend/python/common/ /${BACKEND}/common
+COPY scripts/build/package-gpu-libs.sh /package-gpu-libs.sh

 RUN cd /${BACKEND} && PORTABLE_PYTHON=true make

+# Package GPU libraries into the backend's lib directory
+RUN mkdir -p /${BACKEND}/lib && \
+    TARGET_LIB_DIR="/${BACKEND}/lib" BUILD_TYPE="${BUILD_TYPE}" CUDA_MAJOR_VERSION="${CUDA_MAJOR_VERSION}" \
+    bash /package-gpu-libs.sh "/${BACKEND}/lib"
+
 FROM scratch
 ARG BACKEND=rerankers
 COPY --from=builder /${BACKEND}/ /
--- a/backend/README.md
+++ b/backend/README.md
@@ -65,7 +65,7 @@ The backend system provides language-specific Dockerfiles that handle the build
 ## Hardware Acceleration Support

 ### CUDA (NVIDIA)
- **Versions**: CUDA 11.x, 12.x
+- **Versions**: CUDA 12.x, 13.x
 - **Features**: cuBLAS, cuDNN, TensorRT optimization
 - **Targets**: x86_64, ARM64 (Jetson)

@@ -132,8 +132,7 @@ For ARM64/Mac builds, docker can't be used, and the makefile in the respective b
 ### Build Types

 - **`cpu`**: CPU-only optimization
- **`cublas11`**: CUDA 11.x with cuBLAS
- **`cublas12`**: CUDA 12.x with cuBLAS
+- **`cublas12`**, **`cublas13`**: CUDA 12.x, 13.x with cuBLAS
 - **`hipblas`**: ROCm with rocBLAS
 - **`intel`**: Intel oneAPI optimization
 - **`vulkan`**: Vulkan-based acceleration
@@ -210,4 +209,4 @@ When contributing to the backend system:
 2. **Add Tests**: Include comprehensive test coverage
 3. **Document**: Provide clear usage examples
 4. **Optimize**: Consider performance and resource usage
-5. **Validate**: Test across different hardware targets
+5. **Validate**: Test across different hardware targets
--- a/backend/cpp/llama-cpp/CMakeLists.txt
+++ b/backend/cpp/llama-cpp/CMakeLists.txt
@@ -70,4 +70,4 @@ target_link_libraries(${TARGET} PRIVATE common llama mtmd ${CMAKE_THREAD_LIBS_IN
 target_compile_features(${TARGET} PRIVATE cxx_std_11)
 if(TARGET BUILD_INFO)
  add_dependencies(${TARGET} BUILD_INFO)
-endif()
+endif()
--- a/backend/cpp/llama-cpp/Makefile
+++ b/backend/cpp/llama-cpp/Makefile
@@ -1,5 +1,5 @@

-LLAMA_VERSION?=e57f52334b2e8436a94f7e332462dfc63a08f995
+LLAMA_VERSION?=2fbde785bc106ae1c4102b0e82b9b41d9c466579
 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp

 CMAKE_ARGS?=
@@ -7,7 +7,7 @@ BUILD_TYPE?=
 NATIVE?=false
 ONEAPI_VARS?=/opt/intel/oneapi/setvars.sh
 TARGET?=--target grpc-server
-JOBS?=$(shell nproc)
+JOBS?=$(shell nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)
 ARCH?=$(shell uname -m)

 # Disable Shared libs as we are linking on static gRPC and we can't mix shared and static
@@ -107,39 +107,21 @@ llama-cpp-avx: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build purge
 	$(info ${GREEN}I llama-cpp build info:avx${RESET})
-ifeq ($(OS),Darwin)
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
-else ifeq ($(ARCH),$(filter $(ARCH),aarch64 arm64))
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
-else
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DCMAKE_C_FLAGS=-mno-bmi2 -DCMAKE_CXX_FLAGS=-mno-bmi2" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
-endif
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build/grpc-server llama-cpp-avx

 llama-cpp-fallback: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build purge
 	$(info ${GREEN}I llama-cpp build info:fallback${RESET})
-ifeq ($(OS),Darwin)
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
-else ifeq ($(ARCH),$(filter $(ARCH),aarch64 arm64))
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
-else
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DCMAKE_C_FLAGS='-mno-bmi -mno-bmi2' -DCMAKE_CXX_FLAGS='-mno-bmi -mno-bmi2'" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
-endif
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build/grpc-server llama-cpp-fallback

 llama-cpp-grpc: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build purge
 	$(info ${GREEN}I llama-cpp build info:grpc${RESET})
-ifeq ($(OS),Darwin)
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
-else ifeq ($(ARCH),$(filter $(ARCH),aarch64 arm64))
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
-else
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DCMAKE_C_FLAGS='-mno-bmi -mno-bmi2' -DCMAKE_CXX_FLAGS='-mno-bmi -mno-bmi2'" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
-endif
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build/grpc-server llama-cpp-grpc

 llama-cpp-rpc-server: llama-cpp-grpc
--- a/backend/cpp/llama-cpp/grpc-server.cpp
+++ b/backend/cpp/llama-cpp/grpc-server.cpp
@@ -23,6 +23,7 @@
 #include <grpcpp/health_check_service_interface.h>
 #include <regex>
 #include <atomic>
+#include <mutex>
 #include <signal.h>
 #include <thread>

@@ -293,8 +294,6 @@ json parse_options(bool streaming, const backend::PredictOptions* predict, const
    return data;
 }

-// Sanitize tools JSON to remove null values from tool.parameters.properties
-// This prevents Jinja template errors when processing tools with malformed parameter schemas

 const std::vector<ggml_type> kv_cache_types = {
    GGML_TYPE_F32,
@@ -392,8 +391,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
    // Initialize fit_params options (can be overridden by options)
    // fit_params: whether to auto-adjust params to fit device memory (default: true as in llama.cpp)
    params.fit_params = true;
-    // fit_params_target: target margin per device in bytes (default: 1GB)
-    params.fit_params_target = 1024 * 1024 * 1024;
+    // fit_params_target: target margin per device in bytes (default: 1GB per device)
+    // Initialize as vector with default value for all devices
+    params.fit_params_target = std::vector<size_t>(llama_max_devices(), 1024 * 1024 * 1024);
    // fit_params_min_ctx: minimum context size for fit (default: 4096)
    params.fit_params_min_ctx = 4096;

@@ -470,10 +470,28 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
        } else if (!strcmp(optname, "fit_params_target") || !strcmp(optname, "fit_target")) {
            if (optval != NULL) {
                try {
-                    // Value is in MiB, convert to bytes
-                    params.fit_params_target = static_cast<size_t>(std::stoi(optval_str)) * 1024 * 1024;
+                    // Value is in MiB, can be comma-separated list for multiple devices
+                    // Single value is broadcast across all devices
+                    std::string arg_next = optval_str;
+                    const std::regex regex{ R"([,/]+)" };
+                    std::sregex_token_iterator it{ arg_next.begin(), arg_next.end(), regex, -1 };
+                    std::vector<std::string> split_arg{ it, {} };
+                    if (split_arg.size() >= llama_max_devices()) {
+                        // Too many values provided
+                        continue;
+                    }
+                    if (split_arg.size() == 1) {
+                        // Single value: broadcast to all devices
+                        size_t value_mib = std::stoul(split_arg[0]);
+                        std::fill(params.fit_params_target.begin(), params.fit_params_target.end(), value_mib * 1024 * 1024);
+                    } else {
+                        // Multiple values: set per device
+                        for (size_t i = 0; i < split_arg.size() && i < params.fit_params_target.size(); i++) {
+                            params.fit_params_target[i] = std::stoul(split_arg[i]) * 1024 * 1024;
+                        }
+                    }
                } catch (const std::exception& e) {
-                    // If conversion fails, keep default value (1GB)
+                    // If conversion fails, keep default value (1GB per device)
                }
            }
        } else if (!strcmp(optname, "fit_params_min_ctx") || !strcmp(optname, "fit_ctx")) {
@@ -688,13 +706,13 @@ private:
 public:
    BackendServiceImpl(server_context& ctx) : ctx_server(ctx) {}

-    grpc::Status Health(ServerContext* /*context*/, const backend::HealthMessage* /*request*/, backend::Reply* reply) {
+    grpc::Status Health(ServerContext* /*context*/, const backend::HealthMessage* /*request*/, backend::Reply* reply) override {
        // Implement Health RPC
        reply->set_message("OK");
        return Status::OK;
    }

-    grpc::Status LoadModel(ServerContext* /*context*/, const backend::ModelOptions* request, backend::Result* result) {
+    grpc::Status LoadModel(ServerContext* /*context*/, const backend::ModelOptions* request, backend::Result* result) override {
        // Implement LoadModel RPC
        common_params params;
        params_parse(ctx_server, request, params);
@@ -711,11 +729,72 @@ public:
        LOG_INF("\n");
        LOG_INF("%s\n", common_params_get_system_info(params).c_str());
        LOG_INF("\n");
+        
+        // Capture error messages during model loading
+        struct error_capture {
+            std::string captured_error;
+            std::mutex error_mutex;
+            ggml_log_callback original_callback;
+            void* original_user_data;
+        } error_capture_data;
+        
+        // Get original log callback
+        llama_log_get(&error_capture_data.original_callback, &error_capture_data.original_user_data);
+        
+        // Set custom callback to capture errors
+        llama_log_set([](ggml_log_level level, const char * text, void * user_data) {
+            auto* capture = static_cast<error_capture*>(user_data);
+            
+            // Capture error messages
+            if (level == GGML_LOG_LEVEL_ERROR) {
+                std::lock_guard<std::mutex> lock(capture->error_mutex);
+                // Append error message, removing trailing newlines
+                std::string msg(text);
+                while (!msg.empty() && (msg.back() == '\n' || msg.back() == '\r')) {
+                    msg.pop_back();
+                }
+                if (!msg.empty()) {
+                    if (!capture->captured_error.empty()) {
+                        capture->captured_error.append("; ");
+                    }
+                    capture->captured_error.append(msg);
+                }
+            }
+            
+            // Also call original callback to preserve logging
+            if (capture->original_callback) {
+                capture->original_callback(level, text, capture->original_user_data);
+            }
+        }, &error_capture_data);
+        
        // load the model
-        if (!ctx_server.load_model(params)) {
-            result->set_message("Failed loading model");
+        bool load_success = ctx_server.load_model(params);
+        
+        // Restore original log callback
+        llama_log_set(error_capture_data.original_callback, error_capture_data.original_user_data);
+        
+        if (!load_success) {
+            std::string error_msg = "Failed to load model: " + params.model.path;
+            if (!params.mmproj.path.empty()) {
+                error_msg += " (with mmproj: " + params.mmproj.path + ")";
+            }
+            if (params.has_speculative() && !params.speculative.model.path.empty()) {
+                error_msg += " (with draft model: " + params.speculative.model.path + ")";
+            }
+            
+            // Add captured error details if available
+            {
+                std::lock_guard<std::mutex> lock(error_capture_data.error_mutex);
+                if (!error_capture_data.captured_error.empty()) {
+                    error_msg += ". Error: " + error_capture_data.captured_error;
+                } else {
+                    error_msg += ". Model file may not exist or be invalid.";
+                }
+            }
+            
+            result->set_message(error_msg);
            result->set_success(false);
-            return Status::CANCELLED;
+            return grpc::Status(grpc::StatusCode::INTERNAL, error_msg);
        }

        // Process grammar triggers now that vocab is available
@@ -1494,7 +1573,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Predict(ServerContext* context, const backend::PredictOptions* request, backend::Reply* reply) {
+    grpc::Status Predict(ServerContext* context, const backend::PredictOptions* request, backend::Reply* reply) override {
         if (params_base.model.path.empty()) {
             return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
         }
@@ -2165,7 +2244,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Embedding(ServerContext* context, const backend::PredictOptions* request, backend::EmbeddingResult* embeddingResult) {
+    grpc::Status Embedding(ServerContext* context, const backend::PredictOptions* request, backend::EmbeddingResult* embeddingResult) override {
        if (params_base.model.path.empty()) {
            return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
        }
@@ -2260,7 +2339,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Rerank(ServerContext* context, const backend::RerankRequest* request, backend::RerankResult* rerankResult) {
+    grpc::Status Rerank(ServerContext* context, const backend::RerankRequest* request, backend::RerankResult* rerankResult) override {
        if (!params_base.embedding || params_base.pooling_type != LLAMA_POOLING_TYPE_RANK) {
            return grpc::Status(grpc::StatusCode::UNIMPLEMENTED, "This server does not support reranking. Start it with `--reranking` and without `--embedding`");
        }
@@ -2346,7 +2425,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status TokenizeString(ServerContext* /*context*/, const backend::PredictOptions* request, backend::TokenizationResponse* response) {
+    grpc::Status TokenizeString(ServerContext* /*context*/, const backend::PredictOptions* request, backend::TokenizationResponse* response) override {
        if (params_base.model.path.empty()) {
            return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
        }
@@ -2369,7 +2448,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status GetMetrics(ServerContext* /*context*/, const backend::MetricsRequest* /*request*/, backend::MetricsResponse* response) {
+    grpc::Status GetMetrics(ServerContext* /*context*/, const backend::MetricsRequest* /*request*/, backend::MetricsResponse* response) override {

 // request slots data using task queue
        auto rd = ctx_server.get_response_reader();
--- a/backend/cpp/llama-cpp/package.sh
+++ b/backend/cpp/llama-cpp/package.sh
@@ -6,6 +6,7 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
+REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -37,6 +38,15 @@ else
    exit 1
 fi

+# Package GPU libraries based on BUILD_TYPE
+# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
+GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
+if [ -f "$GPU_LIB_SCRIPT" ]; then
+    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
+    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
+    package_gpu_libs
+fi
+
 echo "Packaging completed successfully" 
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/go/stablediffusion-ggml/Makefile
+++ b/backend/go/stablediffusion-ggml/Makefile
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)

 # stablediffusion.cpp (ggml)
 STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
-STABLEDIFFUSION_GGML_VERSION?=4ff2c8c74bd17c2cfffe3a01be77743fb3efba2f
+STABLEDIFFUSION_GGML_VERSION?=9565c7f6bd5fcff124c589147b2621244f2c4aa1

 CMAKE_ARGS+=-DGGML_MAX_NAME=128

@@ -28,7 +28,12 @@ else ifeq ($(BUILD_TYPE),clblas)
 	CMAKE_ARGS+=-DGGML_CLBLAST=ON -DCLBlast_DIR=/some/path
 # If it's hipblas we do have also to set CC=/opt/rocm/llvm/bin/clang CXX=/opt/rocm/llvm/bin/clang++
 else ifeq ($(BUILD_TYPE),hipblas)
-	CMAKE_ARGS+=-DSD_HIPBLAS=ON -DGGML_HIPBLAS=ON
+	ROCM_HOME ?= /opt/rocm
+	ROCM_PATH ?= /opt/rocm
+	export CXX=$(ROCM_HOME)/llvm/bin/clang++
+	export CC=$(ROCM_HOME)/llvm/bin/clang
+	AMDGPU_TARGETS?=gfx803,gfx900,gfx906,gfx908,gfx90a,gfx942,gfx1010,gfx1030,gfx1032,gfx1100,gfx1101,gfx1102,gfx1200,gfx1201
+	CMAKE_ARGS+=-DSD_HIPBLAS=ON -DGGML_HIPBLAS=ON -DAMDGPU_TARGETS=$(AMDGPU_TARGETS)
 else ifeq ($(BUILD_TYPE),vulkan)
 	CMAKE_ARGS+=-DSD_VULKAN=ON -DGGML_VULKAN=ON
 else ifeq ($(OS),Darwin)
--- a/backend/go/stablediffusion-ggml/package.sh
+++ b/backend/go/stablediffusion-ggml/package.sh
@@ -6,6 +6,7 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
+REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -50,6 +51,15 @@ else
    exit 1
 fi

+# Package GPU libraries based on BUILD_TYPE
+# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
+GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
+if [ -f "$GPU_LIB_SCRIPT" ]; then
+    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
+    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
+    package_gpu_libs
+fi
+
 echo "Packaging completed successfully"
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/go/whisper/Makefile
+++ b/backend/go/whisper/Makefile
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)

 # whisper.cpp version
 WHISPER_REPO?=https://github.com/ggml-org/whisper.cpp
-WHISPER_CPP_VERSION?=e9898ddfb908ffaa7026c66852a023889a5a7202
+WHISPER_CPP_VERSION?=f53dc74843e97f19f94a79241357f74ad5b691a6
 SO_TARGET?=libgowhisper.so

 CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
--- a/backend/go/whisper/package.sh
+++ b/backend/go/whisper/package.sh
@@ -6,6 +6,7 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
+REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -50,6 +51,15 @@ else
    exit 1
 fi

+# Package GPU libraries based on BUILD_TYPE
+# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
+GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
+if [ -f "$GPU_LIB_SCRIPT" ]; then
+    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
+    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
+    package_gpu_libs
+fi
+
 echo "Packaging completed successfully"
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/index.yaml
+++ b/backend/index.yaml
@@ -275,6 +275,24 @@
    amd: "rocm-faster-whisper"
    nvidia-cuda-13: "cuda13-faster-whisper"
    nvidia-cuda-12: "cuda12-faster-whisper"
+- &moonshine
+  description: |
+    Moonshine is a fast, accurate, and efficient speech-to-text transcription model using ONNX Runtime.
+    It provides real-time transcription capabilities with support for multiple model sizes and GPU acceleration.
+  urls:
+    - https://github.com/moonshine-ai/moonshine
+  tags:
+    - speech-to-text
+    - transcription
+    - ONNX
+  license: MIT
+  name: "moonshine"
+  alias: "moonshine"
+  capabilities:
+    nvidia: "cuda12-moonshine"
+    default: "cpu-moonshine"
+    nvidia-cuda-13: "cuda13-moonshine"
+    nvidia-cuda-12: "cuda12-moonshine"
 - &kokoro
  icon: https://avatars.githubusercontent.com/u/166769057?v=4
  description: |
@@ -410,6 +428,28 @@
    nvidia-l4t-cuda-12: "nvidia-l4t-vibevoice"
    nvidia-l4t-cuda-13: "cuda13-nvidia-l4t-arm64-vibevoice"
  icon: https://avatars.githubusercontent.com/u/6154722?s=200&v=4
+- &pocket-tts
+  urls:
+    - https://github.com/kyutai-labs/pocket-tts
+  description: |
+    Pocket TTS is a lightweight text-to-speech model designed to run efficiently on CPUs.
+  tags:
+    - text-to-speech
+    - TTS
+  license: mit
+  name: "pocket-tts"
+  alias: "pocket-tts"
+  capabilities:
+    nvidia: "cuda12-pocket-tts"
+    intel: "intel-pocket-tts"
+    amd: "rocm-pocket-tts"
+    nvidia-l4t: "nvidia-l4t-pocket-tts"
+    default: "cpu-pocket-tts"
+    nvidia-cuda-13: "cuda13-pocket-tts"
+    nvidia-cuda-12: "cuda12-pocket-tts"
+    nvidia-l4t-cuda-12: "nvidia-l4t-pocket-tts"
+    nvidia-l4t-cuda-13: "cuda13-nvidia-l4t-arm64-pocket-tts"
+  icon: https://avatars.githubusercontent.com/u/6154722?s=200&v=4
 - &piper
  name: "piper"
  uri: "quay.io/go-skynet/local-ai-backends:latest-piper"
@@ -497,18 +537,14 @@
    default: "cpu-neutts"
    nvidia: "cuda12-neutts"
    amd: "rocm-neutts"
-    nvidia-l4t: "nvidia-l4t-neutts"
    nvidia-cuda-12: "cuda12-neutts"
-    nvidia-l4t-cuda-12: "nvidia-l4t-arm64-neutts"
 - !!merge <<: *neutts
  name: "neutts-development"
  capabilities:
    default: "cpu-neutts-development"
    nvidia: "cuda12-neutts-development"
    amd: "rocm-neutts-development"
-    nvidia-l4t: "nvidia-l4t-neutts-development"
    nvidia-cuda-12: "cuda12-neutts-development"
-    nvidia-l4t-cuda-12: "nvidia-l4t-arm64-neutts-development"
 - !!merge <<: *llamacpp
  name: "llama-cpp-development"
  capabilities:
@@ -538,11 +574,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-neutts"
  mirrors:
    - localai/localai-backends:latest-gpu-rocm-hipblas-neutts
- !!merge <<: *neutts
-  name: "nvidia-l4t-arm64-neutts"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-nvidia-l4t-arm64-neutts"
-  mirrors:
-    - localai/localai-backends:latest-nvidia-l4t-arm64-neutts
 - !!merge <<: *neutts
  name: "cpu-neutts-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-neutts"
@@ -558,11 +589,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-rocm-hipblas-neutts"
  mirrors:
    - localai/localai-backends:master-gpu-rocm-hipblas-neutts
- !!merge <<: *neutts
-  name: "nvidia-l4t-arm64-neutts-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-arm64-neutts"
-  mirrors:
-    - localai/localai-backends:master-nvidia-l4t-arm64-neutts
 - !!merge <<: *mlx
  name: "mlx-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-metal-darwin-arm64-mlx"
@@ -634,11 +660,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-llama-cpp"
  mirrors:
    - localai/localai-backends:master-cpu-llama-cpp
- !!merge <<: *llamacpp
-  name: "cuda11-llama-cpp"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-llama-cpp"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-llama-cpp
 - !!merge <<: *llamacpp
  name: "cuda12-llama-cpp"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-llama-cpp"
@@ -679,11 +700,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-metal-darwin-arm64-llama-cpp"
  mirrors:
    - localai/localai-backends:master-metal-darwin-arm64-llama-cpp
- !!merge <<: *llamacpp
-  name: "cuda11-llama-cpp-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-llama-cpp"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-llama-cpp
 - !!merge <<: *llamacpp
  name: "cuda12-llama-cpp-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-llama-cpp"
@@ -755,11 +771,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-whisper"
  mirrors:
    - localai/localai-backends:master-cpu-whisper
- !!merge <<: *whispercpp
-  name: "cuda11-whisper"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-whisper"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-whisper
 - !!merge <<: *whispercpp
  name: "cuda12-whisper"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-whisper"
@@ -800,11 +811,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-metal-darwin-arm64-whisper"
  mirrors:
    - localai/localai-backends:master-metal-darwin-arm64-whisper
- !!merge <<: *whispercpp
-  name: "cuda11-whisper-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-whisper"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-whisper
 - !!merge <<: *whispercpp
  name: "cuda12-whisper-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-whisper"
@@ -879,11 +885,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-sycl-f16-stablediffusion-ggml"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-sycl-f16-stablediffusion-ggml
- !!merge <<: *stablediffusionggml
-  name: "cuda11-stablediffusion-ggml"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-stablediffusion-ggml"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-stablediffusion-ggml
 - !!merge <<: *stablediffusionggml
  name: "cuda12-stablediffusion-ggml-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-stablediffusion-ggml"
@@ -899,11 +900,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-intel-sycl-f16-stablediffusion-ggml"
  mirrors:
    - localai/localai-backends:master-gpu-intel-sycl-f16-stablediffusion-ggml
- !!merge <<: *stablediffusionggml
-  name: "cuda11-stablediffusion-ggml-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-stablediffusion-ggml"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-stablediffusion-ggml
 - !!merge <<: *stablediffusionggml
  name: "nvidia-l4t-arm64-stablediffusion-ggml-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-arm64-stablediffusion-ggml"
@@ -1054,11 +1050,6 @@
    intel: "intel-rerankers-development"
    amd: "rocm-rerankers-development"
    nvidia-cuda-13: "cuda13-rerankers-development"
- !!merge <<: *rerankers
-  name: "cuda11-rerankers"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-rerankers"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-rerankers
 - !!merge <<: *rerankers
  name: "cuda12-rerankers"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-rerankers"
@@ -1074,11 +1065,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-rerankers"
  mirrors:
    - localai/localai-backends:latest-gpu-rocm-hipblas-rerankers
- !!merge <<: *rerankers
-  name: "cuda11-rerankers-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-rerankers"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-rerankers
 - !!merge <<: *rerankers
  name: "cuda12-rerankers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-rerankers"
@@ -1127,16 +1113,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-transformers"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-transformers
- !!merge <<: *transformers
-  name: "cuda11-transformers-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-transformers"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-transformers
- !!merge <<: *transformers
-  name: "cuda11-transformers"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-transformers"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-transformers
 - !!merge <<: *transformers
  name: "cuda12-transformers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-transformers"
@@ -1213,21 +1189,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-diffusers"
  mirrors:
    - localai/localai-backends:latest-gpu-rocm-hipblas-diffusers
- !!merge <<: *diffusers
-  name: "cuda11-diffusers"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-diffusers"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-diffusers
 - !!merge <<: *diffusers
  name: "intel-diffusers"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-diffusers"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-diffusers
- !!merge <<: *diffusers
-  name: "cuda11-diffusers-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-diffusers"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-diffusers
 - !!merge <<: *diffusers
  name: "cuda12-diffusers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-diffusers"
@@ -1269,21 +1235,11 @@
  capabilities:
    nvidia: "cuda12-exllama2-development"
    intel: "intel-exllama2-development"
- !!merge <<: *exllama2
-  name: "cuda11-exllama2"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-exllama2"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-exllama2
 - !!merge <<: *exllama2
  name: "cuda12-exllama2"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-exllama2"
  mirrors:
    - localai/localai-backends:latest-gpu-nvidia-cuda-12-exllama2
- !!merge <<: *exllama2
-  name: "cuda11-exllama2-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-exllama2"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-exllama2
 - !!merge <<: *exllama2
  name: "cuda12-exllama2-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-exllama2"
@@ -1297,11 +1253,6 @@
    intel: "intel-kokoro-development"
    amd: "rocm-kokoro-development"
    nvidia-l4t: "nvidia-l4t-kokoro-development"
- !!merge <<: *kokoro
-  name: "cuda11-kokoro-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-kokoro"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-kokoro
 - !!merge <<: *kokoro
  name: "cuda12-kokoro-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-kokoro"
@@ -1332,11 +1283,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-kokoro"
  mirrors:
    - localai/localai-backends:master-nvidia-l4t-kokoro
- !!merge <<: *kokoro
-  name: "cuda11-kokoro"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-kokoro"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-kokoro
 - !!merge <<: *kokoro
  name: "cuda12-kokoro"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-kokoro"
@@ -1365,11 +1311,6 @@
    intel: "intel-faster-whisper-development"
    amd: "rocm-faster-whisper-development"
    nvidia-cuda-13: "cuda13-faster-whisper-development"
- !!merge <<: *faster-whisper
-  name: "cuda11-faster-whisper"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-faster-whisper"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-faster-whisper
 - !!merge <<: *faster-whisper
  name: "cuda12-faster-whisper-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-faster-whisper"
@@ -1400,6 +1341,44 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-13-faster-whisper"
  mirrors:
    - localai/localai-backends:master-gpu-nvidia-cuda-13-faster-whisper
+## moonshine
+- !!merge <<: *moonshine
+  name: "moonshine-development"
+  capabilities:
+    nvidia: "cuda12-moonshine-development"
+    default: "cpu-moonshine-development"
+    nvidia-cuda-13: "cuda13-moonshine-development"
+    nvidia-cuda-12: "cuda12-moonshine-development"
+- !!merge <<: *moonshine
+  name: "cpu-moonshine"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-cpu-moonshine"
+  mirrors:
+    - localai/localai-backends:latest-cpu-moonshine
+- !!merge <<: *moonshine
+  name: "cpu-moonshine-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-moonshine"
+  mirrors:
+    - localai/localai-backends:master-cpu-moonshine
+- !!merge <<: *moonshine
+  name: "cuda12-moonshine"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-moonshine"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-12-moonshine
+- !!merge <<: *moonshine
+  name: "cuda12-moonshine-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-moonshine"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-12-moonshine
+- !!merge <<: *moonshine
+  name: "cuda13-moonshine"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-13-moonshine"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-13-moonshine
+- !!merge <<: *moonshine
+  name: "cuda13-moonshine-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-13-moonshine"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-13-moonshine
 ## coqui

 - !!merge <<: *coqui
@@ -1408,21 +1387,11 @@
    nvidia: "cuda12-coqui-development"
    intel: "intel-coqui-development"
    amd: "rocm-coqui-development"
- !!merge <<: *coqui
-  name: "cuda11-coqui"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-coqui"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-coqui
 - !!merge <<: *coqui
  name: "cuda12-coqui"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-coqui"
  mirrors:
    - localai/localai-backends:latest-gpu-nvidia-cuda-12-coqui
- !!merge <<: *coqui
-  name: "cuda11-coqui-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-coqui"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-coqui
 - !!merge <<: *coqui
  name: "cuda12-coqui-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-coqui"
@@ -1455,16 +1424,6 @@
    nvidia: "cuda12-bark-development"
    intel: "intel-bark-development"
    amd: "rocm-bark-development"
- !!merge <<: *bark
-  name: "cuda11-bark-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-bark"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-bark
- !!merge <<: *bark
-  name: "cuda11-bark"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-bark"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-bark
 - !!merge <<: *bark
  name: "rocm-bark-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-rocm-hipblas-bark"
@@ -1546,16 +1505,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-chatterbox"
  mirrors:
    - localai/localai-backends:master-gpu-nvidia-cuda-12-chatterbox
- !!merge <<: *chatterbox
-  name: "cuda11-chatterbox"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-chatterbox"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-11-chatterbox
- !!merge <<: *chatterbox
-  name: "cuda11-chatterbox-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-chatterbox"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-11-chatterbox
 - !!merge <<: *chatterbox
  name: "cuda12-chatterbox"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-chatterbox"
@@ -1664,3 +1613,86 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-cuda-13-arm64-vibevoice"
  mirrors:
    - localai/localai-backends:master-nvidia-l4t-cuda-13-arm64-vibevoice
+## pocket-tts
+- !!merge <<: *pocket-tts
+  name: "pocket-tts-development"
+  capabilities:
+    nvidia: "cuda12-pocket-tts-development"
+    intel: "intel-pocket-tts-development"
+    amd: "rocm-pocket-tts-development"
+    nvidia-l4t: "nvidia-l4t-pocket-tts-development"
+    default: "cpu-pocket-tts-development"
+    nvidia-cuda-13: "cuda13-pocket-tts-development"
+    nvidia-cuda-12: "cuda12-pocket-tts-development"
+    nvidia-l4t-cuda-12: "nvidia-l4t-pocket-tts-development"
+    nvidia-l4t-cuda-13: "cuda13-nvidia-l4t-arm64-pocket-tts-development"
+- !!merge <<: *pocket-tts
+  name: "cpu-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-cpu-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-cpu-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cpu-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-cpu-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda12-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-12-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda12-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-12-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda13-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-13-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-13-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda13-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-13-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-13-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "intel-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-gpu-intel-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "intel-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-intel-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-gpu-intel-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "rocm-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-gpu-rocm-hipblas-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "rocm-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-rocm-hipblas-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-gpu-rocm-hipblas-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "nvidia-l4t-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-nvidia-l4t-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-nvidia-l4t-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "nvidia-l4t-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-nvidia-l4t-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda13-nvidia-l4t-arm64-pocket-tts"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-nvidia-l4t-cuda-13-arm64-pocket-tts"
+  mirrors:
+    - localai/localai-backends:latest-nvidia-l4t-cuda-13-arm64-pocket-tts
+- !!merge <<: *pocket-tts
+  name: "cuda13-nvidia-l4t-arm64-pocket-tts-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-cuda-13-arm64-pocket-tts"
+  mirrors:
+    - localai/localai-backends:master-nvidia-l4t-cuda-13-arm64-pocket-tts
--- a/backend/python/README.md
+++ b/backend/python/README.md
@@ -85,7 +85,7 @@ runUnittests
 The build system automatically detects and configures for different hardware:

 - **CPU** - Standard CPU-only builds
- **CUDA** - NVIDIA GPU acceleration (supports CUDA 11/12)
+- **CUDA** - NVIDIA GPU acceleration (supports CUDA 12/13)
 - **Intel** - Intel XPU/GPU optimization
 - **MLX** - Apple Silicon (M1/M2/M3) optimization
 - **HIP** - AMD GPU acceleration
@@ -95,8 +95,8 @@ The build system automatically detects and configures for different hardware:
 Backends can specify hardware-specific dependencies:
 - `requirements.txt` - Base requirements
 - `requirements-cpu.txt` - CPU-specific packages
- `requirements-cublas11.txt` - CUDA 11 packages
 - `requirements-cublas12.txt` - CUDA 12 packages
+- `requirements-cublas13.txt` - CUDA 13 packages
 - `requirements-intel.txt` - Intel-optimized packages
 - `requirements-mps.txt` - Apple Silicon packages

--- a/backend/python/bark/requirements-cublas11.txt
+++ b/backend/python/bark/requirements-cublas11.txt
@@ -1,5 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.4.1+cu118
-torchaudio==2.4.1+cu118
-transformers
-accelerate
--- a/backend/python/bark/requirements-hipblas.txt
+++ b/backend/python/bark/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
-torch==2.4.1+rocm6.0
-torchaudio==2.4.1+rocm6.0
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
+torchaudio==2.8.0+rocm6.4
 transformers
 accelerate
--- a/backend/python/chatterbox/install.sh
+++ b/backend/python/chatterbox/install.sh
@@ -17,4 +17,9 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
 fi
 EXTRA_PIP_INSTALL_FLAGS+=" --no-build-isolation"

+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
+
 installRequirements
--- a/backend/python/chatterbox/requirements-cublas11.txt
+++ b/backend/python/chatterbox/requirements-cublas11.txt
@@ -1,8 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.6.0+cu118
-torchaudio==2.6.0+cu118
-transformers==4.46.3
-numpy>=1.24.0,<1.26.0
-# https://github.com/mudler/LocalAI/pull/6240#issuecomment-3329518289
-chatterbox-tts@git+https://git@github.com/mudler/chatterbox.git@faster
-accelerate
--- a/backend/python/chatterbox/requirements-hipblas.txt
+++ b/backend/python/chatterbox/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
-torch==2.6.0+rocm6.1
-torchaudio==2.6.0+rocm6.1
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.9.1+rocm6.4
+torchaudio==2.9.1+rocm6.4
 transformers
 numpy>=1.24.0,<1.26.0
 # https://github.com/mudler/LocalAI/pull/6240#issuecomment-3329518289
--- a/backend/python/chatterbox/requirements-install.txt
+++ b/backend/python/chatterbox/requirements-install.txt
@@ -0,0 +1,5 @@
+# Build dependencies needed for packages installed from source (e.g., git dependencies)
+# When using --no-build-isolation, these must be installed in the venv first
+wheel
+setuptools
+packaging
--- a/backend/python/common/libbackend.sh
+++ b/backend/python/common/libbackend.sh
@@ -1,7 +1,7 @@
 #!/usr/bin/env bash
 set -euo pipefail

-# 
+#
 # use the library by adding the following line to a script:
 # source $(dirname $0)/../common/libbackend.sh
 #
@@ -206,8 +206,8 @@ function init() {

 # getBuildProfile will inspect the system to determine which build profile is appropriate:
 # returns one of the following:
-# - cublas11
 # - cublas12
+# - cublas13
 # - hipblas
 # - intel
 function getBuildProfile() {
@@ -392,7 +392,7 @@ function runProtogen() {
 #  - requirements-${BUILD_TYPE}.txt
 #  - requirements-${BUILD_PROFILE}.txt
 #
-# BUILD_PROFILE is a more specific version of BUILD_TYPE, ex: cuda-11 or cuda-12
+# BUILD_PROFILE is a more specific version of BUILD_TYPE, ex: cuda-12 or cuda-13
 # it can also include some options that we do not have BUILD_TYPES for, ex: intel
 #
 # NOTE: for BUILD_PROFILE==intel, this function does NOT automatically use the Intel python package index.
@@ -465,6 +465,14 @@ function startBackend() {
    if [ "x${PORTABLE_PYTHON}" == "xtrue" ] || [ -x "$(_portable_python)" ]; then
        _makeVenvPortable --update-pyvenv-cfg
    fi
+
+    # Set up GPU library paths if a lib directory exists
+    # This allows backends to include their own GPU libraries (CUDA, ROCm, etc.)
+    if [ -d "${EDIR}/lib" ]; then
+        export LD_LIBRARY_PATH="${EDIR}/lib:${LD_LIBRARY_PATH:-}"
+        echo "Added ${EDIR}/lib to LD_LIBRARY_PATH for GPU libraries"
+    fi
+
    if [ ! -z "${BACKEND_FILE:-}" ]; then
        exec "${EDIR}/venv/bin/python" "${BACKEND_FILE}" "$@"
    elif [ -e "${MY_DIR}/server.py" ]; then
--- a/backend/python/common/template/requirements-hipblas.txt
+++ b/backend/python/common/template/requirements-hipblas.txt
@@ -1,2 +1,2 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
 torch
--- a/backend/python/coqui/requirements-cublas11.txt
+++ b/backend/python/coqui/requirements-cublas11.txt
@@ -1,6 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.4.1+cu118
-torchaudio==2.4.1+cu118
-transformers==4.48.3
-accelerate
-coqui-tts
--- a/backend/python/coqui/requirements-hipblas.txt
+++ b/backend/python/coqui/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
-torch==2.4.1+rocm6.0
-torchaudio==2.4.1+rocm6.0
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
+torchaudio==2.8.0+rocm6.4
 transformers==4.48.3
 accelerate
 coqui-tts
--- a/backend/python/diffusers/backend.py
+++ b/backend/python/diffusers/backend.py
@@ -41,6 +41,14 @@ from optimum.quanto import freeze, qfloat8, quantize
 from transformers import T5EncoderModel
 from safetensors.torch import load_file

+# Import LTX-2 specific utilities
+try:
+    from diffusers.pipelines.ltx2.export_utils import encode_video as ltx2_encode_video
+    LTX2_AVAILABLE = True
+except ImportError:
+    LTX2_AVAILABLE = False
+    ltx2_encode_video = None
+
 _ONE_DAY_IN_SECONDS = 60 * 60 * 24
 COMPEL = os.environ.get("COMPEL", "0") == "1"
 XPU = os.environ.get("XPU", "0") == "1"
@@ -290,6 +298,20 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
                pipe.enable_model_cpu_offload()
            return pipe

+        # LTX2ImageToVideoPipeline - needs img2vid flag, CPU offload, and special handling
+        if pipeline_type == "LTX2ImageToVideoPipeline":
+            self.img2vid = True
+            self.ltx2_pipeline = True
+            pipe = load_diffusers_pipeline(
+                class_name="LTX2ImageToVideoPipeline",
+                model_id=request.Model,
+                torch_dtype=torchType,
+                variant=variant
+            )
+            if not DISABLE_CPU_OFFLOAD:
+                pipe.enable_model_cpu_offload()
+            return pipe
+
        # ================================================================
        # Dynamic pipeline loading - the default path for most pipelines
        # Uses the dynamic loader to instantiate any pipeline by class name
@@ -404,6 +426,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
            fromSingleFile = request.Model.startswith("http") or request.Model.startswith("/") or local
            self.img2vid = False
            self.txt2vid = False
+            self.ltx2_pipeline = False

            # Load pipeline using dynamic loader
            # Special cases that require custom initialization are handled first
@@ -686,7 +709,44 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
            print(f"Generating video with {kwargs=}", file=sys.stderr)

            # Generate video frames based on pipeline type
-            if self.PipelineType == "WanPipeline":
+            if self.ltx2_pipeline or self.PipelineType == "LTX2ImageToVideoPipeline":
+                # LTX-2 image-to-video generation with audio
+                if not LTX2_AVAILABLE:
+                    return backend_pb2.Result(success=False, message="LTX-2 pipeline requires diffusers.pipelines.ltx2.export_utils")
+                
+                # LTX-2 uses 'image' parameter instead of 'start_image'
+                if request.start_image:
+                    image = load_image(request.start_image)
+                    kwargs["image"] = image
+                    # Remove start_image if it was added
+                    kwargs.pop("start_image", None)
+                
+                # LTX-2 uses 'frame_rate' instead of 'fps'
+                frame_rate = float(fps)
+                kwargs["frame_rate"] = frame_rate
+                
+                # LTX-2 requires output_type="np" and return_dict=False
+                kwargs["output_type"] = "np"
+                kwargs["return_dict"] = False
+                
+                # Generate video and audio
+                video, audio = self.pipe(**kwargs)
+                
+                # Convert video to uint8 format
+                video = (video * 255).round().astype("uint8")
+                video = torch.from_numpy(video)
+                
+                # Use LTX-2's encode_video function which handles audio
+                ltx2_encode_video(
+                    video[0],
+                    fps=frame_rate,
+                    audio=audio[0].float().cpu(),
+                    audio_sample_rate=self.pipe.vocoder.config.output_sampling_rate,
+                    output_path=request.dst,
+                )
+                
+                return backend_pb2.Result(message="Video generated successfully", success=True)
+            elif self.PipelineType == "WanPipeline":
                # WAN2.2 text-to-video generation
                output = self.pipe(**kwargs)
                frames = output.frames[0]  # WAN2.2 returns frames in this format
@@ -727,7 +787,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
            else:
                return backend_pb2.Result(success=False, message=f"Pipeline {self.PipelineType} does not support video generation")

-            # Export video
+            # Export video (for non-LTX-2 pipelines)
            export_to_video(frames, request.dst, fps=fps)
            
            return backend_pb2.Result(message="Video generated successfully", success=True)
--- a/backend/python/diffusers/install.sh
+++ b/backend/python/diffusers/install.sh
@@ -16,8 +16,12 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
    EXTRA_PIP_INSTALL_FLAGS+=" --upgrade --index-strategy=unsafe-first-match"
 fi

+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
 # Use python 3.12 for l4t
-if [ "x${BUILD_PROFILE}" == "xl4t12" ] || [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
+if [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
  PYTHON_VERSION="3.12"
  PYTHON_PATCH="12"
  PY_STANDALONE_TAG="20251120"
--- a/backend/python/diffusers/requirements-cublas11.txt
+++ b/backend/python/diffusers/requirements-cublas11.txt
@@ -1,12 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-git+https://github.com/huggingface/diffusers
-opencv-python
-transformers
-torchvision==0.22.1
-accelerate
-compel
-peft
-sentencepiece
-torch==2.7.1
-optimum-quanto
-ftfy
--- a/backend/python/diffusers/requirements-hipblas.txt
+++ b/backend/python/diffusers/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.3
-torch==2.7.1+rocm6.3
-torchvision==0.22.1+rocm6.3
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
+torchvision==0.23.0+rocm6.4
 git+https://github.com/huggingface/diffusers
 opencv-python
 transformers
--- a/backend/python/exllama2/requirements-cublas11.txt
+++ b/backend/python/exllama2/requirements-cublas11.txt
@@ -1,4 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.4.1+cu118
-transformers
-accelerate
--- a/backend/python/faster-whisper/requirements-cublas11.txt
+++ b/backend/python/faster-whisper/requirements-cublas11.txt
@@ -1,9 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.4.1+cu118
-faster-whisper
-opencv-python
-accelerate
-compel
-peft
-sentencepiece
-optimum-quanto
--- a/backend/python/faster-whisper/requirements-hipblas.txt
+++ b/backend/python/faster-whisper/requirements-hipblas.txt
@@ -1,3 +1,3 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
 torch
 faster-whisper
--- a/backend/python/kokoro/install.sh
+++ b/backend/python/kokoro/install.sh
@@ -16,4 +16,8 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
    EXTRA_PIP_INSTALL_FLAGS+=" --upgrade --index-strategy=unsafe-first-match"
 fi

+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
 installRequirements
--- a/backend/python/kokoro/requirements-cublas11.txt
+++ b/backend/python/kokoro/requirements-cublas11.txt
@@ -1,7 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.7.1+cu118
-torchaudio==2.7.1+cu118
-transformers
-accelerate
-kokoro
-soundfile
--- a/backend/python/kokoro/requirements-hipblas.txt
+++ b/backend/python/kokoro/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.3
-torch==2.7.1+rocm6.3
-torchaudio==2.7.1+rocm6.3
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
+torchaudio==2.8.0+rocm6.4
 transformers
 accelerate
 kokoro
--- a/backend/python/moonshine/Makefile
+++ b/backend/python/moonshine/Makefile
@@ -0,0 +1,16 @@
+.DEFAULT_GOAL := install
+
+.PHONY: install
+install:
+	bash install.sh
+
+.PHONY: protogen-clean
+protogen-clean:
+	$(RM) backend_pb2_grpc.py backend_pb2.py
+
+.PHONY: clean
+clean: protogen-clean
+	rm -rf venv __pycache__
+
+test: install
+	bash test.sh
--- a/backend/python/moonshine/backend.py
+++ b/backend/python/moonshine/backend.py
@@ -0,0 +1,113 @@
+#!/usr/bin/env python3
+"""
+This is an extra gRPC server of LocalAI for Moonshine transcription
+"""
+from concurrent import futures
+import time
+import argparse
+import signal
+import sys
+import os
+import backend_pb2
+import backend_pb2_grpc
+import moonshine_onnx
+
+import grpc
+
+
+_ONE_DAY_IN_SECONDS = 60 * 60 * 24
+
+# If MAX_WORKERS are specified in the environment use it, otherwise default to 1
+MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
+
+# Implement the BackendServicer class with the service methods
+class BackendServicer(backend_pb2_grpc.BackendServicer):
+    """
+    BackendServicer is the class that implements the gRPC service
+    """
+    def Health(self, request, context):
+        return backend_pb2.Reply(message=bytes("OK", 'utf-8'))
+    
+    def LoadModel(self, request, context):
+        try:
+            print("Preparing models, please wait", file=sys.stderr)
+            # Store the model name for use in transcription
+            # Model name format: e.g., "moonshine/tiny"
+            self.model_name = request.Model
+            print(f"Model name set to: {self.model_name}", file=sys.stderr)
+        except Exception as err:
+            return backend_pb2.Result(success=False, message=f"Unexpected {err=}, {type(err)=}")
+        return backend_pb2.Result(message="Model loaded successfully", success=True)
+
+    def AudioTranscription(self, request, context):
+        resultSegments = []
+        text = ""
+        try:
+            # moonshine_onnx.transcribe returns a list of strings
+            transcriptions = moonshine_onnx.transcribe(request.dst, self.model_name)
+            
+            # Combine all transcriptions into a single text
+            if isinstance(transcriptions, list):
+                text = " ".join(transcriptions)
+                # Create segments for each transcription in the list
+                for id, trans in enumerate(transcriptions):
+                    # Since moonshine doesn't provide timing info, we'll create a single segment
+                    # with id and text, using approximate timing
+                    resultSegments.append(backend_pb2.TranscriptSegment(
+                        id=id, 
+                        start=0, 
+                        end=0, 
+                        text=trans
+                    ))
+            else:
+                # Handle case where it's not a list (shouldn't happen, but be safe)
+                text = str(transcriptions)
+                resultSegments.append(backend_pb2.TranscriptSegment(
+                    id=0,
+                    start=0,
+                    end=0,
+                    text=text
+                ))
+        except Exception as err:
+            print(f"Unexpected {err=}, {type(err)=}", file=sys.stderr)
+            return backend_pb2.TranscriptResult(segments=[], text="")
+
+        return backend_pb2.TranscriptResult(segments=resultSegments, text=text)
+
+def serve(address):
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=MAX_WORKERS),
+        options=[
+            ('grpc.max_message_length', 50 * 1024 * 1024),  # 50MB
+            ('grpc.max_send_message_length', 50 * 1024 * 1024),  # 50MB
+            ('grpc.max_receive_message_length', 50 * 1024 * 1024),  # 50MB
+        ])
+    backend_pb2_grpc.add_BackendServicer_to_server(BackendServicer(), server)
+    server.add_insecure_port(address)
+    server.start()
+    print("Server started. Listening on: " + address, file=sys.stderr)
+
+    # Define the signal handler function
+    def signal_handler(sig, frame):
+        print("Received termination signal. Shutting down...")
+        server.stop(0)
+        sys.exit(0)
+
+    # Set the signal handlers for SIGINT and SIGTERM
+    signal.signal(signal.SIGINT, signal_handler)
+    signal.signal(signal.SIGTERM, signal_handler)
+
+    try:
+        while True:
+            time.sleep(_ONE_DAY_IN_SECONDS)
+    except KeyboardInterrupt:
+        server.stop(0)
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="Run the gRPC server.")
+    parser.add_argument(
+        "--addr", default="localhost:50051", help="The address to bind the server to."
+    )
+    args = parser.parse_args()
+
+    serve(args.addr)
+
--- a/backend/python/moonshine/install.sh
+++ b/backend/python/moonshine/install.sh
@@ -0,0 +1,12 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+installRequirements
+
--- a/backend/python/moonshine/protogen.sh
+++ b/backend/python/moonshine/protogen.sh
@@ -0,0 +1,12 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+python3 -m grpc_tools.protoc -I../.. -I./ --python_out=. --grpc_python_out=. backend.proto
+
--- a/backend/python/moonshine/requirements.txt
+++ b/backend/python/moonshine/requirements.txt
@@ -0,0 +1,4 @@
+grpcio==1.71.0
+protobuf
+grpcio-tools
+useful-moonshine-onnx@git+https://git@github.com/moonshine-ai/moonshine.git#subdirectory=moonshine-onnx
--- a/backend/python/moonshine/run.sh
+++ b/backend/python/moonshine/run.sh
@@ -0,0 +1,10 @@
+#!/bin/bash
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+startBackend $@
+
--- a/backend/python/moonshine/test.py
+++ b/backend/python/moonshine/test.py
@@ -0,0 +1,139 @@
+"""
+A test script to test the gRPC service for Moonshine transcription
+"""
+import unittest
+import subprocess
+import time
+import os
+import tempfile
+import shutil
+import backend_pb2
+import backend_pb2_grpc
+
+import grpc
+
+
+class TestBackendServicer(unittest.TestCase):
+    """
+    TestBackendServicer is the class that tests the gRPC service
+    """
+    def setUp(self):
+        """
+        This method sets up the gRPC service by starting the server
+        """
+        self.service = subprocess.Popen(["python3", "backend.py", "--addr", "localhost:50051"])
+        time.sleep(10)
+
+    def tearDown(self) -> None:
+        """
+        This method tears down the gRPC service by terminating the server
+        """
+        self.service.terminate()
+        self.service.wait()
+
+    def test_server_startup(self):
+        """
+        This method tests if the server starts up successfully
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                response = stub.Health(backend_pb2.HealthMessage())
+                self.assertEqual(response.message, b'OK')
+        except Exception as err:
+            print(err)
+            self.fail("Server failed to start")
+        finally:
+            self.tearDown()
+
+    def test_load_model(self):
+        """
+        This method tests if the model is loaded successfully
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                response = stub.LoadModel(backend_pb2.ModelOptions(Model="moonshine/tiny"))
+                self.assertTrue(response.success)
+                self.assertEqual(response.message, "Model loaded successfully")
+        except Exception as err:
+            print(err)
+            self.fail("LoadModel service failed")
+        finally:
+            self.tearDown()
+
+    def test_audio_transcription(self):
+        """
+        This method tests if audio transcription works successfully
+        """
+        # Create a temporary directory for the audio file
+        temp_dir = tempfile.mkdtemp()
+        audio_file = os.path.join(temp_dir, 'audio.wav')
+        
+        try:
+            # Download the audio file to the temporary directory
+            print(f"Downloading audio file to {audio_file}...")
+            url = "https://cdn.openai.com/whisper/draft-20220913a/micro-machines.wav"
+            result = subprocess.run(
+                ["wget", "-q", url, "-O", audio_file],
+                capture_output=True,
+                text=True
+            )
+            if result.returncode != 0:
+                self.fail(f"Failed to download audio file: {result.stderr}")
+            
+            # Verify the file was downloaded
+            if not os.path.exists(audio_file):
+                self.fail(f"Audio file was not downloaded to {audio_file}")
+            
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                # Load the model first
+                load_response = stub.LoadModel(backend_pb2.ModelOptions(Model="moonshine/tiny"))
+                self.assertTrue(load_response.success)
+                
+                # Perform transcription
+                transcript_request = backend_pb2.TranscriptRequest(dst=audio_file)
+                transcript_response = stub.AudioTranscription(transcript_request)
+                
+                # Print the transcribed text for debugging
+                print(f"Transcribed text: {transcript_response.text}")
+                print(f"Number of segments: {len(transcript_response.segments)}")
+                
+                # Verify response structure
+                self.assertIsNotNone(transcript_response)
+                self.assertIsNotNone(transcript_response.text)
+                # Protobuf repeated fields return a sequence, not a list
+                self.assertIsNotNone(transcript_response.segments)
+                # Check if segments is iterable (has length)
+                self.assertGreaterEqual(len(transcript_response.segments), 0)
+                
+                # Verify the transcription contains the expected text
+                expected_text = "This is the micro machine man presenting the most midget miniature"
+                self.assertIn(
+                    expected_text.lower(),
+                    transcript_response.text.lower(),
+                    f"Expected text '{expected_text}' not found in transcription: '{transcript_response.text}'"
+                )
+                
+                # If we got segments, verify they have the expected structure
+                if len(transcript_response.segments) > 0:
+                    segment = transcript_response.segments[0]
+                    self.assertIsNotNone(segment.text)
+                    self.assertIsInstance(segment.id, int)
+                else:
+                    # Even if no segments, we should have text
+                    self.assertIsNotNone(transcript_response.text)
+                    self.assertGreater(len(transcript_response.text), 0)
+        except Exception as err:
+            print(err)
+            self.fail("AudioTranscription service failed")
+        finally:
+            self.tearDown()
+            # Clean up the temporary directory
+            if os.path.exists(temp_dir):
+                shutil.rmtree(temp_dir)
+
--- a/backend/python/moonshine/test.sh
+++ b/backend/python/moonshine/test.sh
@@ -0,0 +1,12 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+runUnittests
+
--- a/backend/python/neutts/install.sh
+++ b/backend/python/neutts/install.sh
@@ -26,6 +26,12 @@ fi

 EXTRA_PIP_INSTALL_FLAGS+=" --no-build-isolation"

+
+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
+
 git clone https://github.com/neuphonic/neutts-air neutts-air

 cp -rfv neutts-air/neuttsair ./
--- a/backend/python/neutts/requirements-hipblas.txt
+++ b/backend/python/neutts/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.3
-torch==2.8.0+rocm6.3
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
 transformers==4.56.1
 accelerate
 librosa==0.11.0
--- a/backend/python/pocket-tts/Makefile
+++ b/backend/python/pocket-tts/Makefile
@@ -0,0 +1,23 @@
+.PHONY: pocket-tts
+pocket-tts:
+	bash install.sh
+
+.PHONY: run
+run: pocket-tts
+	@echo "Running pocket-tts..."
+	bash run.sh
+	@echo "pocket-tts run."
+
+.PHONY: test
+test: pocket-tts
+	@echo "Testing pocket-tts..."
+	bash test.sh
+	@echo "pocket-tts tested."
+
+.PHONY: protogen-clean
+protogen-clean:
+	$(RM) backend_pb2_grpc.py backend_pb2.py
+
+.PHONY: clean
+clean: protogen-clean
+	rm -rf venv __pycache__
--- a/backend/python/pocket-tts/backend.py
+++ b/backend/python/pocket-tts/backend.py
@@ -0,0 +1,255 @@
+#!/usr/bin/env python3
+"""
+This is an extra gRPC server of LocalAI for Pocket TTS
+"""
+from concurrent import futures
+import time
+import argparse
+import signal
+import sys
+import os
+import traceback
+import scipy.io.wavfile
+import backend_pb2
+import backend_pb2_grpc
+import torch
+from pocket_tts import TTSModel
+
+import grpc
+
+def is_float(s):
+    """Check if a string can be converted to float."""
+    try:
+        float(s)
+        return True
+    except ValueError:
+        return False
+
+def is_int(s):
+    """Check if a string can be converted to int."""
+    try:
+        int(s)
+        return True
+    except ValueError:
+        return False
+
+_ONE_DAY_IN_SECONDS = 60 * 60 * 24
+
+# If MAX_WORKERS are specified in the environment use it, otherwise default to 1
+MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
+
+# Implement the BackendServicer class with the service methods
+class BackendServicer(backend_pb2_grpc.BackendServicer):
+    """
+    BackendServicer is the class that implements the gRPC service
+    """
+    def Health(self, request, context):
+        return backend_pb2.Reply(message=bytes("OK", 'utf-8'))
+    
+    def LoadModel(self, request, context):
+        # Get device
+        if torch.cuda.is_available():
+            print("CUDA is available", file=sys.stderr)
+            device = "cuda"
+        else:
+            print("CUDA is not available", file=sys.stderr)
+            device = "cpu"
+        mps_available = hasattr(torch.backends, "mps") and torch.backends.mps.is_available()
+        if mps_available:
+            device = "mps"
+        if not torch.cuda.is_available() and request.CUDA:
+            return backend_pb2.Result(success=False, message="CUDA is not available")
+
+        # Normalize potential 'mpx' typo to 'mps'
+        if device == "mpx":
+            print("Note: device 'mpx' detected, treating it as 'mps'.", file=sys.stderr)
+            device = "mps"
+        
+        # Validate mps availability if requested
+        if device == "mps" and not torch.backends.mps.is_available():
+            print("Warning: MPS not available. Falling back to CPU.", file=sys.stderr)
+            device = "cpu"
+
+        self.device = device
+
+        options = request.Options
+
+        # empty dict
+        self.options = {}
+
+        # The options are a list of strings in this form optname:optvalue
+        # We are storing all the options in a dict so we can use it later when
+        # generating the audio
+        for opt in options:
+            if ":" not in opt:
+                continue
+            key, value = opt.split(":", 1)  # Split only on first colon
+            # if value is a number, convert it to the appropriate type
+            if is_float(value):
+                value = float(value)
+            elif is_int(value):
+                value = int(value)
+            elif value.lower() in ["true", "false"]:
+                value = value.lower() == "true"
+            self.options[key] = value
+
+        # Default voice for caching
+        self.default_voice_url = self.options.get("default_voice", None)
+        self._voice_cache = {}
+
+        try:
+            print("Loading Pocket TTS model", file=sys.stderr)
+            self.tts_model = TTSModel.load_model()
+            print(f"Model loaded successfully. Sample rate: {self.tts_model.sample_rate}", file=sys.stderr)
+
+            # Pre-load default voice if specified
+            if self.default_voice_url:
+                try:
+                    print(f"Pre-loading default voice: {self.default_voice_url}", file=sys.stderr)
+                    voice_state = self.tts_model.get_state_for_audio_prompt(self.default_voice_url)
+                    self._voice_cache[self.default_voice_url] = voice_state
+                    print("Default voice loaded successfully", file=sys.stderr)
+                except Exception as e:
+                    print(f"Warning: Failed to pre-load default voice: {e}", file=sys.stderr)
+
+        except Exception as err:
+            return backend_pb2.Result(success=False, message=f"Unexpected {err=}, {type(err)=}")
+        
+        return backend_pb2.Result(message="Model loaded successfully", success=True)
+
+    def _get_voice_state(self, voice_input):
+        """
+        Get voice state from cache or load it.
+        voice_input can be:
+        - HuggingFace URL (e.g., hf://kyutai/tts-voices/alba-mackenna/casual.wav)
+        - Local file path
+        - None (use default)
+        """
+        # Use default if no voice specified
+        if not voice_input:
+            voice_input = self.default_voice_url
+
+        if not voice_input:
+            return None
+
+        # Check cache first
+        if voice_input in self._voice_cache:
+            return self._voice_cache[voice_input]
+
+        # Load voice state
+        try:
+            print(f"Loading voice from: {voice_input}", file=sys.stderr)
+            voice_state = self.tts_model.get_state_for_audio_prompt(voice_input)
+            self._voice_cache[voice_input] = voice_state
+            return voice_state
+        except Exception as e:
+            print(f"Error loading voice from {voice_input}: {e}", file=sys.stderr)
+            return None
+
+    def TTS(self, request, context):
+        try:
+            # Determine voice input
+            # Priority: request.voice > AudioPath (from ModelOptions) > default
+            voice_input = None
+            
+            if request.voice:
+                voice_input = request.voice
+            elif hasattr(request, 'AudioPath') and request.AudioPath:
+                # Use AudioPath as voice file
+                if os.path.isabs(request.AudioPath):
+                    voice_input = request.AudioPath
+                elif hasattr(request, 'ModelFile') and request.ModelFile:
+                    model_file_base = os.path.dirname(request.ModelFile)
+                    voice_input = os.path.join(model_file_base, request.AudioPath)
+                elif hasattr(request, 'ModelPath') and request.ModelPath:
+                    voice_input = os.path.join(request.ModelPath, request.AudioPath)
+                else:
+                    voice_input = request.AudioPath
+
+            # Get voice state
+            voice_state = self._get_voice_state(voice_input)
+            if voice_state is None:
+                return backend_pb2.Result(
+                    success=False,
+                    message=f"Voice not found or failed to load: {voice_input}. Please provide a valid voice URL or file path."
+                )
+
+            # Prepare text
+            text = request.text.strip()
+
+            if not text:
+                return backend_pb2.Result(
+                    success=False,
+                    message="Text is empty"
+                )
+
+            print(f"Generating audio for text: {text[:50]}...", file=sys.stderr)
+
+            # Generate audio
+            audio = self.tts_model.generate_audio(voice_state, text)
+
+            # Audio is a 1D torch tensor containing PCM data
+            if audio is None or audio.numel() == 0:
+                return backend_pb2.Result(
+                    success=False,
+                    message="No audio generated"
+                )
+
+            # Save audio to file
+            output_path = request.dst
+            if not output_path:
+                output_path = "/tmp/pocket-tts-output.wav"
+
+            # Ensure output directory exists
+            output_dir = os.path.dirname(output_path)
+            if output_dir and not os.path.exists(output_dir):
+                os.makedirs(output_dir, exist_ok=True)
+
+            # Convert torch tensor to numpy and save
+            audio_numpy = audio.numpy()
+            scipy.io.wavfile.write(output_path, self.tts_model.sample_rate, audio_numpy)
+            print(f"Saved audio to {output_path}", file=sys.stderr)
+
+        except Exception as err:
+            print(f"Error in TTS: {err}", file=sys.stderr)
+            print(traceback.format_exc(), file=sys.stderr)
+            return backend_pb2.Result(success=False, message=f"Unexpected {err=}, {type(err)=}")
+        
+        return backend_pb2.Result(success=True)
+
+def serve(address):
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=MAX_WORKERS),
+        options=[
+            ('grpc.max_message_length', 50 * 1024 * 1024),  # 50MB
+            ('grpc.max_send_message_length', 50 * 1024 * 1024),  # 50MB
+            ('grpc.max_receive_message_length', 50 * 1024 * 1024),  # 50MB
+        ])
+    backend_pb2_grpc.add_BackendServicer_to_server(BackendServicer(), server)
+    server.add_insecure_port(address)
+    server.start()
+    print("Server started. Listening on: " + address, file=sys.stderr)
+
+    # Define the signal handler function
+    def signal_handler(sig, frame):
+        print("Received termination signal. Shutting down...")
+        server.stop(0)
+        sys.exit(0)
+
+    # Set the signal handlers for SIGINT and SIGTERM
+    signal.signal(signal.SIGINT, signal_handler)
+    signal.signal(signal.SIGTERM, signal_handler)
+
+    try:
+        while True:
+            time.sleep(_ONE_DAY_IN_SECONDS)
+    except KeyboardInterrupt:
+        server.stop(0)
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="Run the gRPC server.")
+    parser.add_argument(
+        "--addr", default="localhost:50051", help="The address to bind the server to."
+    )
+    args = parser.parse_args()
+
+    serve(args.addr)
--- a/backend/python/pocket-tts/install.sh
+++ b/backend/python/pocket-tts/install.sh
@@ -0,0 +1,30 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+# This is here because the Intel pip index is broken and returns 200 status codes for every package name, it just doesn't return any package links.
+# This makes uv think that the package exists in the Intel pip index, and by default it stops looking at other pip indexes once it finds a match.
+# We need uv to continue falling through to the pypi default index to find optimum[openvino] in the pypi index
+# the --upgrade actually allows us to *downgrade* torch to the version provided in the Intel pip index
+if [ "x${BUILD_PROFILE}" == "xintel" ]; then
+    EXTRA_PIP_INSTALL_FLAGS+=" --upgrade --index-strategy=unsafe-first-match"
+fi
+
+# Use python 3.12 for l4t
+if [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
+  PYTHON_VERSION="3.12"
+  PYTHON_PATCH="12"
+  PY_STANDALONE_TAG="20251120"
+fi
+
+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
+installRequirements
--- a/backend/python/pocket-tts/protogen.sh
+++ b/backend/python/pocket-tts/protogen.sh
@@ -0,0 +1,11 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+python3 -m grpc_tools.protoc -I../.. -I./ --python_out=. --grpc_python_out=. backend.proto
--- a/backend/python/pocket-tts/requirements-cpu.txt
+++ b/backend/python/pocket-tts/requirements-cpu.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/cpu
+pocket-tts
+scipy
+torch
--- a/backend/python/pocket-tts/requirements-cublas12.txt
+++ b/backend/python/pocket-tts/requirements-cublas12.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/cu121
+pocket-tts
+scipy
+torch
--- a/backend/python/pocket-tts/requirements-cublas13.txt
+++ b/backend/python/pocket-tts/requirements-cublas13.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/cu130
+pocket-tts
+scipy
+torch
--- a/backend/python/pocket-tts/requirements-hipblas.txt
+++ b/backend/python/pocket-tts/requirements-hipblas.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+pocket-tts
+scipy
+torch==2.7.1+rocm6.3
--- a/backend/python/pocket-tts/requirements-intel.txt
+++ b/backend/python/pocket-tts/requirements-intel.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://pytorch-extension.intel.com/release-whl/stable/xpu/us/
+pocket-tts
+scipy
+torch==2.5.1+cxx11.abi
--- a/backend/python/pocket-tts/requirements-l4t12.txt
+++ b/backend/python/pocket-tts/requirements-l4t12.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://pypi.jetson-ai-lab.io/jp6/cu129/
+pocket-tts
+scipy
+torch
--- a/backend/python/pocket-tts/requirements-l4t13.txt
+++ b/backend/python/pocket-tts/requirements-l4t13.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/cu130
+pocket-tts
+scipy
+torch
--- a/backend/python/pocket-tts/requirements-mps.txt
+++ b/backend/python/pocket-tts/requirements-mps.txt
@@ -0,0 +1,4 @@
+pocket-tts
+scipy
+torch==2.7.1
+torchvision==0.22.1
--- a/backend/python/pocket-tts/requirements.txt
+++ b/backend/python/pocket-tts/requirements.txt
@@ -0,0 +1,4 @@
+grpcio==1.71.0
+protobuf
+certifi
+packaging==24.1
--- a/backend/python/pocket-tts/run.sh
+++ b/backend/python/pocket-tts/run.sh
@@ -0,0 +1,9 @@
+#!/bin/bash
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+startBackend $@
--- a/backend/python/pocket-tts/test.py
+++ b/backend/python/pocket-tts/test.py
@@ -0,0 +1,141 @@
+"""
+A test script to test the gRPC service
+"""
+import unittest
+import subprocess
+import time
+import os
+import tempfile
+import backend_pb2
+import backend_pb2_grpc
+
+import grpc
+
+
+class TestBackendServicer(unittest.TestCase):
+    """
+    TestBackendServicer is the class that tests the gRPC service
+    """
+    def setUp(self):
+        """
+        This method sets up the gRPC service by starting the server
+        """
+        self.service = subprocess.Popen(["python3", "backend.py", "--addr", "localhost:50051"])
+        time.sleep(30)
+
+    def tearDown(self) -> None:
+        """
+        This method tears down the gRPC service by terminating the server
+        """
+        self.service.terminate()
+        self.service.wait()
+
+    def test_server_startup(self):
+        """
+        This method tests if the server starts up successfully
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                response = stub.Health(backend_pb2.HealthMessage())
+                self.assertEqual(response.message, b'OK')
+        except Exception as err:
+            print(err)
+            self.fail("Server failed to start")
+        finally:
+            self.tearDown()
+
+    def test_load_model(self):
+        """
+        This method tests if the model is loaded successfully
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                response = stub.LoadModel(backend_pb2.ModelOptions())
+                print(response)
+                self.assertTrue(response.success)
+                self.assertEqual(response.message, "Model loaded successfully")
+        except Exception as err:
+            print(err)
+            self.fail("LoadModel service failed")
+        finally:
+            self.tearDown()
+
+    def test_tts_with_hf_voice(self):
+        """
+        This method tests TTS generation with HuggingFace voice URL
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                # Load model
+                response = stub.LoadModel(backend_pb2.ModelOptions())
+                self.assertTrue(response.success)
+                
+                # Create temporary output file
+                with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as tmp_file:
+                    output_path = tmp_file.name
+                
+                # Test TTS with HuggingFace voice URL
+                tts_request = backend_pb2.TTSRequest(
+                    text="Hello world, this is a test.",
+                    dst=output_path,
+                    voice="azelma"
+                )
+                tts_response = stub.TTS(tts_request)
+                self.assertTrue(tts_response.success)
+                
+                # Verify output file exists and is not empty
+                self.assertTrue(os.path.exists(output_path))
+                self.assertGreater(os.path.getsize(output_path), 0)
+                
+                # Cleanup
+                os.unlink(output_path)
+        except Exception as err:
+            print(err)
+            self.fail("TTS service failed")
+        finally:
+            self.tearDown()
+
+    def test_tts_with_default_voice(self):
+        """
+        This method tests TTS generation with default voice (via AudioPath in LoadModel)
+        """
+        try:
+            self.setUp()
+            with grpc.insecure_channel("localhost:50051") as channel:
+                stub = backend_pb2_grpc.BackendStub(channel)
+                # Load model with default voice
+                load_request = backend_pb2.ModelOptions(
+                    Options=["default_voice:azelma"]
+                )
+                response = stub.LoadModel(load_request)
+                self.assertTrue(response.success)
+                
+                # Create temporary output file
+                with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as tmp_file:
+                    output_path = tmp_file.name
+                
+                # Test TTS without specifying voice (should use default)
+                tts_request = backend_pb2.TTSRequest(
+                    text="Hello world, this is a test.",
+                    dst=output_path
+                )
+                tts_response = stub.TTS(tts_request)
+                self.assertTrue(tts_response.success)
+                
+                # Verify output file exists and is not empty
+                self.assertTrue(os.path.exists(output_path))
+                self.assertGreater(os.path.getsize(output_path), 0)
+                
+                # Cleanup
+                os.unlink(output_path)
+        except Exception as err:
+            print(err)
+            self.fail("TTS service with default voice failed")
+        finally:
+            self.tearDown()
--- a/backend/python/pocket-tts/test.sh
+++ b/backend/python/pocket-tts/test.sh
@@ -0,0 +1,11 @@
+#!/bin/bash
+set -e
+
+backend_dir=$(dirname $0)
+if [ -d $backend_dir/common ]; then
+    source $backend_dir/common/libbackend.sh
+else
+    source $backend_dir/../common/libbackend.sh
+fi
+
+runUnittests
--- a/backend/python/rerankers/requirements-cublas11.txt
+++ b/backend/python/rerankers/requirements-cublas11.txt
@@ -1,5 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-transformers
-accelerate
-torch==2.4.1+cu118
-rerankers[transformers]
--- a/backend/python/rerankers/requirements-hipblas.txt
+++ b/backend/python/rerankers/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.0
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
 transformers
 accelerate
-torch==2.4.1+rocm6.0
+torch==2.8.0+rocm6.4
 rerankers[transformers]
--- a/backend/python/rfdetr/requirements-cublas11.txt
+++ b/backend/python/rfdetr/requirements-cublas11.txt
@@ -1,8 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.7.1+cu118
-rfdetr
-opencv-python
-accelerate
-inference
-peft
-optimum-quanto
--- a/backend/python/rfdetr/requirements-hipblas.txt
+++ b/backend/python/rfdetr/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.3
-torch==2.7.1+rocm6.3
-torchvision==0.22.1+rocm6.3
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
+torchvision==0.23.0+rocm6.4
 rfdetr
 opencv-python
 accelerate
--- a/backend/python/transformers/requirements-cpu.txt
+++ b/backend/python/transformers/requirements-cpu.txt
@@ -6,4 +6,4 @@ transformers
 bitsandbytes
 outetts
 sentence-transformers==5.2.0
-protobuf==6.33.2
+protobuf==6.33.4
--- a/backend/python/transformers/requirements-cublas11.txt
+++ b/backend/python/transformers/requirements-cublas11.txt
@@ -1,10 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-torch==2.7.1+cu118
-llvmlite==0.43.0
-numba==0.60.0
-accelerate
-transformers
-bitsandbytes
-outetts
-sentence-transformers==5.2.0
-protobuf==6.33.2
--- a/backend/python/transformers/requirements-cublas12.txt
+++ b/backend/python/transformers/requirements-cublas12.txt
@@ -6,4 +6,4 @@ transformers
 bitsandbytes
 outetts
 sentence-transformers==5.2.0
-protobuf==6.33.2
+protobuf==6.33.4
--- a/backend/python/transformers/requirements-cublas13.txt
+++ b/backend/python/transformers/requirements-cublas13.txt
@@ -6,4 +6,4 @@ transformers
 bitsandbytes
 outetts
 sentence-transformers==5.2.0
-protobuf==6.33.2
+protobuf==6.33.4
--- a/backend/python/transformers/requirements-hipblas.txt
+++ b/backend/python/transformers/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.3
-torch==2.7.1+rocm6.3
+--extra-index-url https://download.pytorch.org/whl/rocm6.4
+torch==2.8.0+rocm6.4
 accelerate
 transformers
 llvmlite==0.43.0
@@ -8,4 +8,4 @@ bitsandbytes
 outetts
 bitsandbytes
 sentence-transformers==5.2.0
-protobuf==6.33.2
+protobuf==6.33.4
--- a/backend/python/transformers/requirements-intel.txt
+++ b/backend/python/transformers/requirements-intel.txt
@@ -10,4 +10,4 @@ intel-extension-for-transformers
 bitsandbytes
 outetts
 sentence-transformers==5.2.0
-protobuf==6.33.2
+protobuf==6.33.4
--- a/backend/python/transformers/requirements.txt
+++ b/backend/python/transformers/requirements.txt
@@ -1,5 +1,5 @@
 grpcio==1.76.0
-protobuf==6.33.2
+protobuf==6.33.4
 certifi
 setuptools
 scipy==1.15.1
--- a/backend/python/vibevoice/install.sh
+++ b/backend/python/vibevoice/install.sh
@@ -17,12 +17,16 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
 fi

 # Use python 3.12 for l4t
-if [ "x${BUILD_PROFILE}" == "xl4t12" ] || [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
+if [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
  PYTHON_VERSION="3.12"
  PYTHON_PATCH="12"
  PY_STANDALONE_TAG="20251120"
 fi

+if [ "x${BUILD_PROFILE}" == "xl4t12" ]; then
+    USE_PIP=true
+fi
+
 installRequirements

 git clone https://github.com/microsoft/VibeVoice.git
--- a/backend/python/vibevoice/requirements-cublas11.txt
+++ b/backend/python/vibevoice/requirements-cublas11.txt
@@ -1,22 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-git+https://github.com/huggingface/diffusers
-opencv-python
-transformers==4.51.3
-torchvision==0.22.1
-accelerate
-compel
-peft
-sentencepiece
-torch==2.7.1
-optimum-quanto
-ftfy
-llvmlite>=0.40.0
-numba>=0.57.0
-tqdm
-numpy
-scipy
-librosa
-ml-collections
-absl-py
-gradio
-av
--- a/backend/python/vllm/install.sh
+++ b/backend/python/vllm/install.sh
@@ -28,7 +28,7 @@ fi

 # We don't embed this into the images as it is a large dependency and not always needed.
 # Besides, the speed inference are not actually usable in the current state for production use-cases.
-if [ "x${BUILD_TYPE}" == "x" ] && [ "x${FROM_SOURCE}" == "xtrue" ]; then
+if [ "x${BUILD_TYPE}" == "x" ] && [ "x${FROM_SOURCE:-}" == "xtrue" ]; then
        ensureVenv
        # https://docs.vllm.ai/en/v0.6.1/getting_started/cpu-installation.html
        if [ ! -d vllm ]; then
--- a/backend/python/vllm/requirements-cublas11-after.txt
+++ b/backend/python/vllm/requirements-cublas11-after.txt
@@ -1 +0,0 @@
-flash-attn
--- a/backend/python/vllm/requirements-cublas11.txt
+++ b/backend/python/vllm/requirements-cublas11.txt
@@ -1,5 +0,0 @@
--extra-index-url https://download.pytorch.org/whl/cu118
-accelerate
-torch==2.7.0+cu118
-transformers
-bitsandbytes
--- a/backend/python/vllm/requirements-hipblas.txt
+++ b/backend/python/vllm/requirements-hipblas.txt
@@ -1,4 +1,4 @@
--extra-index-url https://download.pytorch.org/whl/nightly/rocm6.3
+--extra-index-url https://download.pytorch.org/whl/nightly/rocm6.4
 accelerate
 torch
 transformers
--- a/core/cli/run.go
+++ b/core/cli/run.go
@@ -56,7 +56,7 @@ type RunCMD struct {
 	UseSubtleKeyComparison             bool     `env:"LOCALAI_SUBTLE_KEY_COMPARISON" default:"false" help:"If true, API Key validation comparisons will be performed using constant-time comparisons rather than simple equality. This trades off performance on each request for resiliancy against timing attacks." group:"hardening"`
 	DisableApiKeyRequirementForHttpGet bool     `env:"LOCALAI_DISABLE_API_KEY_REQUIREMENT_FOR_HTTP_GET" default:"false" help:"If true, a valid API key is not required to issue GET requests to portions of the web ui. This should only be enabled in secure testing environments" group:"hardening"`
 	DisableMetricsEndpoint             bool     `env:"LOCALAI_DISABLE_METRICS_ENDPOINT,DISABLE_METRICS_ENDPOINT" default:"false" help:"Disable the /metrics endpoint" group:"api"`
-	HttpGetExemptedEndpoints           []string `env:"LOCALAI_HTTP_GET_EXEMPTED_ENDPOINTS" default:"^/$,^/browse/?$,^/talk/?$,^/p2p/?$,^/chat/?$,^/text2image/?$,^/tts/?$,^/static/.*$,^/swagger.*$" help:"If LOCALAI_DISABLE_API_KEY_REQUIREMENT_FOR_HTTP_GET is overriden to true, this is the list of endpoints to exempt. Only adjust this in case of a security incident or as a result of a personal security posture review" group:"hardening"`
+	HttpGetExemptedEndpoints           []string `env:"LOCALAI_HTTP_GET_EXEMPTED_ENDPOINTS" default:"^/$,^/browse/?$,^/talk/?$,^/p2p/?$,^/chat/?$,^/image/?$,^/text2image/?$,^/tts/?$,^/static/.*$,^/swagger.*$" help:"If LOCALAI_DISABLE_API_KEY_REQUIREMENT_FOR_HTTP_GET is overriden to true, this is the list of endpoints to exempt. Only adjust this in case of a security incident or as a result of a personal security posture review" group:"hardening"`
 	Peer2Peer                          bool     `env:"LOCALAI_P2P,P2P" name:"p2p" default:"false" help:"Enable P2P mode" group:"p2p"`
 	Peer2PeerDHTInterval               int      `env:"LOCALAI_P2P_DHT_INTERVAL,P2P_DHT_INTERVAL" default:"360" name:"p2p-dht-interval" help:"Interval for DHT refresh (used during token generation)" group:"p2p"`
 	Peer2PeerOTPInterval               int      `env:"LOCALAI_P2P_OTP_INTERVAL,P2P_OTP_INTERVAL" default:"9000" name:"p2p-otp-interval" help:"Interval for OTP refresh (used during token generation)" group:"p2p"`
@@ -83,6 +83,7 @@ type RunCMD struct {
 	EnableTracing                      bool     `env:"LOCALAI_ENABLE_TRACING,ENABLE_TRACING" help:"Enable API tracing" group:"api"`
 	TracingMaxItems                    int      `env:"LOCALAI_TRACING_MAX_ITEMS" default:"1024" help:"Maximum number of traces to keep" group:"api"`
 	AgentJobRetentionDays              int      `env:"LOCALAI_AGENT_JOB_RETENTION_DAYS,AGENT_JOB_RETENTION_DAYS" default:"30" help:"Number of days to keep agent job history (default: 30)" group:"api"`
+	OpenResponsesStoreTTL               string   `env:"LOCALAI_OPEN_RESPONSES_STORE_TTL,OPEN_RESPONSES_STORE_TTL" default:"0" help:"TTL for Open Responses store (e.g., 1h, 30m, 0 = no expiration)" group:"api"`

 	Version bool
 }
@@ -249,6 +250,15 @@ func (r *RunCMD) Run(ctx *cliContext.Context) error {
 		opts = append(opts, config.WithLRUEvictionRetryInterval(dur))
 	}

+	// Handle Open Responses store TTL
+	if r.OpenResponsesStoreTTL != "" && r.OpenResponsesStoreTTL != "0" {
+		dur, err := time.ParseDuration(r.OpenResponsesStoreTTL)
+		if err != nil {
+			return fmt.Errorf("invalid Open Responses store TTL: %w", err)
+		}
+		opts = append(opts, config.WithOpenResponsesStoreTTL(dur))
+	}
+
 	// split ":" to get backend name and the uri
 	for _, v := range r.ExternalGRPCBackends {
 		backend := v[:strings.IndexByte(v, ':')]
--- a/core/config/application_config.go
+++ b/core/config/application_config.go
@@ -86,6 +86,8 @@ type ApplicationConfig struct {

 	AgentJobRetentionDays int // Default: 30 days

+	OpenResponsesStoreTTL time.Duration // TTL for Open Responses store (0 = no expiration)
+
 	PathWithoutAuth []string
 }

@@ -467,6 +469,12 @@ func WithAgentJobRetentionDays(days int) AppOption {
 	}
 }

+func WithOpenResponsesStoreTTL(ttl time.Duration) AppOption {
+	return func(o *ApplicationConfig) {
+		o.OpenResponsesStoreTTL = ttl
+	}
+}
+
 func WithEnforcedPredownloadScans(enforced bool) AppOption {
 	return func(o *ApplicationConfig) {
 		o.EnforcePredownloadScans = enforced
@@ -594,6 +602,12 @@ func (o *ApplicationConfig) ToRuntimeSettings() RuntimeSettings {
 	} else {
 		lruEvictionRetryInterval = "1s" // default
 	}
+	var openResponsesStoreTTL string
+	if o.OpenResponsesStoreTTL > 0 {
+		openResponsesStoreTTL = o.OpenResponsesStoreTTL.String()
+	} else {
+		openResponsesStoreTTL = "0" // default: no expiration
+	}

 	return RuntimeSettings{
 		WatchdogEnabled:          &watchdogEnabled,
@@ -628,6 +642,7 @@ func (o *ApplicationConfig) ToRuntimeSettings() RuntimeSettings {
 		AutoloadBackendGalleries: &autoloadBackendGalleries,
 		ApiKeys:                  &apiKeys,
 		AgentJobRetentionDays:    &agentJobRetentionDays,
+		OpenResponsesStoreTTL:     &openResponsesStoreTTL,
 	}
 }

@@ -769,6 +784,14 @@ func (o *ApplicationConfig) ApplyRuntimeSettings(settings *RuntimeSettings) (req
 	if settings.AgentJobRetentionDays != nil {
 		o.AgentJobRetentionDays = *settings.AgentJobRetentionDays
 	}
+	if settings.OpenResponsesStoreTTL != nil {
+		if *settings.OpenResponsesStoreTTL == "0" || *settings.OpenResponsesStoreTTL == "" {
+			o.OpenResponsesStoreTTL = 0 // No expiration
+		} else if dur, err := time.ParseDuration(*settings.OpenResponsesStoreTTL); err == nil {
+			o.OpenResponsesStoreTTL = dur
+		}
+		// This setting doesn't require restart, can be updated dynamically
+	}
 	// Note: ApiKeys requires special handling (merging with startup keys) - handled in caller

 	return requireRestart
--- a/core/config/runtime_settings.go
+++ b/core/config/runtime_settings.go
@@ -60,4 +60,7 @@ type RuntimeSettings struct {

 	// Agent settings
 	AgentJobRetentionDays *int `json:"agent_job_retention_days,omitempty"`
+
+	// Open Responses settings
+	OpenResponsesStoreTTL *string `json:"open_responses_store_ttl,omitempty"` // TTL for stored responses (e.g., "1h", "30m", "0" = no expiration)
 }
--- a/core/gallery/backend_types.go
+++ b/core/gallery/backend_types.go
@@ -63,6 +63,25 @@ func (m *GalleryBackend) IsMeta() bool {
 	return len(m.CapabilitiesMap) > 0 && m.URI == ""
 }

+// IsCompatibleWith checks if the backend is compatible with the current system capability.
+// For meta backends, it checks if any of the capabilities in the map match the system capability.
+// For concrete backends, it delegates to SystemState.IsBackendCompatible.
+func (m *GalleryBackend) IsCompatibleWith(systemState *system.SystemState) bool {
+	if systemState == nil {
+		return true
+	}
+
+	// Meta backends are compatible if the system capability matches one of the keys
+	if m.IsMeta() {
+		capability := systemState.Capability(m.CapabilitiesMap)
+		_, exists := m.CapabilitiesMap[capability]
+		return exists
+	}
+
+	// For concrete backends, delegate to the system package
+	return systemState.IsBackendCompatible(m.Name, m.URI)
+}
+
 func (m *GalleryBackend) SetInstalled(installed bool) {
 	m.Installed = installed
 }
--- a/core/gallery/backends_test.go
+++ b/core/gallery/backends_test.go
@@ -172,6 +172,252 @@ var _ = Describe("Gallery Backends", func() {
 			Expect(nilMetaBackend.IsMeta()).To(BeFalse())
 		})

+		It("should check IsCompatibleWith correctly for meta backends", func() {
+			metaBackend := &GalleryBackend{
+				Metadata: Metadata{
+					Name: "meta-backend",
+				},
+				CapabilitiesMap: map[string]string{
+					"nvidia":  "nvidia-backend",
+					"amd":     "amd-backend",
+					"default": "default-backend",
+				},
+			}
+
+			// Test with nil state - should be compatible
+			Expect(metaBackend.IsCompatibleWith(nil)).To(BeTrue())
+
+			// Test with NVIDIA system - should be compatible (has nvidia key)
+			nvidiaState := &system.SystemState{GPUVendor: "nvidia", VRAM: 8 * 1024 * 1024 * 1024}
+			Expect(metaBackend.IsCompatibleWith(nvidiaState)).To(BeTrue())
+
+			// Test with default (no GPU) - should be compatible (has default key)
+			defaultState := &system.SystemState{}
+			Expect(metaBackend.IsCompatibleWith(defaultState)).To(BeTrue())
+		})
+
+		Describe("IsCompatibleWith for concrete backends", func() {
+			Context("CPU backends", func() {
+				It("should be compatible on all systems", func() {
+					cpuBackend := &GalleryBackend{
+						Metadata: Metadata{
+							Name: "cpu-llama-cpp",
+						},
+						URI: "quay.io/go-skynet/local-ai-backends:latest-cpu-llama-cpp",
+					}
+					Expect(cpuBackend.IsCompatibleWith(&system.SystemState{})).To(BeTrue())
+					Expect(cpuBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Nvidia, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					Expect(cpuBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.AMD, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+				})
+			})
+
+			Context("Darwin/Metal backends", func() {
+				When("running on darwin", func() {
+					BeforeEach(func() {
+						if runtime.GOOS != "darwin" {
+							Skip("Skipping darwin-specific tests on non-darwin system")
+						}
+					})
+
+					It("should be compatible for MLX backend", func() {
+						mlxBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "mlx",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-metal-darwin-arm64-mlx",
+						}
+						Expect(mlxBackend.IsCompatibleWith(&system.SystemState{})).To(BeTrue())
+					})
+
+					It("should be compatible for metal-llama-cpp backend", func() {
+						metalBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "metal-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-metal-darwin-arm64-llama-cpp",
+						}
+						Expect(metalBackend.IsCompatibleWith(&system.SystemState{})).To(BeTrue())
+					})
+				})
+
+				When("running on non-darwin", func() {
+					BeforeEach(func() {
+						if runtime.GOOS == "darwin" {
+							Skip("Skipping non-darwin-specific tests on darwin system")
+						}
+					})
+
+					It("should NOT be compatible for MLX backend", func() {
+						mlxBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "mlx",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-metal-darwin-arm64-mlx",
+						}
+						Expect(mlxBackend.IsCompatibleWith(&system.SystemState{})).To(BeFalse())
+					})
+
+					It("should NOT be compatible for metal-llama-cpp backend", func() {
+						metalBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "metal-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-metal-darwin-arm64-llama-cpp",
+						}
+						Expect(metalBackend.IsCompatibleWith(&system.SystemState{})).To(BeFalse())
+					})
+				})
+			})
+
+			Context("NVIDIA/CUDA backends", func() {
+				When("running on non-darwin", func() {
+					BeforeEach(func() {
+						if runtime.GOOS == "darwin" {
+							Skip("Skipping CUDA tests on darwin system")
+						}
+					})
+
+					It("should NOT be compatible without nvidia GPU", func() {
+						cudaBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "cuda12-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-llama-cpp",
+						}
+						Expect(cudaBackend.IsCompatibleWith(&system.SystemState{})).To(BeFalse())
+						Expect(cudaBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.AMD, VRAM: 8 * 1024 * 1024 * 1024})).To(BeFalse())
+					})
+
+					It("should be compatible with nvidia GPU", func() {
+						cudaBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "cuda12-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-llama-cpp",
+						}
+						Expect(cudaBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Nvidia, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+
+					It("should be compatible with cuda13 backend on nvidia GPU", func() {
+						cuda13Backend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "cuda13-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-13-llama-cpp",
+						}
+						Expect(cuda13Backend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Nvidia, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+				})
+			})
+
+			Context("AMD/ROCm backends", func() {
+				When("running on non-darwin", func() {
+					BeforeEach(func() {
+						if runtime.GOOS == "darwin" {
+							Skip("Skipping AMD/ROCm tests on darwin system")
+						}
+					})
+
+					It("should NOT be compatible without AMD GPU", func() {
+						rocmBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "rocm-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-llama-cpp",
+						}
+						Expect(rocmBackend.IsCompatibleWith(&system.SystemState{})).To(BeFalse())
+						Expect(rocmBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Nvidia, VRAM: 8 * 1024 * 1024 * 1024})).To(BeFalse())
+					})
+
+					It("should be compatible with AMD GPU", func() {
+						rocmBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "rocm-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-llama-cpp",
+						}
+						Expect(rocmBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.AMD, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+
+					It("should be compatible with hipblas backend on AMD GPU", func() {
+						hipBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "hip-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-hip-llama-cpp",
+						}
+						Expect(hipBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.AMD, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+				})
+			})
+
+			Context("Intel/SYCL backends", func() {
+				When("running on non-darwin", func() {
+					BeforeEach(func() {
+						if runtime.GOOS == "darwin" {
+							Skip("Skipping Intel/SYCL tests on darwin system")
+						}
+					})
+
+					It("should NOT be compatible without Intel GPU", func() {
+						intelBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "intel-sycl-f16-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-sycl-f16-llama-cpp",
+						}
+						Expect(intelBackend.IsCompatibleWith(&system.SystemState{})).To(BeFalse())
+						Expect(intelBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Nvidia, VRAM: 8 * 1024 * 1024 * 1024})).To(BeFalse())
+					})
+
+					It("should be compatible with Intel GPU", func() {
+						intelBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "intel-sycl-f16-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-sycl-f16-llama-cpp",
+						}
+						Expect(intelBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Intel, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+
+					It("should be compatible with intel-sycl-f32 backend on Intel GPU", func() {
+						intelF32Backend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "intel-sycl-f32-llama-cpp",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-sycl-f32-llama-cpp",
+						}
+						Expect(intelF32Backend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Intel, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+
+					It("should be compatible with intel-transformers backend on Intel GPU", func() {
+						intelTransformersBackend := &GalleryBackend{
+							Metadata: Metadata{
+								Name: "intel-transformers",
+							},
+							URI: "quay.io/go-skynet/local-ai-backends:latest-intel-transformers",
+						}
+						Expect(intelTransformersBackend.IsCompatibleWith(&system.SystemState{GPUVendor: system.Intel, VRAM: 8 * 1024 * 1024 * 1024})).To(BeTrue())
+					})
+				})
+			})
+
+			Context("Vulkan backends", func() {
+				It("should be compatible on CPU-only systems", func() {
+					// Vulkan backends don't have a specific GPU vendor requirement in the current logic
+					// They are compatible if no other GPU-specific pattern matches
+					vulkanBackend := &GalleryBackend{
+						Metadata: Metadata{
+							Name: "vulkan-llama-cpp",
+						},
+						URI: "quay.io/go-skynet/local-ai-backends:latest-gpu-vulkan-llama-cpp",
+					}
+					// Vulkan doesn't have vendor-specific filtering in current implementation
+					Expect(vulkanBackend.IsCompatibleWith(&system.SystemState{})).To(BeTrue())
+				})
+			})
+		})
+
 		It("should find best backend from meta based on system capabilities", func() {

 			metaBackend := &GalleryBackend{
--- a/core/gallery/gallery.go
+++ b/core/gallery/gallery.go
@@ -226,6 +226,16 @@ func AvailableGalleryModels(galleries []config.Gallery, systemState *system.Syst

 // List available backends
 func AvailableBackends(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryBackend], error) {
+	return availableBackendsWithFilter(galleries, systemState, true)
+}
+
+// AvailableBackendsUnfiltered returns all available backends without filtering by system capability.
+func AvailableBackendsUnfiltered(galleries []config.Gallery, systemState *system.SystemState) (GalleryElements[*GalleryBackend], error) {
+	return availableBackendsWithFilter(galleries, systemState, false)
+}
+
+// availableBackendsWithFilter is a helper function that lists available backends with optional filtering.
+func availableBackendsWithFilter(galleries []config.Gallery, systemState *system.SystemState, filterByCapability bool) (GalleryElements[*GalleryBackend], error) {
 	var backends []*GalleryBackend

 	systemBackends, err := ListSystemBackends(systemState)
@@ -241,7 +251,17 @@ func AvailableBackends(galleries []config.Gallery, systemState *system.SystemSta
 		if err != nil {
 			return nil, err
 		}
-		backends = append(backends, galleryBackends...)
+
+		// Filter backends by system capability if requested
+		if filterByCapability {
+			for _, backend := range galleryBackends {
+				if backend.IsCompatibleWith(systemState) {
+					backends = append(backends, backend)
+				}
+			}
+		} else {
+			backends = append(backends, galleryBackends...)
+		}
 	}

 	return backends, nil
--- a/core/http/app.go
+++ b/core/http/app.go
@@ -108,7 +108,15 @@ func API(application *application.Application) (*echo.Echo, error) {
 			req := c.Request()
 			res := c.Response()
 			err := next(c)
-			xlog.Info("HTTP request", "method", req.Method, "path", req.URL.Path, "status", res.Status)
+
+			// Fix for #7989: Reduce log verbosity of Web UI polling
+			// If the path is /api/operations and the request was successful (200),
+			// we log it at DEBUG level (hidden by default) instead of INFO.
+			if req.URL.Path == "/api/operations" && res.Status == 200 {
+				xlog.Debug("HTTP request", "method", req.Method, "path", req.URL.Path, "status", res.Status)
+			} else {
+				xlog.Info("HTTP request", "method", req.Method, "path", req.URL.Path, "status", res.Status)
+			}
 			return err
 		}
 	})
@@ -185,6 +193,8 @@ func API(application *application.Application) (*echo.Echo, error) {
 			corsConfig.AllowOrigins = strings.Split(application.ApplicationConfig().CORSAllowOrigins, ",")
 		}
 		e.Use(middleware.CORSWithConfig(corsConfig))
+	} else {
+		e.Use(middleware.CORS())
 	}

 	// CSRF middleware
@@ -205,6 +215,8 @@ func API(application *application.Application) (*echo.Echo, error) {

 	routes.RegisterLocalAIRoutes(e, requestExtractor, application.ModelConfigLoader(), application.ModelLoader(), application.ApplicationConfig(), application.GalleryService(), opcache, application.TemplatesEvaluator(), application)
 	routes.RegisterOpenAIRoutes(e, requestExtractor, application)
+	routes.RegisterAnthropicRoutes(e, requestExtractor, application)
+	routes.RegisterOpenResponsesRoutes(e, requestExtractor, application)
 	if !application.ApplicationConfig().DisableWebUI {
 		routes.RegisterUIAPIRoutes(e, application.ModelConfigLoader(), application.ModelLoader(), application.ApplicationConfig(), application.GalleryService(), opcache, application)
 		routes.RegisterUIRoutes(e, application.ModelConfigLoader(), application.ModelLoader(), application.ApplicationConfig(), application.GalleryService())
--- a/core/http/endpoints/anthropic/messages.go
+++ b/core/http/endpoints/anthropic/messages.go
@@ -0,0 +1,537 @@
+package anthropic
+
+import (
+	"encoding/json"
+	"fmt"
+
+	"github.com/google/uuid"
+	"github.com/labstack/echo/v4"
+	"github.com/mudler/LocalAI/core/backend"
+	"github.com/mudler/LocalAI/core/config"
+	"github.com/mudler/LocalAI/core/http/middleware"
+	"github.com/mudler/LocalAI/core/schema"
+	"github.com/mudler/LocalAI/core/templates"
+	"github.com/mudler/LocalAI/pkg/functions"
+	"github.com/mudler/LocalAI/pkg/model"
+	"github.com/mudler/xlog"
+)
+
+// MessagesEndpoint is the Anthropic Messages API endpoint
+// https://docs.anthropic.com/claude/reference/messages_post
+// @Summary Generate a message response for the given messages and model.
+// @Param request body schema.AnthropicRequest true "query params"
+// @Success 200 {object} schema.AnthropicResponse "Response"
+// @Router /v1/messages [post]
+func MessagesEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator *templates.Evaluator, appConfig *config.ApplicationConfig) echo.HandlerFunc {
+	return func(c echo.Context) error {
+		id := uuid.New().String()
+
+		input, ok := c.Get(middleware.CONTEXT_LOCALS_KEY_LOCALAI_REQUEST).(*schema.AnthropicRequest)
+		if !ok || input.Model == "" {
+			return sendAnthropicError(c, 400, "invalid_request_error", "model is required")
+		}
+
+		cfg, ok := c.Get(middleware.CONTEXT_LOCALS_KEY_MODEL_CONFIG).(*config.ModelConfig)
+		if !ok || cfg == nil {
+			return sendAnthropicError(c, 400, "invalid_request_error", "model configuration not found")
+		}
+
+		if input.MaxTokens <= 0 {
+			return sendAnthropicError(c, 400, "invalid_request_error", "max_tokens is required and must be greater than 0")
+		}
+
+		xlog.Debug("Anthropic Messages endpoint configuration read", "config", cfg)
+
+		// Convert Anthropic messages to OpenAI format for internal processing
+		openAIMessages := convertAnthropicToOpenAIMessages(input)
+
+		// Convert Anthropic tools to internal Functions format
+		funcs, shouldUseFn := convertAnthropicTools(input, cfg)
+
+		// Create an OpenAI-compatible request for internal processing
+		openAIReq := &schema.OpenAIRequest{
+			PredictionOptions: schema.PredictionOptions{
+				BasicModelRequest: schema.BasicModelRequest{Model: input.Model},
+				Temperature:       input.Temperature,
+				TopK:              input.TopK,
+				TopP:              input.TopP,
+				Maxtokens:         &input.MaxTokens,
+			},
+			Messages: openAIMessages,
+			Stream:   input.Stream,
+			Context:  input.Context,
+			Cancel:   input.Cancel,
+		}
+
+		// Set stop sequences
+		if len(input.StopSequences) > 0 {
+			openAIReq.Stop = input.StopSequences
+		}
+
+		// Merge config settings
+		if input.Temperature != nil {
+			cfg.Temperature = input.Temperature
+		}
+		if input.TopK != nil {
+			cfg.TopK = input.TopK
+		}
+		if input.TopP != nil {
+			cfg.TopP = input.TopP
+		}
+		cfg.Maxtokens = &input.MaxTokens
+		if len(input.StopSequences) > 0 {
+			cfg.StopWords = append(cfg.StopWords, input.StopSequences...)
+		}
+
+		// Template the prompt with tools if available
+		predInput := evaluator.TemplateMessages(*openAIReq, openAIReq.Messages, cfg, funcs, shouldUseFn)
+		xlog.Debug("Anthropic Messages - Prompt (after templating)", "prompt", predInput)
+
+		if input.Stream {
+			return handleAnthropicStream(c, id, input, cfg, ml, predInput, openAIReq, funcs, shouldUseFn)
+		}
+
+		return handleAnthropicNonStream(c, id, input, cfg, ml, predInput, openAIReq, funcs, shouldUseFn)
+	}
+}
+
+func handleAnthropicNonStream(c echo.Context, id string, input *schema.AnthropicRequest, cfg *config.ModelConfig, ml *model.ModelLoader, predInput string, openAIReq *schema.OpenAIRequest, funcs functions.Functions, shouldUseFn bool) error {
+	images := []string{}
+	for _, m := range openAIReq.Messages {
+		images = append(images, m.StringImages...)
+	}
+
+	predFunc, err := backend.ModelInference(
+		input.Context, predInput, openAIReq.Messages, images, nil, nil, ml, cfg, nil, nil, nil, "", "", nil, nil, nil)
+	if err != nil {
+		xlog.Error("Anthropic model inference failed", "error", err)
+		return sendAnthropicError(c, 500, "api_error", fmt.Sprintf("model inference failed: %v", err))
+	}
+
+	prediction, err := predFunc()
+	if err != nil {
+		xlog.Error("Anthropic prediction failed", "error", err)
+		return sendAnthropicError(c, 500, "api_error", fmt.Sprintf("prediction failed: %v", err))
+	}
+
+	result := backend.Finetune(*cfg, predInput, prediction.Response)
+	
+	// Check if the result contains tool calls
+	toolCalls := functions.ParseFunctionCall(result, cfg.FunctionsConfig)
+	
+	var contentBlocks []schema.AnthropicContentBlock
+	var stopReason string
+	
+	if shouldUseFn && len(toolCalls) > 0 {
+		// Model wants to use tools
+		stopReason = "tool_use"
+		for _, tc := range toolCalls {
+			// Parse arguments as JSON
+			var inputArgs map[string]interface{}
+			if err := json.Unmarshal([]byte(tc.Arguments), &inputArgs); err != nil {
+				xlog.Warn("Failed to parse tool call arguments as JSON", "error", err, "args", tc.Arguments)
+				inputArgs = map[string]interface{}{"raw": tc.Arguments}
+			}
+			
+			contentBlocks = append(contentBlocks, schema.AnthropicContentBlock{
+				Type:  "tool_use",
+				ID:    fmt.Sprintf("toolu_%s_%d", id, len(contentBlocks)),
+				Name:  tc.Name,
+				Input: inputArgs,
+			})
+		}
+		
+		// Add any text content before the tool calls
+		textContent := functions.ParseTextContent(result, cfg.FunctionsConfig)
+		if textContent != "" {
+			// Prepend text block
+			contentBlocks = append([]schema.AnthropicContentBlock{{Type: "text", Text: textContent}}, contentBlocks...)
+		}
+	} else {
+		// Normal text response
+		stopReason = "end_turn"
+		contentBlocks = []schema.AnthropicContentBlock{
+			{Type: "text", Text: result},
+		}
+	}
+
+	resp := &schema.AnthropicResponse{
+		ID:         fmt.Sprintf("msg_%s", id),
+		Type:       "message",
+		Role:       "assistant",
+		Model:      input.Model,
+		StopReason: &stopReason,
+		Content:    contentBlocks,
+		Usage: schema.AnthropicUsage{
+			InputTokens:  prediction.Usage.Prompt,
+			OutputTokens: prediction.Usage.Completion,
+		},
+	}
+
+	if respData, err := json.Marshal(resp); err == nil {
+		xlog.Debug("Anthropic Response", "response", string(respData))
+	}
+
+	return c.JSON(200, resp)
+}
+
+func handleAnthropicStream(c echo.Context, id string, input *schema.AnthropicRequest, cfg *config.ModelConfig, ml *model.ModelLoader, predInput string, openAIReq *schema.OpenAIRequest, funcs functions.Functions, shouldUseFn bool) error {
+	c.Response().Header().Set("Content-Type", "text/event-stream")
+	c.Response().Header().Set("Cache-Control", "no-cache")
+	c.Response().Header().Set("Connection", "keep-alive")
+
+	// Create OpenAI messages for inference
+	openAIMessages := openAIReq.Messages
+
+	images := []string{}
+	for _, m := range openAIMessages {
+		images = append(images, m.StringImages...)
+	}
+
+	// Send message_start event
+	messageStart := schema.AnthropicStreamEvent{
+		Type: "message_start",
+		Message: &schema.AnthropicStreamMessage{
+			ID:      fmt.Sprintf("msg_%s", id),
+			Type:    "message",
+			Role:    "assistant",
+			Content: []schema.AnthropicContentBlock{},
+			Model:   input.Model,
+			Usage:   schema.AnthropicUsage{InputTokens: 0, OutputTokens: 0},
+		},
+	}
+	sendAnthropicSSE(c, messageStart)
+
+	// Track accumulated content for tool call detection
+	accumulatedContent := ""
+	currentBlockIndex := 0
+	inToolCall := false
+	toolCallsEmitted := 0
+	
+	// Send initial content_block_start event
+	contentBlockStart := schema.AnthropicStreamEvent{
+		Type:         "content_block_start",
+		Index:        currentBlockIndex,
+		ContentBlock: &schema.AnthropicContentBlock{Type: "text", Text: ""},
+	}
+	sendAnthropicSSE(c, contentBlockStart)
+
+	// Stream content deltas
+	tokenCallback := func(token string, usage backend.TokenUsage) bool {
+		accumulatedContent += token
+		
+		// If we're using functions, try to detect tool calls incrementally
+		if shouldUseFn {
+			cleanedResult := functions.CleanupLLMResult(accumulatedContent, cfg.FunctionsConfig)
+			
+			// Try parsing for tool calls
+			toolCalls := functions.ParseFunctionCall(cleanedResult, cfg.FunctionsConfig)
+			
+			// If we detected new tool calls and haven't emitted them yet
+			if len(toolCalls) > toolCallsEmitted {
+				// Stop the current text block if we were in one
+				if !inToolCall && currentBlockIndex == 0 {
+					sendAnthropicSSE(c, schema.AnthropicStreamEvent{
+						Type:  "content_block_stop",
+						Index: currentBlockIndex,
+					})
+					currentBlockIndex++
+					inToolCall = true
+				}
+				
+				// Emit new tool calls
+				for i := toolCallsEmitted; i < len(toolCalls); i++ {
+					tc := toolCalls[i]
+					
+					// Send content_block_start for tool_use
+					sendAnthropicSSE(c, schema.AnthropicStreamEvent{
+						Type:  "content_block_start",
+						Index: currentBlockIndex,
+						ContentBlock: &schema.AnthropicContentBlock{
+							Type: "tool_use",
+							ID:   fmt.Sprintf("toolu_%s_%d", id, i),
+							Name: tc.Name,
+						},
+					})
+					
+					// Send input_json_delta with the arguments
+					sendAnthropicSSE(c, schema.AnthropicStreamEvent{
+						Type:  "content_block_delta",
+						Index: currentBlockIndex,
+						Delta: &schema.AnthropicStreamDelta{
+							Type:        "input_json_delta",
+							PartialJSON: tc.Arguments,
+						},
+					})
+					
+					// Send content_block_stop
+					sendAnthropicSSE(c, schema.AnthropicStreamEvent{
+						Type:  "content_block_stop",
+						Index: currentBlockIndex,
+					})
+					
+					currentBlockIndex++
+				}
+				toolCallsEmitted = len(toolCalls)
+				return true
+			}
+		}
+		
+		// Send regular text delta if not in tool call mode
+		if !inToolCall {
+			delta := schema.AnthropicStreamEvent{
+				Type:  "content_block_delta",
+				Index: 0,
+				Delta: &schema.AnthropicStreamDelta{
+					Type: "text_delta",
+					Text: token,
+				},
+			}
+			sendAnthropicSSE(c, delta)
+		}
+		return true
+	}
+
+	predFunc, err := backend.ModelInference(
+		input.Context, predInput, openAIMessages, images, nil, nil, ml, cfg, nil, nil, tokenCallback, "", "", nil, nil, nil)
+	if err != nil {
+		xlog.Error("Anthropic stream model inference failed", "error", err)
+		return sendAnthropicError(c, 500, "api_error", fmt.Sprintf("model inference failed: %v", err))
+	}
+
+	prediction, err := predFunc()
+	if err != nil {
+		xlog.Error("Anthropic stream prediction failed", "error", err)
+		return sendAnthropicError(c, 500, "api_error", fmt.Sprintf("prediction failed: %v", err))
+	}
+
+	// Send content_block_stop event for last block if we didn't close it yet
+	if !inToolCall {
+		contentBlockStop := schema.AnthropicStreamEvent{
+			Type:  "content_block_stop",
+			Index: 0,
+		}
+		sendAnthropicSSE(c, contentBlockStop)
+	}
+
+	// Determine stop reason
+	stopReason := "end_turn"
+	if toolCallsEmitted > 0 {
+		stopReason = "tool_use"
+	}
+
+	// Send message_delta event with stop_reason
+	messageDelta := schema.AnthropicStreamEvent{
+		Type: "message_delta",
+		Delta: &schema.AnthropicStreamDelta{
+			StopReason: &stopReason,
+		},
+		Usage: &schema.AnthropicUsage{
+			OutputTokens: prediction.Usage.Completion,
+		},
+	}
+	sendAnthropicSSE(c, messageDelta)
+
+	// Send message_stop event
+	messageStop := schema.AnthropicStreamEvent{
+		Type: "message_stop",
+	}
+	sendAnthropicSSE(c, messageStop)
+
+	return nil
+}
+
+func sendAnthropicSSE(c echo.Context, event schema.AnthropicStreamEvent) {
+	data, err := json.Marshal(event)
+	if err != nil {
+		xlog.Error("Failed to marshal SSE event", "error", err)
+		return
+	}
+	fmt.Fprintf(c.Response().Writer, "event: %s\ndata: %s\n\n", event.Type, string(data))
+	c.Response().Flush()
+}
+
+func sendAnthropicError(c echo.Context, statusCode int, errorType, message string) error {
+	resp := schema.AnthropicErrorResponse{
+		Type: "error",
+		Error: schema.AnthropicError{
+			Type:    errorType,
+			Message: message,
+		},
+	}
+	return c.JSON(statusCode, resp)
+}
+
+func convertAnthropicToOpenAIMessages(input *schema.AnthropicRequest) []schema.Message {
+	var messages []schema.Message
+
+	// Add system message if present
+	if input.System != "" {
+		messages = append(messages, schema.Message{
+			Role:          "system",
+			StringContent: input.System,
+			Content:       input.System,
+		})
+	}
+
+	// Convert Anthropic messages to OpenAI format
+	for _, msg := range input.Messages {
+		openAIMsg := schema.Message{
+			Role: msg.Role,
+		}
+
+		// Handle content (can be string or array of content blocks)
+		switch content := msg.Content.(type) {
+		case string:
+			openAIMsg.StringContent = content
+			openAIMsg.Content = content
+		case []interface{}:
+			// Handle array of content blocks
+			var textContent string
+			var stringImages []string
+			var toolCalls []schema.ToolCall
+			toolCallIndex := 0
+
+			for _, block := range content {
+				if blockMap, ok := block.(map[string]interface{}); ok {
+					blockType, _ := blockMap["type"].(string)
+					switch blockType {
+					case "text":
+						if text, ok := blockMap["text"].(string); ok {
+							textContent += text
+						}
+					case "image":
+						// Handle image content
+						if source, ok := blockMap["source"].(map[string]interface{}); ok {
+							if sourceType, ok := source["type"].(string); ok && sourceType == "base64" {
+								if data, ok := source["data"].(string); ok {
+									mediaType, _ := source["media_type"].(string)
+									// Format as data URI
+									dataURI := fmt.Sprintf("data:%s;base64,%s", mediaType, data)
+									stringImages = append(stringImages, dataURI)
+								}
+							}
+						}
+					case "tool_use":
+						// Convert tool_use to ToolCall format
+						toolID, _ := blockMap["id"].(string)
+						toolName, _ := blockMap["name"].(string)
+						toolInput := blockMap["input"]
+						
+						// Serialize input to JSON string
+						inputJSON, err := json.Marshal(toolInput)
+						if err != nil {
+							xlog.Warn("Failed to marshal tool input", "error", err)
+							inputJSON = []byte("{}")
+						}
+						
+						toolCalls = append(toolCalls, schema.ToolCall{
+							Index: toolCallIndex,
+							ID:    toolID,
+							Type:  "function",
+							FunctionCall: schema.FunctionCall{
+								Name:      toolName,
+								Arguments: string(inputJSON),
+							},
+						})
+						toolCallIndex++
+					case "tool_result":
+						// Convert tool_result to a message with role "tool"
+						// This is handled by creating a separate message after this block
+						// For now, we'll add it as text content
+						toolUseID, _ := blockMap["tool_use_id"].(string)
+						isError := false
+						if isErrorPtr, ok := blockMap["is_error"].(*bool); ok && isErrorPtr != nil {
+							isError = *isErrorPtr
+						}
+						
+						var resultText string
+						if resultContent, ok := blockMap["content"]; ok {
+							switch rc := resultContent.(type) {
+							case string:
+								resultText = rc
+							case []interface{}:
+								// Array of content blocks
+								for _, cb := range rc {
+									if cbMap, ok := cb.(map[string]interface{}); ok {
+										if cbMap["type"] == "text" {
+											if text, ok := cbMap["text"].(string); ok {
+												resultText += text
+											}
+										}
+									}
+								}
+							}
+						}
+						
+						// Add tool result as a tool role message
+						// We need to handle this differently - create a new message
+						if msg.Role == "user" {
+							// Store tool result info for creating separate message
+							prefix := ""
+							if isError {
+								prefix = "Error: "
+							}
+							textContent += fmt.Sprintf("\n[Tool Result for %s]: %s%s", toolUseID, prefix, resultText)
+						}
+					}
+				}
+			}
+			openAIMsg.StringContent = textContent
+			openAIMsg.Content = textContent
+			openAIMsg.StringImages = stringImages
+			
+			// Add tool calls if present
+			if len(toolCalls) > 0 {
+				openAIMsg.ToolCalls = toolCalls
+			}
+		}
+
+		messages = append(messages, openAIMsg)
+	}
+
+	return messages
+}
+
+// convertAnthropicTools converts Anthropic tools to internal Functions format
+func convertAnthropicTools(input *schema.AnthropicRequest, cfg *config.ModelConfig) (functions.Functions, bool) {
+	if len(input.Tools) == 0 {
+		return nil, false
+	}
+	
+	var funcs functions.Functions
+	for _, tool := range input.Tools {
+		f := functions.Function{
+			Name:        tool.Name,
+			Description: tool.Description,
+			Parameters:  tool.InputSchema,
+		}
+		funcs = append(funcs, f)
+	}
+	
+	// Handle tool_choice
+	if input.ToolChoice != nil {
+		switch tc := input.ToolChoice.(type) {
+		case string:
+			// "auto", "any", or "none"
+			if tc == "any" {
+				// Force the model to use one of the tools
+				cfg.SetFunctionCallString("required")
+			} else if tc == "none" {
+				// Don't use tools
+				return nil, false
+			}
+			// "auto" is the default - let model decide
+		case map[string]interface{}:
+			// Specific tool selection: {"type": "tool", "name": "tool_name"}
+			if tcType, ok := tc["type"].(string); ok && tcType == "tool" {
+				if name, ok := tc["name"].(string); ok {
+					// Force specific tool
+					cfg.SetFunctionCallString(name)
+				}
+			}
+		}
+	}
+	
+	return funcs, len(funcs) > 0 && cfg.ShouldUseFunctions()
+}
--- a/Show More
+++ b/Show More