Address feedback from review

Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
fix: resolve duplicate MCP route registration causing 50% failure rate
2026-02-03 11:13:31 -05:00 · 2026-01-02 21:34:23 +01:00 · 2026-01-02 21:29:05 +01:00
92 changed files with 1385 additions and 8120 deletions
--- a/.github/workflows/backend.yml
+++ b/.github/workflows/backend.yml
--- a/.github/workflows/dependabot_auto.yml
+++ b/.github/workflows/dependabot_auto.yml
@@ -14,7 +14,7 @@ jobs:
    steps:
      - name: Dependabot metadata
        id: metadata
-        uses: dependabot/fetch-metadata@v2.5.0
+        uses: dependabot/fetch-metadata@v2.4.0
        with:
          github-token: "${{ secrets.GITHUB_TOKEN }}"
          skip-commit-verification: true
--- a/.github/workflows/generate_grpc_cache.yaml
+++ b/.github/workflows/generate_grpc_cache.yaml
@@ -16,7 +16,7 @@ jobs:
    strategy:
      matrix:
        include:
-          - grpc-base-image: ubuntu:24.04
+          - grpc-base-image: ubuntu:22.04
            runs-on: 'ubuntu-latest'
            platforms: 'linux/amd64,linux/arm64'
    runs-on: ${{matrix.runs-on}}
--- a/.github/workflows/generate_intel_image.yaml
+++ b/.github/workflows/generate_intel_image.yaml
@@ -15,7 +15,7 @@ jobs:
    strategy:
      matrix:
        include:
-          - base-image: intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04
+          - base-image: intel/oneapi-basekit:2025.2.0-0-devel-ubuntu22.04
            runs-on: 'arc-runner-set'
            platforms: 'linux/amd64'
    runs-on: ${{matrix.runs-on}}
@@ -53,7 +53,7 @@ jobs:
            BASE_IMAGE=${{ matrix.base-image }}
          context: .
          file: ./Dockerfile
-          tags: quay.io/go-skynet/intel-oneapi-base:24.04
+          tags: quay.io/go-skynet/intel-oneapi-base:latest
          push: true
          target: intel
          platforms: ${{ matrix.platforms }}
--- a/.github/workflows/image-pr.yml
+++ b/.github/workflows/image-pr.yml
@@ -1,95 +1,94 @@
 ---
-  name: 'build container images tests'
-  
-  on:
-    pull_request:
-  
-  concurrency:
-    group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
-    cancel-in-progress: true
-  
-  jobs:
-    image-build:
-      uses: ./.github/workflows/image_build.yml
-      with:
-        tag-latest: ${{ matrix.tag-latest }}
-        tag-suffix: ${{ matrix.tag-suffix }}
-        build-type: ${{ matrix.build-type }}
-        cuda-major-version: ${{ matrix.cuda-major-version }}
-        cuda-minor-version: ${{ matrix.cuda-minor-version }}
-        platforms: ${{ matrix.platforms }}
-        runs-on: ${{ matrix.runs-on }}
-        base-image: ${{ matrix.base-image }}
-        grpc-base-image: ${{ matrix.grpc-base-image }}
-        makeflags: ${{ matrix.makeflags }}
-        ubuntu-version: ${{ matrix.ubuntu-version }}
-      secrets:
-        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-      strategy:
-        # Pushing with all jobs in parallel
-        # eats the bandwidth of all the nodes
-        max-parallel: ${{ github.event_name != 'pull_request' && 4 || 8 }}
-        fail-fast: false
-        matrix:
-          include:
-            - build-type: 'cublas'
-              cuda-major-version: "12"
-              cuda-minor-version: "9"
-              platforms: 'linux/amd64'
-              tag-latest: 'false'
-              tag-suffix: '-gpu-nvidia-cuda-12'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:24.04"
-              makeflags: "--jobs=3 --output-sync=target"
-              ubuntu-version: '2404'
-            - build-type: 'cublas'
-              cuda-major-version: "13"
-              cuda-minor-version: "0"
-              platforms: 'linux/amd64'
-              tag-latest: 'false'
-              tag-suffix: '-gpu-nvidia-cuda-13'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:22.04"
-              makeflags: "--jobs=3 --output-sync=target"
-              ubuntu-version: '2404'
-            - build-type: 'hipblas'
-              platforms: 'linux/amd64'
-              tag-latest: 'false'
-              tag-suffix: '-hipblas'
-              base-image: "rocm/dev-ubuntu-24.04:6.4.4"
-              grpc-base-image: "ubuntu:24.04"
-              runs-on: 'ubuntu-latest'
-              makeflags: "--jobs=3 --output-sync=target"
-              ubuntu-version: '2404'
-            - build-type: 'sycl'
-              platforms: 'linux/amd64'
-              tag-latest: 'false'
-              base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
-              grpc-base-image: "ubuntu:24.04"
-              tag-suffix: 'sycl'
-              runs-on: 'ubuntu-latest'
-              makeflags: "--jobs=3 --output-sync=target"
-              ubuntu-version: '2404'
-            - build-type: 'vulkan'
-              platforms: 'linux/amd64,linux/arm64'
-              tag-latest: 'false'
-              tag-suffix: '-vulkan-core'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:24.04"
-              makeflags: "--jobs=4 --output-sync=target"
-              ubuntu-version: '2404'
-            - build-type: 'cublas'
-              cuda-major-version: "13"
-              cuda-minor-version: "0"
-              platforms: 'linux/arm64'
-              tag-latest: 'false'
-              tag-suffix: '-nvidia-l4t-arm64-cuda-13'
-              base-image: "ubuntu:24.04"
-              runs-on: 'ubuntu-24.04-arm'
-              makeflags: "--jobs=4 --output-sync=target"
-              skip-drivers: 'false'
-              ubuntu-version: '2404'
-  
+name: 'build container images tests'
+
+on:
+  pull_request:
+
+concurrency:
+  group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
+  cancel-in-progress: true
+
+jobs:
+  image-build:
+    uses: ./.github/workflows/image_build.yml
+    with:
+      tag-latest: ${{ matrix.tag-latest }}
+      tag-suffix: ${{ matrix.tag-suffix }}
+      build-type: ${{ matrix.build-type }}
+      cuda-major-version: ${{ matrix.cuda-major-version }}
+      cuda-minor-version: ${{ matrix.cuda-minor-version }}
+      platforms: ${{ matrix.platforms }}
+      runs-on: ${{ matrix.runs-on }}
+      base-image: ${{ matrix.base-image }}
+      grpc-base-image: ${{ matrix.grpc-base-image }}
+      makeflags: ${{ matrix.makeflags }}
+      ubuntu-version: ${{ matrix.ubuntu-version }}
+    secrets:
+      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+    strategy:
+      # Pushing with all jobs in parallel
+      # eats the bandwidth of all the nodes
+      max-parallel: ${{ github.event_name != 'pull_request' && 4 || 8 }}
+      fail-fast: false
+      matrix:
+        include:
+          - build-type: 'cublas'
+            cuda-major-version: "12"
+            cuda-minor-version: "0"
+            platforms: 'linux/amd64'
+            tag-latest: 'false'
+            tag-suffix: '-gpu-nvidia-cuda-12'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            makeflags: "--jobs=3 --output-sync=target"
+            ubuntu-version: '2204'
+          - build-type: 'cublas'
+            cuda-major-version: "13"
+            cuda-minor-version: "0"
+            platforms: 'linux/amd64'
+            tag-latest: 'false'
+            tag-suffix: '-gpu-nvidia-cuda-13'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            makeflags: "--jobs=3 --output-sync=target"
+            ubuntu-version: '2204'
+          - build-type: 'hipblas'
+            platforms: 'linux/amd64'
+            tag-latest: 'false'
+            tag-suffix: '-hipblas'
+            base-image: "rocm/dev-ubuntu-22.04:6.4.3"
+            grpc-base-image: "ubuntu:22.04"
+            runs-on: 'ubuntu-latest'
+            makeflags: "--jobs=3 --output-sync=target"
+            ubuntu-version: '2204'
+          - build-type: 'sycl'
+            platforms: 'linux/amd64'
+            tag-latest: 'false'
+            base-image: "quay.io/go-skynet/intel-oneapi-base:latest"
+            grpc-base-image: "ubuntu:22.04"
+            tag-suffix: 'sycl'
+            runs-on: 'ubuntu-latest'
+            makeflags: "--jobs=3 --output-sync=target"
+            ubuntu-version: '2204'
+          - build-type: 'vulkan'
+            platforms: 'linux/amd64'
+            tag-latest: 'false'
+            tag-suffix: '-vulkan-core'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            makeflags: "--jobs=4 --output-sync=target"
+            ubuntu-version: '2204'
+          - build-type: 'cublas'
+            cuda-major-version: "13"
+            cuda-minor-version: "0"
+            platforms: 'linux/arm64'
+            tag-latest: 'false'
+            tag-suffix: '-nvidia-l4t-arm64-cuda-13'
+            base-image: "ubuntu:24.04"
+            runs-on: 'ubuntu-24.04-arm'
+            makeflags: "--jobs=4 --output-sync=target"
+            skip-drivers: 'false'
+            ubuntu-version: '2404'
--- a/.github/workflows/image.yml
+++ b/.github/workflows/image.yml
@@ -1,187 +1,187 @@
 ---
-  name: 'build container images'
-  
-  on:
-    push:
-      branches:
-        - master
-      tags:
-        - '*'
-  
-  concurrency:
-    group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
-    cancel-in-progress: true
-  
-  jobs:
-    hipblas-jobs:
-      uses: ./.github/workflows/image_build.yml
-      with:
-        tag-latest: ${{ matrix.tag-latest }}
-        tag-suffix: ${{ matrix.tag-suffix }}
-        build-type: ${{ matrix.build-type }}
-        cuda-major-version: ${{ matrix.cuda-major-version }}
-        cuda-minor-version: ${{ matrix.cuda-minor-version }}
-        platforms: ${{ matrix.platforms }}
-        runs-on: ${{ matrix.runs-on }}
-        base-image: ${{ matrix.base-image }}
-        grpc-base-image: ${{ matrix.grpc-base-image }}
-        aio: ${{ matrix.aio }}
-        makeflags: ${{ matrix.makeflags }}
-        ubuntu-version: ${{ matrix.ubuntu-version }}
-        ubuntu-codename: ${{ matrix.ubuntu-codename }}
-      secrets:
-        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-      strategy:
-        matrix:
-          include:
-            - build-type: 'hipblas'
-              platforms: 'linux/amd64'
-              tag-latest: 'auto'
-              tag-suffix: '-gpu-hipblas'
-              base-image: "rocm/dev-ubuntu-24.04:6.4.4"
-              grpc-base-image: "ubuntu:24.04"
-              runs-on: 'ubuntu-latest'
-              makeflags: "--jobs=3 --output-sync=target"
-              aio: "-aio-gpu-hipblas"
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-  
-    core-image-build:
-      uses: ./.github/workflows/image_build.yml
-      with:
-        tag-latest: ${{ matrix.tag-latest }}
-        tag-suffix: ${{ matrix.tag-suffix }}
-        build-type: ${{ matrix.build-type }}
-        cuda-major-version: ${{ matrix.cuda-major-version }}
-        cuda-minor-version: ${{ matrix.cuda-minor-version }}
-        platforms: ${{ matrix.platforms }}
-        runs-on: ${{ matrix.runs-on }}
-        aio: ${{ matrix.aio }}
-        base-image: ${{ matrix.base-image }}
-        grpc-base-image: ${{ matrix.grpc-base-image }}
-        makeflags: ${{ matrix.makeflags }}
-        skip-drivers: ${{ matrix.skip-drivers }}
-        ubuntu-version: ${{ matrix.ubuntu-version }}
-        ubuntu-codename: ${{ matrix.ubuntu-codename }}
-      secrets:
-        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-      strategy:
-        #max-parallel: ${{ github.event_name != 'pull_request' && 2 || 4 }}
-        matrix:
-          include:
-            - build-type: ''
-              platforms: 'linux/amd64,linux/arm64'
-              tag-latest: 'auto'
-              tag-suffix: ''
-              base-image: "ubuntu:24.04"
-              runs-on: 'ubuntu-latest'
-              aio: "-aio-cpu"
-              makeflags: "--jobs=4 --output-sync=target"
-              skip-drivers: 'false'
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-            - build-type: 'cublas'
-              cuda-major-version: "12"
-              cuda-minor-version: "9"
-              platforms: 'linux/amd64'
-              tag-latest: 'auto'
-              tag-suffix: '-gpu-nvidia-cuda-12'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:24.04"
-              skip-drivers: 'false'
-              makeflags: "--jobs=4 --output-sync=target"
-              aio: "-aio-gpu-nvidia-cuda-12"
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-            - build-type: 'cublas'
-              cuda-major-version: "13"
-              cuda-minor-version: "0"
-              platforms: 'linux/amd64'
-              tag-latest: 'auto'
-              tag-suffix: '-gpu-nvidia-cuda-13'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:22.04"
-              skip-drivers: 'false'
-              makeflags: "--jobs=4 --output-sync=target"
-              aio: "-aio-gpu-nvidia-cuda-13"
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-            - build-type: 'vulkan'
-              platforms: 'linux/amd64,linux/arm64'
-              tag-latest: 'auto'
-              tag-suffix: '-gpu-vulkan'
-              runs-on: 'ubuntu-latest'
-              base-image: "ubuntu:24.04"
-              skip-drivers: 'false'
-              makeflags: "--jobs=4 --output-sync=target"
-              aio: "-aio-gpu-vulkan"
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-            - build-type: 'intel'
-              platforms: 'linux/amd64'
-              tag-latest: 'auto'
-              base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"
-              grpc-base-image: "ubuntu:24.04"
-              tag-suffix: '-gpu-intel'
-              runs-on: 'ubuntu-latest'
-              makeflags: "--jobs=3 --output-sync=target"
-              aio: "-aio-gpu-intel"
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-  
-    gh-runner:
-      uses: ./.github/workflows/image_build.yml
-      with:
-        tag-latest: ${{ matrix.tag-latest }}
-        tag-suffix: ${{ matrix.tag-suffix }}
-        build-type: ${{ matrix.build-type }}
-        cuda-major-version: ${{ matrix.cuda-major-version }}
-        cuda-minor-version: ${{ matrix.cuda-minor-version }}
-        platforms: ${{ matrix.platforms }}
-        runs-on: ${{ matrix.runs-on }}
-        aio: ${{ matrix.aio }}
-        base-image: ${{ matrix.base-image }}
-        grpc-base-image: ${{ matrix.grpc-base-image }}
-        makeflags: ${{ matrix.makeflags }}
-        skip-drivers: ${{ matrix.skip-drivers }}
-        ubuntu-version: ${{ matrix.ubuntu-version }}
-        ubuntu-codename: ${{ matrix.ubuntu-codename }}
-      secrets:
-        dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
-        dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
-        quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
-        quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
-      strategy:
-        matrix:
-          include:
-            - build-type: 'cublas'
-              cuda-major-version: "12"
-              cuda-minor-version: "0"
-              platforms: 'linux/arm64'
-              tag-latest: 'auto'
-              tag-suffix: '-nvidia-l4t-arm64'
-              base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
-              runs-on: 'ubuntu-24.04-arm'
-              makeflags: "--jobs=4 --output-sync=target"
-              skip-drivers: 'true'
-              ubuntu-version: "2204"
-              ubuntu-codename: 'jammy'
-            - build-type: 'cublas'
-              cuda-major-version: "13"
-              cuda-minor-version: "0"
-              platforms: 'linux/arm64'
-              tag-latest: 'auto'
-              tag-suffix: '-nvidia-l4t-arm64-cuda-13'
-              base-image: "ubuntu:24.04"
-              runs-on: 'ubuntu-24.04-arm'
-              makeflags: "--jobs=4 --output-sync=target"
-              skip-drivers: 'false'
-              ubuntu-version: '2404'
-              ubuntu-codename: 'noble'
-  
+name: 'build container images'
+
+on:
+  push:
+    branches:
+      - master
+    tags:
+      - '*'
+
+concurrency:
+  group: ci-${{ github.head_ref || github.ref }}-${{ github.repository }}
+  cancel-in-progress: true
+
+jobs:
+  hipblas-jobs:
+    uses: ./.github/workflows/image_build.yml
+    with:
+      tag-latest: ${{ matrix.tag-latest }}
+      tag-suffix: ${{ matrix.tag-suffix }}
+      build-type: ${{ matrix.build-type }}
+      cuda-major-version: ${{ matrix.cuda-major-version }}
+      cuda-minor-version: ${{ matrix.cuda-minor-version }}
+      platforms: ${{ matrix.platforms }}
+      runs-on: ${{ matrix.runs-on }}
+      base-image: ${{ matrix.base-image }}
+      grpc-base-image: ${{ matrix.grpc-base-image }}
+      aio: ${{ matrix.aio }}
+      makeflags: ${{ matrix.makeflags }}
+      ubuntu-version: ${{ matrix.ubuntu-version }}
+    secrets:
+      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+    strategy:
+      matrix:
+        include:
+          - build-type: 'hipblas'
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            tag-suffix: '-gpu-hipblas'
+            base-image: "rocm/dev-ubuntu-22.04:6.4.3"
+            grpc-base-image: "ubuntu:22.04"
+            runs-on: 'ubuntu-latest'
+            makeflags: "--jobs=3 --output-sync=target"
+            aio: "-aio-gpu-hipblas"
+            ubuntu-version: '2204'
+
+  core-image-build:
+    uses: ./.github/workflows/image_build.yml
+    with:
+      tag-latest: ${{ matrix.tag-latest }}
+      tag-suffix: ${{ matrix.tag-suffix }}
+      build-type: ${{ matrix.build-type }}
+      cuda-major-version: ${{ matrix.cuda-major-version }}
+      cuda-minor-version: ${{ matrix.cuda-minor-version }}
+      platforms: ${{ matrix.platforms }}
+      runs-on: ${{ matrix.runs-on }}
+      aio: ${{ matrix.aio }}
+      base-image: ${{ matrix.base-image }}
+      grpc-base-image: ${{ matrix.grpc-base-image }}
+      makeflags: ${{ matrix.makeflags }}
+      skip-drivers: ${{ matrix.skip-drivers }}
+      ubuntu-version: ${{ matrix.ubuntu-version }}
+    secrets:
+      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+    strategy:
+      #max-parallel: ${{ github.event_name != 'pull_request' && 2 || 4 }}
+      matrix:
+        include:
+          - build-type: ''
+            platforms: 'linux/amd64,linux/arm64'
+            tag-latest: 'auto'
+            tag-suffix: ''
+            base-image: "ubuntu:22.04"
+            runs-on: 'ubuntu-latest'
+            aio: "-aio-cpu"
+            makeflags: "--jobs=4 --output-sync=target"
+            skip-drivers: 'false'
+            ubuntu-version: '2204'
+          - build-type: 'cublas'
+            cuda-major-version: "11"
+            cuda-minor-version: "7"
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            tag-suffix: '-gpu-nvidia-cuda-11'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            makeflags: "--jobs=4 --output-sync=target"
+            skip-drivers: 'false'
+            aio: "-aio-gpu-nvidia-cuda-11"
+            ubuntu-version: '2204'
+          - build-type: 'cublas'
+            cuda-major-version: "12"
+            cuda-minor-version: "0"
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            tag-suffix: '-gpu-nvidia-cuda-12'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            skip-drivers: 'false'
+            makeflags: "--jobs=4 --output-sync=target"
+            aio: "-aio-gpu-nvidia-cuda-12"
+            ubuntu-version: '2204'
+          - build-type: 'cublas'
+            cuda-major-version: "13"
+            cuda-minor-version: "0"
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            tag-suffix: '-gpu-nvidia-cuda-13'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            skip-drivers: 'false'
+            makeflags: "--jobs=4 --output-sync=target"
+            aio: "-aio-gpu-nvidia-cuda-13"
+            ubuntu-version: '2204'
+          - build-type: 'vulkan'
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            tag-suffix: '-gpu-vulkan'
+            runs-on: 'ubuntu-latest'
+            base-image: "ubuntu:22.04"
+            skip-drivers: 'false'
+            makeflags: "--jobs=4 --output-sync=target"
+            aio: "-aio-gpu-vulkan"
+            ubuntu-version: '2204'
+          - build-type: 'intel'
+            platforms: 'linux/amd64'
+            tag-latest: 'auto'
+            base-image: "quay.io/go-skynet/intel-oneapi-base:latest"
+            grpc-base-image: "ubuntu:22.04"
+            tag-suffix: '-gpu-intel'
+            runs-on: 'ubuntu-latest'
+            makeflags: "--jobs=3 --output-sync=target"
+            aio: "-aio-gpu-intel"
+            ubuntu-version: '2204'
+
+  gh-runner:
+    uses: ./.github/workflows/image_build.yml
+    with:
+      tag-latest: ${{ matrix.tag-latest }}
+      tag-suffix: ${{ matrix.tag-suffix }}
+      build-type: ${{ matrix.build-type }}
+      cuda-major-version: ${{ matrix.cuda-major-version }}
+      cuda-minor-version: ${{ matrix.cuda-minor-version }}
+      platforms: ${{ matrix.platforms }}
+      runs-on: ${{ matrix.runs-on }}
+      aio: ${{ matrix.aio }}
+      base-image: ${{ matrix.base-image }}
+      grpc-base-image: ${{ matrix.grpc-base-image }}
+      makeflags: ${{ matrix.makeflags }}
+      skip-drivers: ${{ matrix.skip-drivers }}
+      ubuntu-version: ${{ matrix.ubuntu-version }}
+    secrets:
+      dockerUsername: ${{ secrets.DOCKERHUB_USERNAME }}
+      dockerPassword: ${{ secrets.DOCKERHUB_PASSWORD }}
+      quayUsername: ${{ secrets.LOCALAI_REGISTRY_USERNAME }}
+      quayPassword: ${{ secrets.LOCALAI_REGISTRY_PASSWORD }}
+    strategy:
+      matrix:
+        include:
+          - build-type: 'cublas'
+            cuda-major-version: "12"
+            cuda-minor-version: "0"
+            platforms: 'linux/arm64'
+            tag-latest: 'auto'
+            tag-suffix: '-nvidia-l4t-arm64'
+            base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0"
+            runs-on: 'ubuntu-24.04-arm'
+            makeflags: "--jobs=4 --output-sync=target"
+            skip-drivers: 'true'
+            ubuntu-version: "2204"
+          - build-type: 'cublas'
+            cuda-major-version: "13"
+            cuda-minor-version: "0"
+            platforms: 'linux/arm64'
+            tag-latest: 'auto'
+            tag-suffix: '-nvidia-l4t-arm64-cuda-13'
+            base-image: "ubuntu:24.04"
+            runs-on: 'ubuntu-24.04-arm'
+            makeflags: "--jobs=4 --output-sync=target"
+            skip-drivers: 'false'
+            ubuntu-version: '2404'
--- a/.github/workflows/image_build.yml
+++ b/.github/workflows/image_build.yml
@@ -23,7 +23,7 @@ on:
        type: string
      cuda-minor-version:
        description: 'CUDA minor version'
-        default: "9"
+        default: "4"
        type: string
      platforms:
        description: 'Platforms'
@@ -61,11 +61,6 @@ on:
        required: false
        default: '2204'
        type: string
-      ubuntu-codename:
-        description: 'Ubuntu codename'
-        required: false
-        default: 'noble'
-        type: string
    secrets:
      dockerUsername:
        required: true
@@ -249,7 +244,6 @@ jobs:
            MAKEFLAGS=${{ inputs.makeflags }}
            SKIP_DRIVERS=${{ inputs.skip-drivers }}
            UBUNTU_VERSION=${{ inputs.ubuntu-version }}
-            UBUNTU_CODENAME=${{ inputs.ubuntu-codename }}
          context: .
          file: ./Dockerfile
          cache-from: type=gha
@@ -278,7 +272,6 @@ jobs:
            MAKEFLAGS=${{ inputs.makeflags }}
            SKIP_DRIVERS=${{ inputs.skip-drivers }}
            UBUNTU_VERSION=${{ inputs.ubuntu-version }}
-            UBUNTU_CODENAME=${{ inputs.ubuntu-codename }}
          context: .
          file: ./Dockerfile
          cache-from: type=gha
--- a/.github/workflows/test-extra.yml
+++ b/.github/workflows/test-extra.yml
@@ -247,22 +247,3 @@ jobs:
        run: |
          make --jobs=5 --output-sync=target -C backend/python/coqui
          make --jobs=5 --output-sync=target -C backend/python/coqui test
-  tests-moonshine:
-    runs-on: ubuntu-latest
-    steps:
-      - name: Clone
-        uses: actions/checkout@v6
-        with:
-          submodules: true
-      - name: Dependencies
-        run: |
-          sudo apt-get update
-          sudo apt-get install build-essential ffmpeg
-          sudo apt-get install -y ca-certificates cmake curl patch python3-pip
-          # Install UV
-          curl -LsSf https://astral.sh/uv/install.sh | sh
-          pip install --user --no-cache-dir grpcio-tools==1.64.1
-      - name: Test moonshine
-        run: |
-          make --jobs=5 --output-sync=target -C backend/python/moonshine
-          make --jobs=5 --output-sync=target -C backend/python/moonshine test
--- a/.gitignore
+++ b/.gitignore
@@ -25,7 +25,6 @@ go-bert
 # LocalAI build binary
 LocalAI
 /local-ai
-/local-ai-launcher
 # prevent above rules from omitting the helm chart
 !charts/*
 # prevent above rules from omitting the api/localai folder
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -2,163 +2,6 @@

 Building and testing the project depends on the components involved and the platform where development is taking place. Due to the amount of context required it's usually best not to try building or testing the project unless the user requests it. If you must build the project then inspect the Makefile in the project root and the Makefiles of any backends that are effected by changes you are making. In addition the workflows in .github/workflows can be used as a reference when it is unclear how to build or test a component. The primary Makefile contains targets for building inside or outside Docker, if the user has not previously specified a preference then ask which they would like to use.

-## Building a specified backend
-
-Let's say the user wants to build a particular backend for a given platform. For example let's say they want to build bark for ROCM/hipblas
-
- The Makefile has targets like `docker-build-bark` created with `generate-docker-build-target` at the time of writing. Recently added backends may require a new target.
- At a minimum we need to set the BUILD_TYPE, BASE_IMAGE build-args
-  - Use .github/workflows/backend.yml as a reference it lists the needed args in the `include` job strategy matrix
-  - l4t and cublas also requires the CUDA major and minor version
- You can pretty print a command like `DOCKER_MAKEFLAGS=-j$(nproc --ignore=1) BUILD_TYPE=hipblas BASE_IMAGE=rocm/dev-ubuntu-24.04:6.4.4 make docker-build-bark`
- Unless the user specifies that they want you to run the command, then just print it because not all agent frontends handle long running jobs well and the output may overflow your context
- The user may say they want to build AMD or ROCM instead of hipblas, or Intel instead of SYCL or NVIDIA insted of l4t or cublas. Ask for confirmation if there is ambiguity.
- Sometimes the user may need extra parameters to be added to `docker build` (e.g. `--platform` for cross-platform builds or `--progress` to view the full logs), in which case you can generate the `docker build` command directly.
-
-## Adding a New Backend
-
-When adding a new backend to LocalAI, you need to update several files to ensure the backend is properly built, tested, and registered. Here's a step-by-step guide based on the pattern used for adding backends like `moonshine`:
-
-### 1. Create Backend Directory Structure
-
-Create the backend directory under the appropriate location:
- **Python backends**: `backend/python/<backend-name>/`
- **Go backends**: `backend/go/<backend-name>/`
- **C++ backends**: `backend/cpp/<backend-name>/`
-
-For Python backends, you'll typically need:
- `backend.py` - Main gRPC server implementation
- `Makefile` - Build configuration
- `install.sh` - Installation script for dependencies
- `protogen.sh` - Protocol buffer generation script
- `requirements.txt` - Python dependencies
- `run.sh` - Runtime script
- `test.py` / `test.sh` - Test files
-
-### 2. Add Build Configurations to `.github/workflows/backend.yml`
-
-Add build matrix entries for each platform/GPU type you want to support. Look at similar backends (e.g., `chatterbox`, `faster-whisper`) for reference.
-
-**Placement in file:**
- CPU builds: Add after other CPU builds (e.g., after `cpu-chatterbox`)
- CUDA 12 builds: Add after other CUDA 12 builds (e.g., after `gpu-nvidia-cuda-12-chatterbox`)
- CUDA 13 builds: Add after other CUDA 13 builds (e.g., after `gpu-nvidia-cuda-13-chatterbox`)
-
-**Additional build types you may need:**
- ROCm/HIP: Use `build-type: 'hipblas'` with `base-image: "rocm/dev-ubuntu-24.04:6.4.4"`
- Intel/SYCL: Use `build-type: 'intel'` or `build-type: 'sycl_f16'`/`sycl_f32` with `base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04"`
- L4T (ARM): Use `build-type: 'l4t'` with `platforms: 'linux/arm64'` and `runs-on: 'ubuntu-24.04-arm'`
-
-### 3. Add Backend Metadata to `backend/index.yaml`
-
-**Step 3a: Add Meta Definition**
-
-Add a YAML anchor definition in the `## metas` section (around line 2-300). Look for similar backends to use as a template such as `diffusers` or `chatterbox`
-
-**Step 3b: Add Image Entries**
-
-Add image entries at the end of the file, following the pattern of similar backends such as `diffusers` or `chatterbox`. Include both `latest` (production) and `master` (development) tags.
-
-### 4. Update the Makefile
-
-The Makefile needs to be updated in several places to support building and testing the new backend:
-
-**Step 4a: Add to `.NOTPARALLEL`**
-
-Add `backends/<backend-name>` to the `.NOTPARALLEL` line (around line 2) to prevent parallel execution conflicts:
-
-```makefile
-.NOTPARALLEL: ... backends/<backend-name>
-```
-
-**Step 4b: Add to `prepare-test-extra`**
-
-Add the backend to the `prepare-test-extra` target (around line 312) to prepare it for testing:
-
-```makefile
-prepare-test-extra: protogen-python
-	...
-	$(MAKE) -C backend/python/<backend-name>
-```
-
-**Step 4c: Add to `test-extra`**
-
-Add the backend to the `test-extra` target (around line 319) to run its tests:
-
-```makefile
-test-extra: prepare-test-extra
-	...
-	$(MAKE) -C backend/python/<backend-name> test
-```
-
-**Step 4d: Add Backend Definition**
-
-Add a backend definition variable in the backend definitions section (around line 428-457). The format depends on the backend type:
-
-**For Python backends with root context** (like `faster-whisper`, `bark`):
-```makefile
-BACKEND_<BACKEND_NAME> = <backend-name>|python|.|false|true
-```
-
-**For Python backends with `./backend` context** (like `chatterbox`, `moonshine`):
-```makefile
-BACKEND_<BACKEND_NAME> = <backend-name>|python|./backend|false|true
-```
-
-**For Go backends**:
-```makefile
-BACKEND_<BACKEND_NAME> = <backend-name>|golang|.|false|true
-```
-
-**Step 4e: Generate Docker Build Target**
-
-Add an eval call to generate the docker-build target (around line 480-501):
-
-```makefile
-$(eval $(call generate-docker-build-target,$(BACKEND_<BACKEND_NAME>)))
-```
-
-**Step 4f: Add to `docker-build-backends`**
-
-Add `docker-build-<backend-name>` to the `docker-build-backends` target (around line 507):
-
-```makefile
-docker-build-backends: ... docker-build-<backend-name>
-```
-
-**Determining the Context:**
-
- If the backend is in `backend/python/<backend-name>/` and uses `./backend` as context in the workflow file, use `./backend` context
- If the backend is in `backend/python/<backend-name>/` but uses `.` as context in the workflow file, use `.` context
- Check similar backends to determine the correct context
-
-### 5. Verification Checklist
-
-After adding a new backend, verify:
-
- [ ] Backend directory structure is complete with all necessary files
- [ ] Build configurations added to `.github/workflows/backend.yml` for all desired platforms
- [ ] Meta definition added to `backend/index.yaml` in the `## metas` section
- [ ] Image entries added to `backend/index.yaml` for all build variants (latest + development)
- [ ] Tag suffixes match between workflow file and index.yaml
- [ ] Makefile updated with all 6 required changes (`.NOTPARALLEL`, `prepare-test-extra`, `test-extra`, backend definition, docker-build target eval, `docker-build-backends`)
- [ ] No YAML syntax errors (check with linter)
- [ ] No Makefile syntax errors (check with linter)
- [ ] Follows the same pattern as similar backends (e.g., if it's a transcription backend, follow `faster-whisper` pattern)
-
-### 6. Example: Adding a Python Backend
-
-For reference, when `moonshine` was added:
- **Files created**: `backend/python/moonshine/{backend.py, Makefile, install.sh, protogen.sh, requirements.txt, run.sh, test.py, test.sh}`
- **Workflow entries**: 3 build configurations (CPU, CUDA 12, CUDA 13)
- **Index entries**: 1 meta definition + 6 image entries (cpu, cuda12, cuda13 × latest/development)
- **Makefile updates**: 
-  - Added to `.NOTPARALLEL` line
-  - Added to `prepare-test-extra` and `test-extra` targets
-  - Added `BACKEND_MOONSHINE = moonshine|python|./backend|false|true`
-  - Added eval for docker-build target generation
-  - Added `docker-build-moonshine` to `docker-build-backends`
-
 # Coding style

 - The project has the following .editorconfig
@@ -234,49 +77,3 @@ When fixing compilation errors after upstream changes:
 - HTTP server uses `server_routes` with HTTP handlers
 - Both use the same `server_context` and task queue infrastructure
 - gRPC methods: `LoadModel`, `Predict`, `PredictStream`, `Embedding`, `Rerank`, `TokenizeString`, `GetMetrics`, `Health`
-
-## Tool Call Parsing Maintenance
-
-When working on JSON/XML tool call parsing functionality, always check llama.cpp for reference implementation and updates:
-
-### Checking for XML Parsing Changes
-
-1. **Review XML Format Definitions**: Check `llama.cpp/common/chat-parser-xml-toolcall.h` for `xml_tool_call_format` struct changes
-2. **Review Parsing Logic**: Check `llama.cpp/common/chat-parser-xml-toolcall.cpp` for parsing algorithm updates
-3. **Review Format Presets**: Check `llama.cpp/common/chat-parser.cpp` for new XML format presets (search for `xml_tool_call_format form`)
-4. **Review Model Lists**: Check `llama.cpp/common/chat.h` for `COMMON_CHAT_FORMAT_*` enum values that use XML parsing:
-   - `COMMON_CHAT_FORMAT_GLM_4_5`
-   - `COMMON_CHAT_FORMAT_MINIMAX_M2`
-   - `COMMON_CHAT_FORMAT_KIMI_K2`
-   - `COMMON_CHAT_FORMAT_QWEN3_CODER_XML`
-   - `COMMON_CHAT_FORMAT_APRIEL_1_5`
-   - `COMMON_CHAT_FORMAT_XIAOMI_MIMO`
-   - Any new formats added
-
-### Model Configuration Options
-
-Always check `llama.cpp` for new model configuration options that should be supported in LocalAI:
-
-1. **Check Server Context**: Review `llama.cpp/tools/server/server-context.cpp` for new parameters
-2. **Check Chat Params**: Review `llama.cpp/common/chat.h` for `common_chat_params` struct changes
-3. **Check Server Options**: Review `llama.cpp/tools/server/server.cpp` for command-line argument changes
-4. **Examples of options to check**:
-   - `ctx_shift` - Context shifting support
-   - `parallel_tool_calls` - Parallel tool calling
-   - `reasoning_format` - Reasoning format options
-   - Any new flags or parameters
-
-### Implementation Guidelines
-
-1. **Feature Parity**: Always aim for feature parity with llama.cpp's implementation
-2. **Test Coverage**: Add tests for new features matching llama.cpp's behavior
-3. **Documentation**: Update relevant documentation when adding new formats or options
-4. **Backward Compatibility**: Ensure changes don't break existing functionality
-
-### Files to Monitor
-
- `llama.cpp/common/chat-parser-xml-toolcall.h` - Format definitions
- `llama.cpp/common/chat-parser-xml-toolcall.cpp` - Parsing logic
- `llama.cpp/common/chat-parser.cpp` - Format presets and model-specific handlers
- `llama.cpp/common/chat.h` - Format enums and parameter structures
- `llama.cpp/tools/server/server-context.cpp` - Server configuration options
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -78,20 +78,6 @@ LOCALAI_IMAGE_TAG=test LOCALAI_IMAGE=local-ai-aio make run-e2e-aio

 We are welcome the contribution of the documents, please open new PR or create a new issue. The documentation is available under `docs/` https://github.com/mudler/LocalAI/tree/master/docs

-### Gallery YAML Schema
-
-LocalAI provides a JSON Schema for gallery model YAML files at:
-
-`core/schema/gallery-model.schema.json`
-
-This schema mirrors the internal gallery model configuration and can be used by editors (such as VS Code) to enable autocomplete, validation, and inline documentation when creating or modifying gallery files.
-
-To use it with the YAML language server, add the following comment at the top of a gallery YAML file:
-
-```yaml
-# yaml-language-server: $schema=../core/schema/gallery-model.schema.json
-```
-
 ## Community and Communication

 - You can reach out via the Github issue tracker.
--- a/59
+++ b/59
@@ -1,7 +1,6 @@
-ARG BASE_IMAGE=ubuntu:24.04
+ARG BASE_IMAGE=ubuntu:22.04
 ARG GRPC_BASE_IMAGE=${BASE_IMAGE}
 ARG INTEL_BASE_IMAGE=${BASE_IMAGE}
-ARG UBUNTU_CODENAME=noble

 FROM ${BASE_IMAGE} AS requirements

@@ -10,7 +9,7 @@ ENV DEBIAN_FRONTEND=noninteractive
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        ca-certificates curl wget espeak-ng libgomp1 \
-        ffmpeg libopenblas0 libopenblas-dev && \
+        ffmpeg && \
    apt-get clean && \
    rm -rf /var/lib/apt/lists/*

@@ -24,7 +23,7 @@ ARG SKIP_DRIVERS=false
 ARG TARGETARCH
 ARG TARGETVARIANT
 ENV BUILD_TYPE=${BUILD_TYPE}
-ARG UBUNTU_VERSION=2404
+ARG UBUNTU_VERSION=2204

 RUN mkdir -p /run/localai
 RUN echo "default" > /run/localai/capability
@@ -35,45 +34,11 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
-            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
-            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
-            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
-            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
-            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils mesa-vulkan-drivers
-        if [ "amd64" = "$TARGETARCH" ]; then
-            wget "https://sdk.lunarg.com/sdk/download/1.4.328.1/linux/vulkansdk-linux-x86_64-1.4.328.1.tar.xz" && \
-            tar -xf vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            rm vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            mkdir -p /opt/vulkan-sdk && \
-            mv 1.4.328.1 /opt/vulkan-sdk/ && \
-            cd /opt/vulkan-sdk/1.4.328.1 && \
-            ./vulkansdk --no-deps --maxjobs \
-                vulkan-loader \
-                vulkan-validationlayers \
-                vulkan-extensionlayer \
-                vulkan-tools \
-                shaderc && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/bin/* /usr/bin/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/include/* /usr/include/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/share/* /usr/share/ && \
-            rm -rf /opt/vulkan-sdk
-        fi
-        if [ "arm64" = "$TARGETARCH" ]; then
-            mkdir vulkan && cd vulkan && \
-            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
-            tar -xvf vulkan-sdk.tar.xz && \
-            rm vulkan-sdk.tar.xz && \
-            cd 1.4.335.0 && \
-            cp -rfv aarch64/bin/* /usr/bin/ && \
-            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
-            cp -rfv aarch64/include/* /usr/include/ && \
-            cp -rfv aarch64/share/* /usr/share/ && \
-            cd ../.. && \
-            rm -rf vulkan
-        fi
-        ldconfig && \
+        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
+        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
+        apt-get update && \
+        apt-get install -y \
+            vulkan-sdk && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/* && \
        echo "vulkan" > /run/localai/capability
@@ -176,12 +141,13 @@ ENV PATH=/opt/rocm/bin:${PATH}
 # The requirements-core target is common to all images.  It should not be placed in requirements-core unless every single build will use it.
 FROM requirements-drivers AS build-requirements

-ARG GO_VERSION=1.25.4
-ARG CMAKE_VERSION=3.31.10
+ARG GO_VERSION=1.22.6
+ARG CMAKE_VERSION=3.26.4
 ARG CMAKE_FROM_SOURCE=false
 ARG TARGETARCH
 ARG TARGETVARIANT

+
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        build-essential \
@@ -238,10 +204,9 @@ WORKDIR /build
 # https://community.intel.com/t5/Intel-oneAPI-Math-Kernel-Library/APT-Repository-not-working-signatures-invalid/m-p/1599436/highlight/true#M36143
 # This is a temporary workaround until Intel fixes their repository
 FROM ${INTEL_BASE_IMAGE} AS intel
-ARG UBUNTU_CODENAME=noble
 RUN wget -qO - https://repositories.intel.com/gpu/intel-graphics.key | \
 gpg --yes --dearmor --output /usr/share/keyrings/intel-graphics.gpg
-RUN echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] https://repositories.intel.com/gpu/ubuntu ${UBUNTU_CODENAME}/lts/2350 unified" > /etc/apt/sources.list.d/intel-graphics.list
+RUN echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] https://repositories.intel.com/gpu/ubuntu jammy/lts/2350 unified" > /etc/apt/sources.list.d/intel-graphics.list
 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
        intel-oneapi-runtime-libs && \
--- a/Dockerfile.aio
+++ b/Dockerfile.aio
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:24.04
+ARG BASE_IMAGE=ubuntu:22.04

 FROM ${BASE_IMAGE} 

--- a/293
+++ b/293
@@ -1,6 +1,3 @@
-# Disable parallel execution for backend builds
-.NOTPARALLEL: backends/diffusers backends/llama-cpp backends/piper backends/stablediffusion-ggml backends/whisper backends/faster-whisper backends/silero-vad backends/local-store backends/huggingface backends/rfdetr backends/kitten-tts backends/kokoro backends/chatterbox backends/llama-cpp-darwin backends/neutts build-darwin-python-backend build-darwin-go-backend backends/mlx backends/diffuser-darwin backends/mlx-vlm backends/mlx-audio backends/stablediffusion-ggml-darwin backends/vllm backends/moonshine
-
 GOCMD=go
 GOTEST=$(GOCMD) test
 GOVET=$(GOCMD) vet
@@ -9,14 +6,10 @@ LAUNCHER_BINARY_NAME=local-ai-launcher

 CUDA_MAJOR_VERSION?=13
 CUDA_MINOR_VERSION?=0
-UBUNTU_VERSION?=2204
-UBUNTU_CODENAME?=noble

 GORELEASER?=

 export BUILD_TYPE?=
-export CUDA_MAJOR_VERSION?=12
-export CUDA_MINOR_VERSION?=9

 GO_TAGS?=
 BUILD_ID?=
@@ -162,17 +155,7 @@ test: test-models/testmodel.ggml protogen-go
 ########################################################

 docker-build-aio:
-	docker build \
-		--build-arg MAKEFLAGS="--jobs=5 --output-sync=target" \
-		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
-		--build-arg IMAGE_TYPE=$(IMAGE_TYPE) \
-		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
-		--build-arg GO_TAGS="$(GO_TAGS)" \
-		-t local-ai:tests -f Dockerfile .
+	docker build --build-arg MAKEFLAGS="--jobs=5 --output-sync=target" -t local-ai:tests -f Dockerfile .
 	BASE_IMAGE=local-ai:tests DOCKER_AIO_IMAGE=local-ai-aio:test $(MAKE) docker-aio

 e2e-aio:
@@ -194,17 +177,7 @@ prepare-e2e:
 	mkdir -p $(TEST_DIR)
 	cp -rfv $(abspath ./tests/e2e-fixtures)/gpu.yaml $(TEST_DIR)/gpu.yaml
 	test -e $(TEST_DIR)/ggllm-test-model.bin || wget -q https://huggingface.co/TheBloke/CodeLlama-7B-Instruct-GGUF/resolve/main/codellama-7b-instruct.Q2_K.gguf -O $(TEST_DIR)/ggllm-test-model.bin
-	docker build \
-		--build-arg IMAGE_TYPE=core \
-		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
-		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
-		--build-arg GO_TAGS="$(GO_TAGS)" \
-		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
-		-t localai-tests .
+	docker build --build-arg IMAGE_TYPE=core --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg CUDA_MAJOR_VERSION=12 --build-arg CUDA_MINOR_VERSION=0 -t localai-tests .

 run-e2e-image:
 	ls -liah $(abspath ./tests/e2e-fixtures)
@@ -315,7 +288,6 @@ prepare-test-extra: protogen-python
 	$(MAKE) -C backend/python/chatterbox
 	$(MAKE) -C backend/python/vllm
 	$(MAKE) -C backend/python/vibevoice
-	$(MAKE) -C backend/python/moonshine

 test-extra: prepare-test-extra
 	$(MAKE) -C backend/python/transformers test
@@ -323,12 +295,11 @@ test-extra: prepare-test-extra
 	$(MAKE) -C backend/python/chatterbox test
 	$(MAKE) -C backend/python/vllm test
 	$(MAKE) -C backend/python/vibevoice test
-	$(MAKE) -C backend/python/moonshine test

 DOCKER_IMAGE?=local-ai
 DOCKER_AIO_IMAGE?=local-ai-aio
 IMAGE_TYPE?=core
-BASE_IMAGE?=ubuntu:24.04
+BASE_IMAGE?=ubuntu:22.04

 docker:
 	docker build \
@@ -337,34 +308,24 @@ docker:
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
 		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		-t $(DOCKER_IMAGE) .

-docker-cuda12:
+docker-cuda11:
 	docker build \
-		--build-arg CUDA_MAJOR_VERSION=${CUDA_MAJOR_VERSION} \
-		--build-arg CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION} \
+		--build-arg CUDA_MAJOR_VERSION=11 \
+		--build-arg CUDA_MINOR_VERSION=8 \
 		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
 		--build-arg IMAGE_TYPE=$(IMAGE_TYPE) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
 		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
-		-t $(DOCKER_IMAGE)-cuda-12 .
+		-t $(DOCKER_IMAGE)-cuda-11 .

 docker-aio:
 	@echo "Building AIO image with base $(BASE_IMAGE) as $(DOCKER_AIO_IMAGE)"
 	docker build \
 		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
 		-t $(DOCKER_AIO_IMAGE) -f Dockerfile.aio .

 docker-aio-all:
@@ -373,31 +334,66 @@ docker-aio-all:

 docker-image-intel:
 	docker build \
-		--build-arg BASE_IMAGE=intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04 \
+		--build-arg BASE_IMAGE=quay.io/go-skynet/intel-oneapi-base:latest \
 		--build-arg IMAGE_TYPE=$(IMAGE_TYPE) \
 		--build-arg GO_TAGS="$(GO_TAGS)" \
 		--build-arg MAKEFLAGS="$(DOCKER_MAKEFLAGS)" \
-		--build-arg BUILD_TYPE=intel \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
-		-t $(DOCKER_IMAGE) .
+		--build-arg BUILD_TYPE=intel -t $(DOCKER_IMAGE) .

 ########################################################
 ## Backends
 ########################################################

-# Pattern rule for standard backends (docker-based)
-# This matches all backends that use docker-build-* and docker-save-*
-backends/%: docker-build-% docker-save-% build
-	./local-ai backends install "ocifile://$(abspath ./backend-images/$*.tar)"

-# Darwin-specific backends (keep as explicit targets since they have special build logic)
+backends/diffusers: docker-build-diffusers docker-save-diffusers build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/diffusers.tar)"
+
+backends/llama-cpp: docker-build-llama-cpp docker-save-llama-cpp build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/llama-cpp.tar)"
+
+backends/piper: docker-build-piper docker-save-piper build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/piper.tar)"
+
+backends/stablediffusion-ggml: docker-build-stablediffusion-ggml docker-save-stablediffusion-ggml build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/stablediffusion-ggml.tar)"
+
+backends/whisper: docker-build-whisper docker-save-whisper build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/whisper.tar)"
+
+backends/silero-vad: docker-build-silero-vad docker-save-silero-vad build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/silero-vad.tar)"
+
+backends/local-store: docker-build-local-store docker-save-local-store build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/local-store.tar)"
+
+backends/huggingface: docker-build-huggingface docker-save-huggingface build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/huggingface.tar)"
+
+backends/rfdetr: docker-build-rfdetr docker-save-rfdetr build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/rfdetr.tar)"
+
+backends/kitten-tts: docker-build-kitten-tts docker-save-kitten-tts build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/kitten-tts.tar)"
+
+backends/kokoro: docker-build-kokoro docker-save-kokoro build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/kokoro.tar)"
+
+backends/chatterbox: docker-build-chatterbox docker-save-chatterbox build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/chatterbox.tar)"
+
 backends/llama-cpp-darwin: build
 	bash ./scripts/build/llama-cpp-darwin.sh
 	./local-ai backends install "ocifile://$(abspath ./backend-images/llama-cpp.tar)"

+backends/neutts: docker-build-neutts docker-save-neutts build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/neutts.tar)"
+
+backends/vllm: docker-build-vllm docker-save-vllm build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/vllm.tar)"
+
+backends/vibevoice: docker-build-vibevoice docker-save-vibevoice build
+	./local-ai backends install "ocifile://$(abspath ./backend-images/vibevoice.tar)"
+
 build-darwin-python-backend: build
 	bash ./scripts/build/python-darwin.sh

@@ -427,88 +423,121 @@ backends/stablediffusion-ggml-darwin:
 backend-images:
 	mkdir -p backend-images

-# Backend metadata: BACKEND_NAME | DOCKERFILE_TYPE | BUILD_CONTEXT | PROGRESS_FLAG | NEEDS_BACKEND_ARG
-# llama-cpp is special - uses llama-cpp Dockerfile and doesn't need BACKEND arg
-BACKEND_LLAMA_CPP = llama-cpp|llama-cpp|.|false|false
+docker-build-llama-cpp:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:llama-cpp -f backend/Dockerfile.llama-cpp .

-# Golang backends
-BACKEND_BARK_CPP = bark-cpp|golang|.|false|true
-BACKEND_PIPER = piper|golang|.|false|true
-BACKEND_LOCAL_STORE = local-store|golang|.|false|true
-BACKEND_HUGGINGFACE = huggingface|golang|.|false|true
-BACKEND_SILERO_VAD = silero-vad|golang|.|false|true
-BACKEND_STABLEDIFFUSION_GGML = stablediffusion-ggml|golang|.|--progress=plain|true
-BACKEND_WHISPER = whisper|golang|.|false|true
+docker-build-bark-cpp:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:bark-cpp -f backend/Dockerfile.golang --build-arg BACKEND=bark-cpp .

-# Python backends with root context
-BACKEND_RERANKERS = rerankers|python|.|false|true
-BACKEND_TRANSFORMERS = transformers|python|.|false|true
-BACKEND_FASTER_WHISPER = faster-whisper|python|.|false|true
-BACKEND_COQUI = coqui|python|.|false|true
-BACKEND_BARK = bark|python|.|false|true
-BACKEND_EXLLAMA2 = exllama2|python|.|false|true
+docker-build-piper:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:piper -f backend/Dockerfile.golang --build-arg BACKEND=piper .

-# Python backends with ./backend context
-BACKEND_RFDETR = rfdetr|python|./backend|false|true
-BACKEND_KITTEN_TTS = kitten-tts|python|./backend|false|true
-BACKEND_NEUTTS = neutts|python|./backend|false|true
-BACKEND_KOKORO = kokoro|python|./backend|false|true
-BACKEND_VLLM = vllm|python|./backend|false|true
-BACKEND_DIFFUSERS = diffusers|python|./backend|--progress=plain|true
-BACKEND_CHATTERBOX = chatterbox|python|./backend|false|true
-BACKEND_VIBEVOICE = vibevoice|python|./backend|--progress=plain|true
-BACKEND_MOONSHINE = moonshine|python|./backend|false|true
+docker-build-local-store:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:local-store -f backend/Dockerfile.golang --build-arg BACKEND=local-store .

-# Helper function to build docker image for a backend
-# Usage: $(call docker-build-backend,BACKEND_NAME,DOCKERFILE_TYPE,BUILD_CONTEXT,PROGRESS_FLAG,NEEDS_BACKEND_ARG)
-define docker-build-backend
-	docker build $(if $(filter-out false,$(4)),$(4)) \
-		--build-arg BUILD_TYPE=$(BUILD_TYPE) \
-		--build-arg BASE_IMAGE=$(BASE_IMAGE) \
-		--build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) \
-		--build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) \
-		--build-arg UBUNTU_VERSION=$(UBUNTU_VERSION) \
-		--build-arg UBUNTU_CODENAME=$(UBUNTU_CODENAME) \
-		$(if $(filter true,$(5)),--build-arg BACKEND=$(1)) \
-		-t local-ai-backend:$(1) -f backend/Dockerfile.$(2) $(3)
-endef
+docker-build-huggingface:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:huggingface -f backend/Dockerfile.golang --build-arg BACKEND=huggingface .

-# Generate docker-build targets from backend definitions
-define generate-docker-build-target
-docker-build-$(word 1,$(subst |, ,$(1))):
-	$$(call docker-build-backend,$(word 1,$(subst |, ,$(1))),$(word 2,$(subst |, ,$(1))),$(word 3,$(subst |, ,$(1))),$(word 4,$(subst |, ,$(1))),$(word 5,$(subst |, ,$(1))))
-endef
+docker-build-rfdetr:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:rfdetr -f backend/Dockerfile.python --build-arg BACKEND=rfdetr ./backend

-# Generate all docker-build targets
-$(eval $(call generate-docker-build-target,$(BACKEND_LLAMA_CPP)))
-$(eval $(call generate-docker-build-target,$(BACKEND_BARK_CPP)))
-$(eval $(call generate-docker-build-target,$(BACKEND_PIPER)))
-$(eval $(call generate-docker-build-target,$(BACKEND_LOCAL_STORE)))
-$(eval $(call generate-docker-build-target,$(BACKEND_HUGGINGFACE)))
-$(eval $(call generate-docker-build-target,$(BACKEND_SILERO_VAD)))
-$(eval $(call generate-docker-build-target,$(BACKEND_STABLEDIFFUSION_GGML)))
-$(eval $(call generate-docker-build-target,$(BACKEND_WHISPER)))
-$(eval $(call generate-docker-build-target,$(BACKEND_RERANKERS)))
-$(eval $(call generate-docker-build-target,$(BACKEND_TRANSFORMERS)))
-$(eval $(call generate-docker-build-target,$(BACKEND_FASTER_WHISPER)))
-$(eval $(call generate-docker-build-target,$(BACKEND_COQUI)))
-$(eval $(call generate-docker-build-target,$(BACKEND_BARK)))
-$(eval $(call generate-docker-build-target,$(BACKEND_EXLLAMA2)))
-$(eval $(call generate-docker-build-target,$(BACKEND_RFDETR)))
-$(eval $(call generate-docker-build-target,$(BACKEND_KITTEN_TTS)))
-$(eval $(call generate-docker-build-target,$(BACKEND_NEUTTS)))
-$(eval $(call generate-docker-build-target,$(BACKEND_KOKORO)))
-$(eval $(call generate-docker-build-target,$(BACKEND_VLLM)))
-$(eval $(call generate-docker-build-target,$(BACKEND_DIFFUSERS)))
-$(eval $(call generate-docker-build-target,$(BACKEND_CHATTERBOX)))
-$(eval $(call generate-docker-build-target,$(BACKEND_VIBEVOICE)))
-$(eval $(call generate-docker-build-target,$(BACKEND_MOONSHINE)))
+docker-build-kitten-tts:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:kitten-tts -f backend/Dockerfile.python --build-arg BACKEND=kitten-tts ./backend

-# Pattern rule for docker-save targets
-docker-save-%: backend-images
-	docker save local-ai-backend:$* -o backend-images/$*.tar
+docker-save-kitten-tts: backend-images
+	docker save local-ai-backend:kitten-tts -o backend-images/kitten-tts.tar

-docker-build-backends: docker-build-llama-cpp docker-build-rerankers docker-build-vllm docker-build-transformers docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-coqui docker-build-bark docker-build-chatterbox docker-build-vibevoice docker-build-exllama2 docker-build-moonshine
+docker-save-chatterbox: backend-images
+	docker save local-ai-backend:chatterbox -o backend-images/chatterbox.tar
+
+docker-save-vibevoice: backend-images
+	docker save local-ai-backend:vibevoice -o backend-images/vibevoice.tar
+
+docker-build-neutts:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:neutts -f backend/Dockerfile.python --build-arg BACKEND=neutts ./backend
+
+docker-save-neutts: backend-images
+	docker save local-ai-backend:neutts -o backend-images/neutts.tar
+
+docker-build-kokoro:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:kokoro -f backend/Dockerfile.python --build-arg BACKEND=kokoro ./backend
+
+docker-build-vllm:
+	docker build --build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) --build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:vllm -f backend/Dockerfile.python --build-arg BACKEND=vllm ./backend
+
+docker-save-vllm: backend-images
+	docker save local-ai-backend:vllm -o backend-images/vllm.tar
+
+docker-save-kokoro: backend-images
+	docker save local-ai-backend:kokoro -o backend-images/kokoro.tar
+
+docker-save-rfdetr: backend-images
+	docker save local-ai-backend:rfdetr -o backend-images/rfdetr.tar
+
+docker-save-huggingface: backend-images
+	docker save local-ai-backend:huggingface -o backend-images/huggingface.tar
+
+docker-save-local-store: backend-images
+	docker save local-ai-backend:local-store -o backend-images/local-store.tar
+
+docker-build-silero-vad:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:silero-vad -f backend/Dockerfile.golang --build-arg BACKEND=silero-vad .
+
+docker-save-silero-vad: backend-images
+	docker save local-ai-backend:silero-vad -o backend-images/silero-vad.tar
+
+docker-save-piper: backend-images
+	docker save local-ai-backend:piper -o backend-images/piper.tar
+
+docker-save-llama-cpp: backend-images
+	docker save local-ai-backend:llama-cpp -o backend-images/llama-cpp.tar
+
+docker-save-bark-cpp: backend-images
+	docker save local-ai-backend:bark-cpp -o backend-images/bark-cpp.tar
+
+docker-build-stablediffusion-ggml:
+	docker build --progress=plain --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) --build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) --build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) -t local-ai-backend:stablediffusion-ggml -f backend/Dockerfile.golang --build-arg BACKEND=stablediffusion-ggml .
+
+docker-save-stablediffusion-ggml: backend-images
+	docker save local-ai-backend:stablediffusion-ggml -o backend-images/stablediffusion-ggml.tar
+
+docker-build-rerankers:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:rerankers -f backend/Dockerfile.python --build-arg BACKEND=rerankers .
+
+docker-build-transformers:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:transformers -f backend/Dockerfile.python --build-arg BACKEND=transformers .
+
+docker-build-diffusers:
+	docker build --progress=plain --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:diffusers -f backend/Dockerfile.python --build-arg BACKEND=diffusers ./backend
+
+docker-save-diffusers: backend-images
+	docker save local-ai-backend:diffusers -o backend-images/diffusers.tar
+
+docker-build-whisper:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) --build-arg CUDA_MAJOR_VERSION=$(CUDA_MAJOR_VERSION) --build-arg CUDA_MINOR_VERSION=$(CUDA_MINOR_VERSION) -t local-ai-backend:whisper -f backend/Dockerfile.golang --build-arg BACKEND=whisper  .
+
+docker-save-whisper: backend-images
+	docker save local-ai-backend:whisper -o backend-images/whisper.tar
+
+docker-build-faster-whisper:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:faster-whisper -f backend/Dockerfile.python --build-arg BACKEND=faster-whisper .
+
+docker-build-coqui:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:coqui -f backend/Dockerfile.python --build-arg BACKEND=coqui .
+
+docker-build-bark:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:bark -f backend/Dockerfile.python --build-arg BACKEND=bark .
+
+docker-build-chatterbox:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:chatterbox -f backend/Dockerfile.python --build-arg BACKEND=chatterbox ./backend
+
+docker-build-vibevoice:
+	docker build --progress=plain --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:vibevoice -f backend/Dockerfile.python --build-arg BACKEND=vibevoice ./backend
+
+docker-build-exllama2:
+	docker build --build-arg BUILD_TYPE=$(BUILD_TYPE) --build-arg BASE_IMAGE=$(BASE_IMAGE) -t local-ai-backend:exllama2 -f backend/Dockerfile.python --build-arg BACKEND=exllama2 .
+
+docker-build-backends: docker-build-llama-cpp docker-build-rerankers docker-build-vllm docker-build-transformers docker-build-diffusers docker-build-kokoro docker-build-faster-whisper docker-build-coqui docker-build-bark docker-build-chatterbox docker-build-vibevoice docker-build-exllama2

 ########################################################
 ### END Backends
--- a/README.md
+++ b/README.md
@@ -152,6 +152,9 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gp
 # CUDA 12.0
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-12

+# CUDA 11.7
+docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-11
+
 # NVIDIA Jetson (L4T) ARM64
 # CUDA 12 (for Nvidia AGX Orin and similar platforms)
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-nvidia-l4t-arm64
@@ -190,6 +193,9 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-ai
 # NVIDIA CUDA 12 version
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-12

+# NVIDIA CUDA 11 version
+docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-11
+
 # Intel GPU version
 docker run -ti --name local-ai -p 8080:8080 localai/localai:latest-aio-gpu-intel

@@ -273,9 +279,9 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration
 ### Text Generation & Language Models
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
-| **llama.cpp** | LLM inference in C/C++ | CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
+| **llama.cpp** | LLM inference in C/C++ | CUDA 11/12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
 | **vLLM** | Fast LLM inference with PagedAttention | CUDA 12/13, ROCm, Intel |
-| **transformers** | HuggingFace transformers framework | CUDA 12/13, ROCm, Intel, CPU |
+| **transformers** | HuggingFace transformers framework | CUDA 11/12/13, ROCm, Intel, CPU |
 | **exllama2** | GPTQ inference library | CUDA 12/13 |
 | **MLX** | Apple Silicon LLM inference | Metal (M1/M2/M3+) |
 | **MLX-VLM** | Apple Silicon Vision-Language Models | Metal (M1/M2/M3+) |
@@ -289,7 +295,7 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration
 | **bark-cpp** | C++ implementation of Bark | CUDA, Metal, CPU |
 | **coqui** | Advanced TTS with 1100+ languages | CUDA 12/13, ROCm, Intel, CPU |
 | **kokoro** | Lightweight TTS model | CUDA 12/13, ROCm, Intel, CPU |
-| **chatterbox** | Production-grade TTS | CUDA 12/13, CPU |
+| **chatterbox** | Production-grade TTS | CUDA 11/12/13, CPU |
 | **piper** | Fast neural TTS system | CPU |
 | **kitten-tts** | Kitten TTS models | CPU |
 | **silero-vad** | Voice Activity Detection | CPU |
@@ -300,13 +306,13 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
 | **stablediffusion.cpp** | Stable Diffusion in C/C++ | CUDA 12/13, Intel SYCL, Vulkan, CPU |
-| **diffusers** | HuggingFace diffusion models | CUDA 12/13, ROCm, Intel, Metal, CPU |
+| **diffusers** | HuggingFace diffusion models | CUDA 11/12/13, ROCm, Intel, Metal, CPU |

 ### Specialized AI Tasks
 | Backend | Description | Acceleration Support |
 |---------|-------------|---------------------|
 | **rfdetr** | Real-time object detection | CUDA 12/13, Intel, CPU |
-| **rerankers** | Document reranking API | CUDA 12/13, ROCm, Intel, CPU |
+| **rerankers** | Document reranking API | CUDA 11/12/13, ROCm, Intel, CPU |
 | **local-store** | Vector database | CPU |
 | **huggingface** | HuggingFace API integration | API-based |

@@ -314,6 +320,7 @@ LocalAI supports a comprehensive range of AI backends with multiple acceleration

 | Acceleration Type | Supported Backends | Hardware Support |
 |-------------------|-------------------|------------------|
+| **NVIDIA CUDA 11** | llama.cpp, whisper, stablediffusion, diffusers, rerankers, bark, chatterbox | Nvidia hardware |
 | **NVIDIA CUDA 12** | All CUDA-compatible backends | Nvidia hardware |
 | **NVIDIA CUDA 13** | All CUDA-compatible backends | Nvidia hardware |
 | **AMD ROCm** | llama.cpp, whisper, vllm, transformers, diffusers, rerankers, coqui, kokoro, bark, neutts, vibevoice | AMD Graphics |
--- a/backend/Dockerfile.golang
+++ b/backend/Dockerfile.golang
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:24.04
+ARG BASE_IMAGE=ubuntu:22.04

 FROM ${BASE_IMAGE} AS builder
 ARG BACKEND=rerankers
@@ -12,8 +12,8 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG GO_VERSION=1.25.4
-ARG UBUNTU_VERSION=2404
+ARG GO_VERSION=1.22.6
+ARG UBUNTU_VERSION=2204

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -40,45 +40,11 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
-            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
-            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
-            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
-            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
-            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
-        if [ "amd64" = "$TARGETARCH" ]; then
-            wget "https://sdk.lunarg.com/sdk/download/1.4.328.1/linux/vulkansdk-linux-x86_64-1.4.328.1.tar.xz" && \
-            tar -xf vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            rm vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            mkdir -p /opt/vulkan-sdk && \
-            mv 1.4.328.1 /opt/vulkan-sdk/ && \
-            cd /opt/vulkan-sdk/1.4.328.1 && \
-            ./vulkansdk --no-deps --maxjobs \
-                vulkan-loader \
-                vulkan-validationlayers \
-                vulkan-extensionlayer \
-                vulkan-tools \
-                shaderc && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/bin/* /usr/bin/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/include/* /usr/include/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/share/* /usr/share/ && \
-            rm -rf /opt/vulkan-sdk
-        fi
-        if [ "arm64" = "$TARGETARCH" ]; then
-            mkdir vulkan && cd vulkan && \
-            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
-            tar -xvf vulkan-sdk.tar.xz && \
-            rm vulkan-sdk.tar.xz && \
-            cd 1.4.335.0 && \
-            cp -rfv aarch64/bin/* /usr/bin/ && \
-            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
-            cp -rfv aarch64/include/* /usr/include/ && \
-            cp -rfv aarch64/share/* /usr/share/ && \
-            cd ../.. && \
-            rm -rf vulkan
-        fi
-        ldconfig && \
+        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
+        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
+        apt-get update && \
+        apt-get install -y \
+            vulkan-sdk && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -182,8 +148,6 @@ EOT

 COPY . /LocalAI

-RUN git config --global --add safe.directory /LocalAI
-
 RUN cd /LocalAI && make protogen-go && make -C /LocalAI/backend/go/${BACKEND} build

 FROM scratch
--- a/backend/Dockerfile.llama-cpp
+++ b/backend/Dockerfile.llama-cpp
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:24.04
+ARG BASE_IMAGE=ubuntu:22.04
 ARG GRPC_BASE_IMAGE=${BASE_IMAGE}


@@ -10,8 +10,7 @@ FROM ${GRPC_BASE_IMAGE} AS grpc
 ARG GRPC_MAKEFLAGS="-j4 -Otarget"
 ARG GRPC_VERSION=v1.65.0
 ARG CMAKE_FROM_SOURCE=false
-# CUDA Toolkit 13.x compatibility: CMake 3.31.9+ fixes toolchain detection/arch table issues
-ARG CMAKE_VERSION=3.31.10
+ARG CMAKE_VERSION=3.26.4

 ENV MAKEFLAGS=${GRPC_MAKEFLAGS}

@@ -27,7 +26,7 @@ RUN apt-get update && \

 # Install CMake (the version in 22.04 is too old)
 RUN <<EOT bash
-    if [ "${CMAKE_FROM_SOURCE}" = "true" ]; then
+    if [ "${CMAKE_FROM_SOURCE}}" = "true" ]; then
        curl -L -s https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}.tar.gz -o cmake.tar.gz && tar xvf cmake.tar.gz && cd cmake-${CMAKE_VERSION} && ./configure && make && make install
    else
        apt-get update && \
@@ -51,13 +50,6 @@ RUN git clone --recurse-submodules --jobs 4 -b ${GRPC_VERSION} --depth 1 --shall
    rm -rf /build

 FROM ${BASE_IMAGE} AS builder
-ARG CMAKE_FROM_SOURCE=false
-ARG CMAKE_VERSION=3.31.10
-# We can target specific CUDA ARCHITECTURES like --build-arg CUDA_DOCKER_ARCH='75;86;89;120'
-ARG CUDA_DOCKER_ARCH
-ENV CUDA_DOCKER_ARCH=${CUDA_DOCKER_ARCH}
-ARG CMAKE_ARGS
-ENV CMAKE_ARGS=${CMAKE_ARGS}
 ARG BACKEND=rerankers
 ARG BUILD_TYPE
 ENV BUILD_TYPE=${BUILD_TYPE}
@@ -69,8 +61,8 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG GO_VERSION=1.25.4
-ARG UBUNTU_VERSION=2404
+ARG GO_VERSION=1.22.6
+ARG UBUNTU_VERSION=2204

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -78,7 +70,6 @@ RUN apt-get update && \
        ccache git \
        ca-certificates \
        make \
-        pkg-config libcurl4-openssl-dev \
        curl unzip \
        libssl-dev wget && \
    apt-get clean && \
@@ -97,45 +88,11 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
-            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
-            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
-            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
-            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
-            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
-        if [ "amd64" = "$TARGETARCH" ]; then
-            wget "https://sdk.lunarg.com/sdk/download/1.4.328.1/linux/vulkansdk-linux-x86_64-1.4.328.1.tar.xz" && \
-            tar -xf vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            rm vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            mkdir -p /opt/vulkan-sdk && \
-            mv 1.4.328.1 /opt/vulkan-sdk/ && \
-            cd /opt/vulkan-sdk/1.4.328.1 && \
-            ./vulkansdk --no-deps --maxjobs \
-                vulkan-loader \
-                vulkan-validationlayers \
-                vulkan-extensionlayer \
-                vulkan-tools \
-                shaderc && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/bin/* /usr/bin/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/include/* /usr/include/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/share/* /usr/share/ && \
-            rm -rf /opt/vulkan-sdk
-        fi
-        if [ "arm64" = "$TARGETARCH" ]; then
-            mkdir vulkan && cd vulkan && \
-            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
-            tar -xvf vulkan-sdk.tar.xz && \
-            rm vulkan-sdk.tar.xz && \
-            cd 1.4.335.0 && \
-            cp -rfv aarch64/bin/* /usr/bin/ && \
-            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
-            cp -rfv aarch64/include/* /usr/include/ && \
-            cp -rfv aarch64/share/* /usr/share/ && \
-            cd ../.. && \
-            rm -rf vulkan
-        fi
-        ldconfig && \
+        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
+        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
+        apt-get update && \
+        apt-get install -y \
+            vulkan-sdk && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -232,7 +189,7 @@ EOT

 # Install CMake (the version in 22.04 is too old)
 RUN <<EOT bash
-    if [ "${CMAKE_FROM_SOURCE}" = "true" ]; then
+    if [ "${CMAKE_FROM_SOURCE}}" = "true" ]; then
        curl -L -s https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}.tar.gz -o cmake.tar.gz && tar xvf cmake.tar.gz && cd cmake-${CMAKE_VERSION} && ./configure && make && make install
    else
        apt-get update && \
@@ -248,30 +205,19 @@ COPY --from=grpc /opt/grpc /usr/local

 COPY . /LocalAI

-RUN <<'EOT' bash
-set -euxo pipefail
-
-if [[ -n "${CUDA_DOCKER_ARCH:-}" ]]; then
-  CUDA_ARCH_ESC="${CUDA_DOCKER_ARCH//;/\\;}"
-  export CMAKE_ARGS="${CMAKE_ARGS:-} -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCH_ESC}"
-  echo "CMAKE_ARGS(env) = ${CMAKE_ARGS}"
-  rm -rf /LocalAI/backend/cpp/llama-cpp-*-build
-fi
-
-if [ "${TARGETARCH}" = "arm64" ] || [ "${BUILD_TYPE}" = "hipblas" ]; then
-  cd /LocalAI/backend/cpp/llama-cpp
-  make llama-cpp-fallback
-  make llama-cpp-grpc
-  make llama-cpp-rpc-server
-else
-  cd /LocalAI/backend/cpp/llama-cpp
-  make llama-cpp-avx
-  make llama-cpp-avx2
-  make llama-cpp-avx512
-  make llama-cpp-fallback
-  make llama-cpp-grpc
-  make llama-cpp-rpc-server
-fi
+## Otherwise just run the normal build
+RUN <<EOT bash
+if [ "${TARGETARCH}" = "arm64" ] || [ "${BUILD_TYPE}" = "hipblas" ]; then \
+        cd /LocalAI/backend/cpp/llama-cpp && make llama-cpp-fallback && \
+        make llama-cpp-grpc && make llama-cpp-rpc-server; \
+    else \
+        cd /LocalAI/backend/cpp/llama-cpp && make llama-cpp-avx && \
+        make llama-cpp-avx2 && \
+        make llama-cpp-avx512 && \
+        make llama-cpp-fallback && \
+        make llama-cpp-grpc && \
+        make llama-cpp-rpc-server; \
+    fi  
 EOT


--- a/backend/Dockerfile.python
+++ b/backend/Dockerfile.python
@@ -1,4 +1,4 @@
-ARG BASE_IMAGE=ubuntu:24.04
+ARG BASE_IMAGE=ubuntu:22.04

 FROM ${BASE_IMAGE} AS builder
 ARG BACKEND=rerankers
@@ -12,7 +12,7 @@ ENV CUDA_MINOR_VERSION=${CUDA_MINOR_VERSION}
 ENV DEBIAN_FRONTEND=noninteractive
 ARG TARGETARCH
 ARG TARGETVARIANT
-ARG UBUNTU_VERSION=2404
+ARG UBUNTU_VERSION=2204

 RUN apt-get update && \
    apt-get install -y --no-install-recommends \
@@ -54,45 +54,11 @@ RUN <<EOT bash
        apt-get update && \
        apt-get install -y  --no-install-recommends \
            software-properties-common pciutils wget gpg-agent && \
-        apt-get install -y libglm-dev cmake libxcb-dri3-0 libxcb-present0 libpciaccess0 \
-            libpng-dev libxcb-keysyms1-dev libxcb-dri3-dev libx11-dev g++ gcc \
-            libwayland-dev libxrandr-dev libxcb-randr0-dev libxcb-ewmh-dev \
-            git python-is-python3 bison libx11-xcb-dev liblz4-dev libzstd-dev \
-            ocaml-core ninja-build pkg-config libxml2-dev wayland-protocols python3-jsonschema \
-            clang-format qtbase5-dev qt6-base-dev libxcb-glx0-dev sudo xz-utils
-        if [ "amd64" = "$TARGETARCH" ]; then
-            wget "https://sdk.lunarg.com/sdk/download/1.4.328.1/linux/vulkansdk-linux-x86_64-1.4.328.1.tar.xz" && \
-            tar -xf vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            rm vulkansdk-linux-x86_64-1.4.328.1.tar.xz && \
-            mkdir -p /opt/vulkan-sdk && \
-            mv 1.4.328.1 /opt/vulkan-sdk/ && \
-            cd /opt/vulkan-sdk/1.4.328.1 && \
-            ./vulkansdk --no-deps --maxjobs \
-                vulkan-loader \
-                vulkan-validationlayers \
-                vulkan-extensionlayer \
-                vulkan-tools \
-                shaderc && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/bin/* /usr/bin/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/lib/* /usr/lib/x86_64-linux-gnu/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/include/* /usr/include/ && \
-            cp -rfv /opt/vulkan-sdk/1.4.328.1/x86_64/share/* /usr/share/ && \
-            rm -rf /opt/vulkan-sdk
-        fi
-        if [ "arm64" = "$TARGETARCH" ]; then
-            mkdir vulkan && cd vulkan && \
-            curl -L -o vulkan-sdk.tar.xz https://github.com/mudler/vulkan-sdk-arm/releases/download/1.4.335.0/vulkansdk-ubuntu-24.04-arm-1.4.335.0.tar.xz && \
-            tar -xvf vulkan-sdk.tar.xz && \
-            rm vulkan-sdk.tar.xz && \
-            cd 1.4.335.0 && \
-            cp -rfv aarch64/bin/* /usr/bin/ && \
-            cp -rfv aarch64/lib/* /usr/lib/aarch64-linux-gnu/ && \
-            cp -rfv aarch64/include/* /usr/include/ && \
-            cp -rfv aarch64/share/* /usr/share/ && \
-            cd ../.. && \
-            rm -rf vulkan
-        fi
-        ldconfig && \
+        wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | apt-key add - && \
+        wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list && \
+        apt-get update && \
+        apt-get install -y \
+            vulkan-sdk && \
        apt-get clean && \
        rm -rf /var/lib/apt/lists/*
    fi
@@ -176,8 +142,7 @@ RUN if [ "${BUILD_TYPE}" = "hipblas" ]; then \
 # Install uv as a system package
 RUN curl -LsSf https://astral.sh/uv/install.sh | UV_INSTALL_DIR=/usr/bin sh
 ENV PATH="/root/.cargo/bin:${PATH}"
-# Increase timeout for uv installs behind slow networks
-ENV UV_HTTP_TIMEOUT=180
+
 RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y

 # Install grpcio-tools (the version in 22.04 is too old)
@@ -190,18 +155,12 @@ RUN <<EOT bash
 EOT


-COPY backend/python/${BACKEND} /${BACKEND}
-COPY backend/backend.proto /${BACKEND}/backend.proto
-COPY backend/python/common/ /${BACKEND}/common
-COPY scripts/build/package-gpu-libs.sh /package-gpu-libs.sh
+COPY python/${BACKEND} /${BACKEND}
+COPY backend.proto /${BACKEND}/backend.proto
+COPY python/common/ /${BACKEND}/common

 RUN cd /${BACKEND} && PORTABLE_PYTHON=true make

-# Package GPU libraries into the backend's lib directory
-RUN mkdir -p /${BACKEND}/lib && \
-    TARGET_LIB_DIR="/${BACKEND}/lib" BUILD_TYPE="${BUILD_TYPE}" CUDA_MAJOR_VERSION="${CUDA_MAJOR_VERSION}" \
-    bash /package-gpu-libs.sh "/${BACKEND}/lib"
-
 FROM scratch
 ARG BACKEND=rerankers
 COPY --from=builder /${BACKEND}/ /
--- a/backend/README.md
+++ b/backend/README.md
@@ -65,7 +65,7 @@ The backend system provides language-specific Dockerfiles that handle the build
 ## Hardware Acceleration Support

 ### CUDA (NVIDIA)
- **Versions**: CUDA 12.x, 13.x
+- **Versions**: CUDA 11.x, 12.x
 - **Features**: cuBLAS, cuDNN, TensorRT optimization
 - **Targets**: x86_64, ARM64 (Jetson)

@@ -132,7 +132,8 @@ For ARM64/Mac builds, docker can't be used, and the makefile in the respective b
 ### Build Types

 - **`cpu`**: CPU-only optimization
- **`cublas12`**, **`cublas13`**: CUDA 12.x, 13.x with cuBLAS
+- **`cublas11`**: CUDA 11.x with cuBLAS
+- **`cublas12`**: CUDA 12.x with cuBLAS
 - **`hipblas`**: ROCm with rocBLAS
 - **`intel`**: Intel oneAPI optimization
 - **`vulkan`**: Vulkan-based acceleration
@@ -209,4 +210,4 @@ When contributing to the backend system:
 2. **Add Tests**: Include comprehensive test coverage
 3. **Document**: Provide clear usage examples
 4. **Optimize**: Consider performance and resource usage
-5. **Validate**: Test across different hardware targets
+5. **Validate**: Test across different hardware targets
--- a/backend/cpp/llama-cpp/CMakeLists.txt
+++ b/backend/cpp/llama-cpp/CMakeLists.txt
@@ -70,4 +70,4 @@ target_link_libraries(${TARGET} PRIVATE common llama mtmd ${CMAKE_THREAD_LIBS_IN
 target_compile_features(${TARGET} PRIVATE cxx_std_11)
 if(TARGET BUILD_INFO)
  add_dependencies(${TARGET} BUILD_INFO)
-endif()
+endif()
--- a/backend/cpp/llama-cpp/Makefile
+++ b/backend/cpp/llama-cpp/Makefile
@@ -1,5 +1,5 @@

-LLAMA_VERSION?=593da7fa49503b68f9f01700be9f508f1e528992
+LLAMA_VERSION?=ced765be44ce173c374f295b3c6f4175f8fd109b
 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp

 CMAKE_ARGS?=
@@ -7,8 +7,7 @@ BUILD_TYPE?=
 NATIVE?=false
 ONEAPI_VARS?=/opt/intel/oneapi/setvars.sh
 TARGET?=--target grpc-server
-JOBS?=$(shell nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 1)
-ARCH?=$(shell uname -m)
+JOBS?=$(shell nproc)

 # Disable Shared libs as we are linking on static gRPC and we can't mix shared and static
 CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF -DLLAMA_CURL=OFF
@@ -107,21 +106,21 @@ llama-cpp-avx: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build purge
 	$(info ${GREEN}I llama-cpp build info:avx${RESET})
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=on -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-avx-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-avx-build/grpc-server llama-cpp-avx

 llama-cpp-fallback: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build purge
 	$(info ${GREEN}I llama-cpp build info:fallback${RESET})
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" $(MAKE) VARIANT="llama-cpp-fallback-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-fallback-build/grpc-server llama-cpp-fallback

 llama-cpp-grpc: llama.cpp
 	cp -rf $(CURRENT_MAKEFILE_DIR)/../llama-cpp $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build
 	$(MAKE) -C $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build purge
 	$(info ${GREEN}I llama-cpp build info:grpc${RESET})
-	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off -DGGML_BMI2=off" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
+	CMAKE_ARGS="$(CMAKE_ARGS) -DGGML_RPC=ON -DGGML_AVX=off -DGGML_AVX2=off -DGGML_AVX512=off -DGGML_FMA=off -DGGML_F16C=off" TARGET="--target grpc-server --target rpc-server" $(MAKE) VARIANT="llama-cpp-grpc-build" build-llama-cpp-grpc-server
 	cp -rfv $(CURRENT_MAKEFILE_DIR)/../llama-cpp-grpc-build/grpc-server llama-cpp-grpc

 llama-cpp-rpc-server: llama-cpp-grpc
--- a/backend/cpp/llama-cpp/grpc-server.cpp
+++ b/backend/cpp/llama-cpp/grpc-server.cpp
@@ -23,7 +23,6 @@
 #include <grpcpp/health_check_service_interface.h>
 #include <regex>
 #include <atomic>
-#include <mutex>
 #include <signal.h>
 #include <thread>

@@ -391,9 +390,8 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
    // Initialize fit_params options (can be overridden by options)
    // fit_params: whether to auto-adjust params to fit device memory (default: true as in llama.cpp)
    params.fit_params = true;
-    // fit_params_target: target margin per device in bytes (default: 1GB per device)
-    // Initialize as vector with default value for all devices
-    params.fit_params_target = std::vector<size_t>(llama_max_devices(), 1024 * 1024 * 1024);
+    // fit_params_target: target margin per device in bytes (default: 1GB)
+    params.fit_params_target = 1024 * 1024 * 1024;
    // fit_params_min_ctx: minimum context size for fit (default: 4096)
    params.fit_params_min_ctx = 4096;

@@ -470,28 +468,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
        } else if (!strcmp(optname, "fit_params_target") || !strcmp(optname, "fit_target")) {
            if (optval != NULL) {
                try {
-                    // Value is in MiB, can be comma-separated list for multiple devices
-                    // Single value is broadcast across all devices
-                    std::string arg_next = optval_str;
-                    const std::regex regex{ R"([,/]+)" };
-                    std::sregex_token_iterator it{ arg_next.begin(), arg_next.end(), regex, -1 };
-                    std::vector<std::string> split_arg{ it, {} };
-                    if (split_arg.size() >= llama_max_devices()) {
-                        // Too many values provided
-                        continue;
-                    }
-                    if (split_arg.size() == 1) {
-                        // Single value: broadcast to all devices
-                        size_t value_mib = std::stoul(split_arg[0]);
-                        std::fill(params.fit_params_target.begin(), params.fit_params_target.end(), value_mib * 1024 * 1024);
-                    } else {
-                        // Multiple values: set per device
-                        for (size_t i = 0; i < split_arg.size() && i < params.fit_params_target.size(); i++) {
-                            params.fit_params_target[i] = std::stoul(split_arg[i]) * 1024 * 1024;
-                        }
-                    }
+                    // Value is in MiB, convert to bytes
+                    params.fit_params_target = static_cast<size_t>(std::stoi(optval_str)) * 1024 * 1024;
                } catch (const std::exception& e) {
-                    // If conversion fails, keep default value (1GB per device)
+                    // If conversion fails, keep default value (1GB)
                }
            }
        } else if (!strcmp(optname, "fit_params_min_ctx") || !strcmp(optname, "fit_ctx")) {
@@ -706,13 +686,13 @@ private:
 public:
    BackendServiceImpl(server_context& ctx) : ctx_server(ctx) {}

-    grpc::Status Health(ServerContext* /*context*/, const backend::HealthMessage* /*request*/, backend::Reply* reply) override {
+    grpc::Status Health(ServerContext* /*context*/, const backend::HealthMessage* /*request*/, backend::Reply* reply) {
        // Implement Health RPC
        reply->set_message("OK");
        return Status::OK;
    }

-    grpc::Status LoadModel(ServerContext* /*context*/, const backend::ModelOptions* request, backend::Result* result) override {
+    grpc::Status LoadModel(ServerContext* /*context*/, const backend::ModelOptions* request, backend::Result* result) {
        // Implement LoadModel RPC
        common_params params;
        params_parse(ctx_server, request, params);
@@ -729,72 +709,11 @@ public:
        LOG_INF("\n");
        LOG_INF("%s\n", common_params_get_system_info(params).c_str());
        LOG_INF("\n");
-        
-        // Capture error messages during model loading
-        struct error_capture {
-            std::string captured_error;
-            std::mutex error_mutex;
-            ggml_log_callback original_callback;
-            void* original_user_data;
-        } error_capture_data;
-        
-        // Get original log callback
-        llama_log_get(&error_capture_data.original_callback, &error_capture_data.original_user_data);
-        
-        // Set custom callback to capture errors
-        llama_log_set([](ggml_log_level level, const char * text, void * user_data) {
-            auto* capture = static_cast<error_capture*>(user_data);
-            
-            // Capture error messages
-            if (level == GGML_LOG_LEVEL_ERROR) {
-                std::lock_guard<std::mutex> lock(capture->error_mutex);
-                // Append error message, removing trailing newlines
-                std::string msg(text);
-                while (!msg.empty() && (msg.back() == '\n' || msg.back() == '\r')) {
-                    msg.pop_back();
-                }
-                if (!msg.empty()) {
-                    if (!capture->captured_error.empty()) {
-                        capture->captured_error.append("; ");
-                    }
-                    capture->captured_error.append(msg);
-                }
-            }
-            
-            // Also call original callback to preserve logging
-            if (capture->original_callback) {
-                capture->original_callback(level, text, capture->original_user_data);
-            }
-        }, &error_capture_data);
-        
        // load the model
-        bool load_success = ctx_server.load_model(params);
-        
-        // Restore original log callback
-        llama_log_set(error_capture_data.original_callback, error_capture_data.original_user_data);
-        
-        if (!load_success) {
-            std::string error_msg = "Failed to load model: " + params.model.path;
-            if (!params.mmproj.path.empty()) {
-                error_msg += " (with mmproj: " + params.mmproj.path + ")";
-            }
-            if (params.has_speculative() && !params.speculative.model.path.empty()) {
-                error_msg += " (with draft model: " + params.speculative.model.path + ")";
-            }
-            
-            // Add captured error details if available
-            {
-                std::lock_guard<std::mutex> lock(error_capture_data.error_mutex);
-                if (!error_capture_data.captured_error.empty()) {
-                    error_msg += ". Error: " + error_capture_data.captured_error;
-                } else {
-                    error_msg += ". Model file may not exist or be invalid.";
-                }
-            }
-            
-            result->set_message(error_msg);
+        if (!ctx_server.load_model(params)) {
+            result->set_message("Failed loading model");
            result->set_success(false);
-            return grpc::Status(grpc::StatusCode::INTERNAL, error_msg);
+            return Status::CANCELLED;
        }

        // Process grammar triggers now that vocab is available
@@ -1573,7 +1492,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Predict(ServerContext* context, const backend::PredictOptions* request, backend::Reply* reply) override {
+    grpc::Status Predict(ServerContext* context, const backend::PredictOptions* request, backend::Reply* reply) {
         if (params_base.model.path.empty()) {
             return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
         }
@@ -2244,7 +2163,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Embedding(ServerContext* context, const backend::PredictOptions* request, backend::EmbeddingResult* embeddingResult) override {
+    grpc::Status Embedding(ServerContext* context, const backend::PredictOptions* request, backend::EmbeddingResult* embeddingResult) {
        if (params_base.model.path.empty()) {
            return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
        }
@@ -2339,7 +2258,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status Rerank(ServerContext* context, const backend::RerankRequest* request, backend::RerankResult* rerankResult) override {
+    grpc::Status Rerank(ServerContext* context, const backend::RerankRequest* request, backend::RerankResult* rerankResult) {
        if (!params_base.embedding || params_base.pooling_type != LLAMA_POOLING_TYPE_RANK) {
            return grpc::Status(grpc::StatusCode::UNIMPLEMENTED, "This server does not support reranking. Start it with `--reranking` and without `--embedding`");
        }
@@ -2425,7 +2344,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status TokenizeString(ServerContext* /*context*/, const backend::PredictOptions* request, backend::TokenizationResponse* response) override {
+    grpc::Status TokenizeString(ServerContext* /*context*/, const backend::PredictOptions* request, backend::TokenizationResponse* response) {
        if (params_base.model.path.empty()) {
            return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, "Model not loaded");
        }
@@ -2448,7 +2367,7 @@ public:
        return grpc::Status::OK;
    }

-    grpc::Status GetMetrics(ServerContext* /*context*/, const backend::MetricsRequest* /*request*/, backend::MetricsResponse* response) override {
+    grpc::Status GetMetrics(ServerContext* /*context*/, const backend::MetricsRequest* /*request*/, backend::MetricsResponse* response) {

 // request slots data using task queue
        auto rd = ctx_server.get_response_reader();
--- a/backend/cpp/llama-cpp/package.sh
+++ b/backend/cpp/llama-cpp/package.sh
@@ -6,7 +6,6 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
-REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -38,15 +37,6 @@ else
    exit 1
 fi

-# Package GPU libraries based on BUILD_TYPE
-# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
-GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
-if [ -f "$GPU_LIB_SCRIPT" ]; then
-    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
-    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
-    package_gpu_libs
-fi
-
 echo "Packaging completed successfully" 
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/go/stablediffusion-ggml/Makefile
+++ b/backend/go/stablediffusion-ggml/Makefile
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)

 # stablediffusion.cpp (ggml)
 STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
-STABLEDIFFUSION_GGML_VERSION?=0e52afc6513cc2dea9a1a017afc4a008d5acf2b0
+STABLEDIFFUSION_GGML_VERSION?=4ff2c8c74bd17c2cfffe3a01be77743fb3efba2f

 CMAKE_ARGS+=-DGGML_MAX_NAME=128

@@ -28,12 +28,7 @@ else ifeq ($(BUILD_TYPE),clblas)
 	CMAKE_ARGS+=-DGGML_CLBLAST=ON -DCLBlast_DIR=/some/path
 # If it's hipblas we do have also to set CC=/opt/rocm/llvm/bin/clang CXX=/opt/rocm/llvm/bin/clang++
 else ifeq ($(BUILD_TYPE),hipblas)
-	ROCM_HOME ?= /opt/rocm
-	ROCM_PATH ?= /opt/rocm
-	export CXX=$(ROCM_HOME)/llvm/bin/clang++
-	export CC=$(ROCM_HOME)/llvm/bin/clang
-	AMDGPU_TARGETS?=gfx803,gfx900,gfx906,gfx908,gfx90a,gfx942,gfx1010,gfx1030,gfx1032,gfx1100,gfx1101,gfx1102,gfx1200,gfx1201
-	CMAKE_ARGS+=-DSD_HIPBLAS=ON -DGGML_HIPBLAS=ON -DAMDGPU_TARGETS=$(AMDGPU_TARGETS)
+	CMAKE_ARGS+=-DSD_HIPBLAS=ON -DGGML_HIPBLAS=ON
 else ifeq ($(BUILD_TYPE),vulkan)
 	CMAKE_ARGS+=-DSD_VULKAN=ON -DGGML_VULKAN=ON
 else ifeq ($(OS),Darwin)
--- a/backend/go/stablediffusion-ggml/package.sh
+++ b/backend/go/stablediffusion-ggml/package.sh
@@ -6,7 +6,6 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
-REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -51,15 +50,6 @@ else
    exit 1
 fi

-# Package GPU libraries based on BUILD_TYPE
-# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
-GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
-if [ -f "$GPU_LIB_SCRIPT" ]; then
-    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
-    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
-    package_gpu_libs
-fi
-
 echo "Packaging completed successfully"
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/go/whisper/Makefile
+++ b/backend/go/whisper/Makefile
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)

 # whisper.cpp version
 WHISPER_REPO?=https://github.com/ggml-org/whisper.cpp
-WHISPER_CPP_VERSION?=679bdb53dbcbfb3e42685f50c7ff367949fd4d48
+WHISPER_CPP_VERSION?=e9898ddfb908ffaa7026c66852a023889a5a7202
 SO_TARGET?=libgowhisper.so

 CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
--- a/backend/go/whisper/package.sh
+++ b/backend/go/whisper/package.sh
@@ -6,7 +6,6 @@
 set -e

 CURDIR=$(dirname "$(realpath $0)")
-REPO_ROOT="${CURDIR}/../../.."

 # Create lib directory
 mkdir -p $CURDIR/package/lib
@@ -51,15 +50,6 @@ else
    exit 1
 fi

-# Package GPU libraries based on BUILD_TYPE
-# The GPU library packaging script will detect BUILD_TYPE and copy appropriate GPU libraries
-GPU_LIB_SCRIPT="${REPO_ROOT}/scripts/build/package-gpu-libs.sh"
-if [ -f "$GPU_LIB_SCRIPT" ]; then
-    echo "Packaging GPU libraries for BUILD_TYPE=${BUILD_TYPE:-cpu}..."
-    source "$GPU_LIB_SCRIPT" "$CURDIR/package/lib"
-    package_gpu_libs
-fi
-
 echo "Packaging completed successfully"
 ls -liah $CURDIR/package/
 ls -liah $CURDIR/package/lib/
--- a/backend/index.yaml
+++ b/backend/index.yaml
@@ -275,24 +275,6 @@
    amd: "rocm-faster-whisper"
    nvidia-cuda-13: "cuda13-faster-whisper"
    nvidia-cuda-12: "cuda12-faster-whisper"
- &moonshine
-  description: |
-    Moonshine is a fast, accurate, and efficient speech-to-text transcription model using ONNX Runtime.
-    It provides real-time transcription capabilities with support for multiple model sizes and GPU acceleration.
-  urls:
-    - https://github.com/moonshine-ai/moonshine
-  tags:
-    - speech-to-text
-    - transcription
-    - ONNX
-  license: MIT
-  name: "moonshine"
-  alias: "moonshine"
-  capabilities:
-    nvidia: "cuda12-moonshine"
-    default: "cpu-moonshine"
-    nvidia-cuda-13: "cuda13-moonshine"
-    nvidia-cuda-12: "cuda12-moonshine"
 - &kokoro
  icon: https://avatars.githubusercontent.com/u/166769057?v=4
  description: |
@@ -652,6 +634,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-llama-cpp"
  mirrors:
    - localai/localai-backends:master-cpu-llama-cpp
+- !!merge <<: *llamacpp
+  name: "cuda11-llama-cpp"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-llama-cpp"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-llama-cpp
 - !!merge <<: *llamacpp
  name: "cuda12-llama-cpp"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-llama-cpp"
@@ -692,6 +679,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-metal-darwin-arm64-llama-cpp"
  mirrors:
    - localai/localai-backends:master-metal-darwin-arm64-llama-cpp
+- !!merge <<: *llamacpp
+  name: "cuda11-llama-cpp-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-llama-cpp"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-llama-cpp
 - !!merge <<: *llamacpp
  name: "cuda12-llama-cpp-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-llama-cpp"
@@ -763,6 +755,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-whisper"
  mirrors:
    - localai/localai-backends:master-cpu-whisper
+- !!merge <<: *whispercpp
+  name: "cuda11-whisper"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-whisper"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-whisper
 - !!merge <<: *whispercpp
  name: "cuda12-whisper"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-whisper"
@@ -803,6 +800,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-metal-darwin-arm64-whisper"
  mirrors:
    - localai/localai-backends:master-metal-darwin-arm64-whisper
+- !!merge <<: *whispercpp
+  name: "cuda11-whisper-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-whisper"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-whisper
 - !!merge <<: *whispercpp
  name: "cuda12-whisper-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-whisper"
@@ -877,6 +879,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-sycl-f16-stablediffusion-ggml"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-sycl-f16-stablediffusion-ggml
+- !!merge <<: *stablediffusionggml
+  name: "cuda11-stablediffusion-ggml"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-stablediffusion-ggml"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-stablediffusion-ggml
 - !!merge <<: *stablediffusionggml
  name: "cuda12-stablediffusion-ggml-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-stablediffusion-ggml"
@@ -892,6 +899,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-intel-sycl-f16-stablediffusion-ggml"
  mirrors:
    - localai/localai-backends:master-gpu-intel-sycl-f16-stablediffusion-ggml
+- !!merge <<: *stablediffusionggml
+  name: "cuda11-stablediffusion-ggml-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-stablediffusion-ggml"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-stablediffusion-ggml
 - !!merge <<: *stablediffusionggml
  name: "nvidia-l4t-arm64-stablediffusion-ggml-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-arm64-stablediffusion-ggml"
@@ -1042,6 +1054,11 @@
    intel: "intel-rerankers-development"
    amd: "rocm-rerankers-development"
    nvidia-cuda-13: "cuda13-rerankers-development"
+- !!merge <<: *rerankers
+  name: "cuda11-rerankers"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-rerankers"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-rerankers
 - !!merge <<: *rerankers
  name: "cuda12-rerankers"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-rerankers"
@@ -1057,6 +1074,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-rerankers"
  mirrors:
    - localai/localai-backends:latest-gpu-rocm-hipblas-rerankers
+- !!merge <<: *rerankers
+  name: "cuda11-rerankers-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-rerankers"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-rerankers
 - !!merge <<: *rerankers
  name: "cuda12-rerankers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-rerankers"
@@ -1105,6 +1127,16 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-transformers"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-transformers
+- !!merge <<: *transformers
+  name: "cuda11-transformers-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-transformers"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-transformers
+- !!merge <<: *transformers
+  name: "cuda11-transformers"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-transformers"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-transformers
 - !!merge <<: *transformers
  name: "cuda12-transformers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-transformers"
@@ -1181,11 +1213,21 @@
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-rocm-hipblas-diffusers"
  mirrors:
    - localai/localai-backends:latest-gpu-rocm-hipblas-diffusers
+- !!merge <<: *diffusers
+  name: "cuda11-diffusers"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-diffusers"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-diffusers
 - !!merge <<: *diffusers
  name: "intel-diffusers"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-intel-diffusers"
  mirrors:
    - localai/localai-backends:latest-gpu-intel-diffusers
+- !!merge <<: *diffusers
+  name: "cuda11-diffusers-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-diffusers"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-diffusers
 - !!merge <<: *diffusers
  name: "cuda12-diffusers-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-diffusers"
@@ -1227,11 +1269,21 @@
  capabilities:
    nvidia: "cuda12-exllama2-development"
    intel: "intel-exllama2-development"
+- !!merge <<: *exllama2
+  name: "cuda11-exllama2"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-exllama2"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-exllama2
 - !!merge <<: *exllama2
  name: "cuda12-exllama2"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-exllama2"
  mirrors:
    - localai/localai-backends:latest-gpu-nvidia-cuda-12-exllama2
+- !!merge <<: *exllama2
+  name: "cuda11-exllama2-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-exllama2"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-exllama2
 - !!merge <<: *exllama2
  name: "cuda12-exllama2-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-exllama2"
@@ -1245,6 +1297,11 @@
    intel: "intel-kokoro-development"
    amd: "rocm-kokoro-development"
    nvidia-l4t: "nvidia-l4t-kokoro-development"
+- !!merge <<: *kokoro
+  name: "cuda11-kokoro-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-kokoro"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-kokoro
 - !!merge <<: *kokoro
  name: "cuda12-kokoro-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-kokoro"
@@ -1275,6 +1332,11 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-nvidia-l4t-kokoro"
  mirrors:
    - localai/localai-backends:master-nvidia-l4t-kokoro
+- !!merge <<: *kokoro
+  name: "cuda11-kokoro"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-kokoro"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-kokoro
 - !!merge <<: *kokoro
  name: "cuda12-kokoro"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-kokoro"
@@ -1303,6 +1365,11 @@
    intel: "intel-faster-whisper-development"
    amd: "rocm-faster-whisper-development"
    nvidia-cuda-13: "cuda13-faster-whisper-development"
+- !!merge <<: *faster-whisper
+  name: "cuda11-faster-whisper"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-faster-whisper"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-faster-whisper
 - !!merge <<: *faster-whisper
  name: "cuda12-faster-whisper-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-faster-whisper"
@@ -1333,44 +1400,6 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-13-faster-whisper"
  mirrors:
    - localai/localai-backends:master-gpu-nvidia-cuda-13-faster-whisper
-## moonshine
- !!merge <<: *moonshine
-  name: "moonshine-development"
-  capabilities:
-    nvidia: "cuda12-moonshine-development"
-    default: "cpu-moonshine-development"
-    nvidia-cuda-13: "cuda13-moonshine-development"
-    nvidia-cuda-12: "cuda12-moonshine-development"
- !!merge <<: *moonshine
-  name: "cpu-moonshine"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-cpu-moonshine"
-  mirrors:
-    - localai/localai-backends:latest-cpu-moonshine
- !!merge <<: *moonshine
-  name: "cpu-moonshine-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-cpu-moonshine"
-  mirrors:
-    - localai/localai-backends:master-cpu-moonshine
- !!merge <<: *moonshine
-  name: "cuda12-moonshine"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-moonshine"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-12-moonshine
- !!merge <<: *moonshine
-  name: "cuda12-moonshine-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-moonshine"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-12-moonshine
- !!merge <<: *moonshine
-  name: "cuda13-moonshine"
-  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-13-moonshine"
-  mirrors:
-    - localai/localai-backends:latest-gpu-nvidia-cuda-13-moonshine
- !!merge <<: *moonshine
-  name: "cuda13-moonshine-development"
-  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-13-moonshine"
-  mirrors:
-    - localai/localai-backends:master-gpu-nvidia-cuda-13-moonshine
 ## coqui

 - !!merge <<: *coqui
@@ -1379,11 +1408,21 @@
    nvidia: "cuda12-coqui-development"
    intel: "intel-coqui-development"
    amd: "rocm-coqui-development"
+- !!merge <<: *coqui
+  name: "cuda11-coqui"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-coqui"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-coqui
 - !!merge <<: *coqui
  name: "cuda12-coqui"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-coqui"
  mirrors:
    - localai/localai-backends:latest-gpu-nvidia-cuda-12-coqui
+- !!merge <<: *coqui
+  name: "cuda11-coqui-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-coqui"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-coqui
 - !!merge <<: *coqui
  name: "cuda12-coqui-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-coqui"
@@ -1416,6 +1455,16 @@
    nvidia: "cuda12-bark-development"
    intel: "intel-bark-development"
    amd: "rocm-bark-development"
+- !!merge <<: *bark
+  name: "cuda11-bark-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-bark"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-bark
+- !!merge <<: *bark
+  name: "cuda11-bark"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-bark"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-bark
 - !!merge <<: *bark
  name: "rocm-bark-development"
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-rocm-hipblas-bark"
@@ -1497,6 +1546,16 @@
  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-12-chatterbox"
  mirrors:
    - localai/localai-backends:master-gpu-nvidia-cuda-12-chatterbox
+- !!merge <<: *chatterbox
+  name: "cuda11-chatterbox"
+  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-11-chatterbox"
+  mirrors:
+    - localai/localai-backends:latest-gpu-nvidia-cuda-11-chatterbox
+- !!merge <<: *chatterbox
+  name: "cuda11-chatterbox-development"
+  uri: "quay.io/go-skynet/local-ai-backends:master-gpu-nvidia-cuda-11-chatterbox"
+  mirrors:
+    - localai/localai-backends:master-gpu-nvidia-cuda-11-chatterbox
 - !!merge <<: *chatterbox
  name: "cuda12-chatterbox"
  uri: "quay.io/go-skynet/local-ai-backends:latest-gpu-nvidia-cuda-12-chatterbox"
--- a/backend/python/README.md
+++ b/backend/python/README.md
@@ -85,7 +85,7 @@ runUnittests
 The build system automatically detects and configures for different hardware:

 - **CPU** - Standard CPU-only builds
- **CUDA** - NVIDIA GPU acceleration (supports CUDA 12/13)
+- **CUDA** - NVIDIA GPU acceleration (supports CUDA 11/12)
 - **Intel** - Intel XPU/GPU optimization
 - **MLX** - Apple Silicon (M1/M2/M3) optimization
 - **HIP** - AMD GPU acceleration
@@ -95,8 +95,8 @@ The build system automatically detects and configures for different hardware:
 Backends can specify hardware-specific dependencies:
 - `requirements.txt` - Base requirements
 - `requirements-cpu.txt` - CPU-specific packages
+- `requirements-cublas11.txt` - CUDA 11 packages
 - `requirements-cublas12.txt` - CUDA 12 packages
- `requirements-cublas13.txt` - CUDA 13 packages
 - `requirements-intel.txt` - Intel-optimized packages
 - `requirements-mps.txt` - Apple Silicon packages

--- a/backend/python/bark/requirements-cublas11.txt
+++ b/backend/python/bark/requirements-cublas11.txt
@@ -0,0 +1,5 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.4.1+cu118
+torchaudio==2.4.1+cu118
+transformers
+accelerate
--- a/backend/python/bark/requirements-hipblas.txt
+++ b/backend/python/bark/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
-torchaudio==2.8.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
+torch==2.4.1+rocm6.0
+torchaudio==2.4.1+rocm6.0
 transformers
 accelerate
--- a/backend/python/chatterbox/requirements-cublas11.txt
+++ b/backend/python/chatterbox/requirements-cublas11.txt
@@ -0,0 +1,8 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.6.0+cu118
+torchaudio==2.6.0+cu118
+transformers==4.46.3
+numpy>=1.24.0,<1.26.0
+# https://github.com/mudler/LocalAI/pull/6240#issuecomment-3329518289
+chatterbox-tts@git+https://git@github.com/mudler/chatterbox.git@faster
+accelerate
--- a/backend/python/chatterbox/requirements-hipblas.txt
+++ b/backend/python/chatterbox/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.9.1+rocm6.4
-torchaudio==2.9.1+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
+torch==2.6.0+rocm6.1
+torchaudio==2.6.0+rocm6.1
 transformers
 numpy>=1.24.0,<1.26.0
 # https://github.com/mudler/LocalAI/pull/6240#issuecomment-3329518289
--- a/backend/python/common/libbackend.sh
+++ b/backend/python/common/libbackend.sh
@@ -1,7 +1,7 @@
 #!/usr/bin/env bash
 set -euo pipefail

-#
+# 
 # use the library by adding the following line to a script:
 # source $(dirname $0)/../common/libbackend.sh
 #
@@ -206,8 +206,8 @@ function init() {

 # getBuildProfile will inspect the system to determine which build profile is appropriate:
 # returns one of the following:
+# - cublas11
 # - cublas12
-# - cublas13
 # - hipblas
 # - intel
 function getBuildProfile() {
@@ -392,7 +392,7 @@ function runProtogen() {
 #  - requirements-${BUILD_TYPE}.txt
 #  - requirements-${BUILD_PROFILE}.txt
 #
-# BUILD_PROFILE is a more specific version of BUILD_TYPE, ex: cuda-12 or cuda-13
+# BUILD_PROFILE is a more specific version of BUILD_TYPE, ex: cuda-11 or cuda-12
 # it can also include some options that we do not have BUILD_TYPES for, ex: intel
 #
 # NOTE: for BUILD_PROFILE==intel, this function does NOT automatically use the Intel python package index.
@@ -465,14 +465,6 @@ function startBackend() {
    if [ "x${PORTABLE_PYTHON}" == "xtrue" ] || [ -x "$(_portable_python)" ]; then
        _makeVenvPortable --update-pyvenv-cfg
    fi
-
-    # Set up GPU library paths if a lib directory exists
-    # This allows backends to include their own GPU libraries (CUDA, ROCm, etc.)
-    if [ -d "${EDIR}/lib" ]; then
-        export LD_LIBRARY_PATH="${EDIR}/lib:${LD_LIBRARY_PATH:-}"
-        echo "Added ${EDIR}/lib to LD_LIBRARY_PATH for GPU libraries"
-    fi
-
    if [ ! -z "${BACKEND_FILE:-}" ]; then
        exec "${EDIR}/venv/bin/python" "${BACKEND_FILE}" "$@"
    elif [ -e "${MY_DIR}/server.py" ]; then
--- a/backend/python/common/template/requirements-hipblas.txt
+++ b/backend/python/common/template/requirements-hipblas.txt
@@ -1,2 +1,2 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
 torch
--- a/backend/python/coqui/requirements-cublas11.txt
+++ b/backend/python/coqui/requirements-cublas11.txt
@@ -0,0 +1,6 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.4.1+cu118
+torchaudio==2.4.1+cu118
+transformers==4.48.3
+accelerate
+coqui-tts
--- a/backend/python/coqui/requirements-hipblas.txt
+++ b/backend/python/coqui/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
-torchaudio==2.8.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
+torch==2.4.1+rocm6.0
+torchaudio==2.4.1+rocm6.0
 transformers==4.48.3
 accelerate
 coqui-tts
--- a/backend/python/diffusers/install.sh
+++ b/backend/python/diffusers/install.sh
@@ -17,7 +17,7 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
 fi

 # Use python 3.12 for l4t
-if [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
+if [ "x${BUILD_PROFILE}" == "xl4t12" ] || [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
  PYTHON_VERSION="3.12"
  PYTHON_PATCH="12"
  PY_STANDALONE_TAG="20251120"
--- a/backend/python/diffusers/requirements-cublas11.txt
+++ b/backend/python/diffusers/requirements-cublas11.txt
@@ -0,0 +1,12 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+git+https://github.com/huggingface/diffusers
+opencv-python
+transformers
+torchvision==0.22.1
+accelerate
+compel
+peft
+sentencepiece
+torch==2.7.1
+optimum-quanto
+ftfy
--- a/backend/python/diffusers/requirements-hipblas.txt
+++ b/backend/python/diffusers/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
-torchvision==0.23.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+torch==2.7.1+rocm6.3
+torchvision==0.22.1+rocm6.3
 git+https://github.com/huggingface/diffusers
 opencv-python
 transformers
--- a/backend/python/exllama2/requirements-cublas11.txt
+++ b/backend/python/exllama2/requirements-cublas11.txt
@@ -0,0 +1,4 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.4.1+cu118
+transformers
+accelerate
--- a/backend/python/faster-whisper/requirements-cublas11.txt
+++ b/backend/python/faster-whisper/requirements-cublas11.txt
@@ -0,0 +1,9 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.4.1+cu118
+faster-whisper
+opencv-python
+accelerate
+compel
+peft
+sentencepiece
+optimum-quanto
--- a/backend/python/faster-whisper/requirements-hipblas.txt
+++ b/backend/python/faster-whisper/requirements-hipblas.txt
@@ -1,3 +1,3 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
 torch
 faster-whisper
--- a/backend/python/kokoro/requirements-cublas11.txt
+++ b/backend/python/kokoro/requirements-cublas11.txt
@@ -0,0 +1,7 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.7.1+cu118
+torchaudio==2.7.1+cu118
+transformers
+accelerate
+kokoro
+soundfile
--- a/backend/python/kokoro/requirements-hipblas.txt
+++ b/backend/python/kokoro/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
-torchaudio==2.8.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+torch==2.7.1+rocm6.3
+torchaudio==2.7.1+rocm6.3
 transformers
 accelerate
 kokoro
--- a/backend/python/moonshine/Makefile
+++ b/backend/python/moonshine/Makefile
@@ -1,16 +0,0 @@
-.DEFAULT_GOAL := install
-
-.PHONY: install
-install:
-	bash install.sh
-
-.PHONY: protogen-clean
-protogen-clean:
-	$(RM) backend_pb2_grpc.py backend_pb2.py
-
-.PHONY: clean
-clean: protogen-clean
-	rm -rf venv __pycache__
-
-test: install
-	bash test.sh
--- a/backend/python/moonshine/backend.py
+++ b/backend/python/moonshine/backend.py
@@ -1,113 +0,0 @@
-#!/usr/bin/env python3
-"""
-This is an extra gRPC server of LocalAI for Moonshine transcription
-"""
-from concurrent import futures
-import time
-import argparse
-import signal
-import sys
-import os
-import backend_pb2
-import backend_pb2_grpc
-import moonshine_onnx
-
-import grpc
-
-
-_ONE_DAY_IN_SECONDS = 60 * 60 * 24
-
-# If MAX_WORKERS are specified in the environment use it, otherwise default to 1
-MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
-
-# Implement the BackendServicer class with the service methods
-class BackendServicer(backend_pb2_grpc.BackendServicer):
-    """
-    BackendServicer is the class that implements the gRPC service
-    """
-    def Health(self, request, context):
-        return backend_pb2.Reply(message=bytes("OK", 'utf-8'))
-    
-    def LoadModel(self, request, context):
-        try:
-            print("Preparing models, please wait", file=sys.stderr)
-            # Store the model name for use in transcription
-            # Model name format: e.g., "moonshine/tiny"
-            self.model_name = request.Model
-            print(f"Model name set to: {self.model_name}", file=sys.stderr)
-        except Exception as err:
-            return backend_pb2.Result(success=False, message=f"Unexpected {err=}, {type(err)=}")
-        return backend_pb2.Result(message="Model loaded successfully", success=True)
-
-    def AudioTranscription(self, request, context):
-        resultSegments = []
-        text = ""
-        try:
-            # moonshine_onnx.transcribe returns a list of strings
-            transcriptions = moonshine_onnx.transcribe(request.dst, self.model_name)
-            
-            # Combine all transcriptions into a single text
-            if isinstance(transcriptions, list):
-                text = " ".join(transcriptions)
-                # Create segments for each transcription in the list
-                for id, trans in enumerate(transcriptions):
-                    # Since moonshine doesn't provide timing info, we'll create a single segment
-                    # with id and text, using approximate timing
-                    resultSegments.append(backend_pb2.TranscriptSegment(
-                        id=id, 
-                        start=0, 
-                        end=0, 
-                        text=trans
-                    ))
-            else:
-                # Handle case where it's not a list (shouldn't happen, but be safe)
-                text = str(transcriptions)
-                resultSegments.append(backend_pb2.TranscriptSegment(
-                    id=0,
-                    start=0,
-                    end=0,
-                    text=text
-                ))
-        except Exception as err:
-            print(f"Unexpected {err=}, {type(err)=}", file=sys.stderr)
-            return backend_pb2.TranscriptResult(segments=[], text="")
-
-        return backend_pb2.TranscriptResult(segments=resultSegments, text=text)
-
-def serve(address):
-    server = grpc.server(futures.ThreadPoolExecutor(max_workers=MAX_WORKERS),
-        options=[
-            ('grpc.max_message_length', 50 * 1024 * 1024),  # 50MB
-            ('grpc.max_send_message_length', 50 * 1024 * 1024),  # 50MB
-            ('grpc.max_receive_message_length', 50 * 1024 * 1024),  # 50MB
-        ])
-    backend_pb2_grpc.add_BackendServicer_to_server(BackendServicer(), server)
-    server.add_insecure_port(address)
-    server.start()
-    print("Server started. Listening on: " + address, file=sys.stderr)
-
-    # Define the signal handler function
-    def signal_handler(sig, frame):
-        print("Received termination signal. Shutting down...")
-        server.stop(0)
-        sys.exit(0)
-
-    # Set the signal handlers for SIGINT and SIGTERM
-    signal.signal(signal.SIGINT, signal_handler)
-    signal.signal(signal.SIGTERM, signal_handler)
-
-    try:
-        while True:
-            time.sleep(_ONE_DAY_IN_SECONDS)
-    except KeyboardInterrupt:
-        server.stop(0)
-
-if __name__ == "__main__":
-    parser = argparse.ArgumentParser(description="Run the gRPC server.")
-    parser.add_argument(
-        "--addr", default="localhost:50051", help="The address to bind the server to."
-    )
-    args = parser.parse_args()
-
-    serve(args.addr)
-
--- a/backend/python/moonshine/install.sh
+++ b/backend/python/moonshine/install.sh
@@ -1,12 +0,0 @@
-#!/bin/bash
-set -e
-
-backend_dir=$(dirname $0)
-if [ -d $backend_dir/common ]; then
-    source $backend_dir/common/libbackend.sh
-else
-    source $backend_dir/../common/libbackend.sh
-fi
-
-installRequirements
-
--- a/backend/python/moonshine/protogen.sh
+++ b/backend/python/moonshine/protogen.sh
@@ -1,12 +0,0 @@
-#!/bin/bash
-set -e
-
-backend_dir=$(dirname $0)
-if [ -d $backend_dir/common ]; then
-    source $backend_dir/common/libbackend.sh
-else
-    source $backend_dir/../common/libbackend.sh
-fi
-
-python3 -m grpc_tools.protoc -I../.. -I./ --python_out=. --grpc_python_out=. backend.proto
-
--- a/backend/python/moonshine/requirements.txt
+++ b/backend/python/moonshine/requirements.txt
@@ -1,4 +0,0 @@
-grpcio==1.71.0
-protobuf
-grpcio-tools
-useful-moonshine-onnx@git+https://git@github.com/moonshine-ai/moonshine.git#subdirectory=moonshine-onnx
--- a/backend/python/moonshine/run.sh
+++ b/backend/python/moonshine/run.sh
@@ -1,10 +0,0 @@
-#!/bin/bash
-backend_dir=$(dirname $0)
-if [ -d $backend_dir/common ]; then
-    source $backend_dir/common/libbackend.sh
-else
-    source $backend_dir/../common/libbackend.sh
-fi
-
-startBackend $@
-
--- a/backend/python/moonshine/test.py
+++ b/backend/python/moonshine/test.py
@@ -1,139 +0,0 @@
-"""
-A test script to test the gRPC service for Moonshine transcription
-"""
-import unittest
-import subprocess
-import time
-import os
-import tempfile
-import shutil
-import backend_pb2
-import backend_pb2_grpc
-
-import grpc
-
-
-class TestBackendServicer(unittest.TestCase):
-    """
-    TestBackendServicer is the class that tests the gRPC service
-    """
-    def setUp(self):
-        """
-        This method sets up the gRPC service by starting the server
-        """
-        self.service = subprocess.Popen(["python3", "backend.py", "--addr", "localhost:50051"])
-        time.sleep(10)
-
-    def tearDown(self) -> None:
-        """
-        This method tears down the gRPC service by terminating the server
-        """
-        self.service.terminate()
-        self.service.wait()
-
-    def test_server_startup(self):
-        """
-        This method tests if the server starts up successfully
-        """
-        try:
-            self.setUp()
-            with grpc.insecure_channel("localhost:50051") as channel:
-                stub = backend_pb2_grpc.BackendStub(channel)
-                response = stub.Health(backend_pb2.HealthMessage())
-                self.assertEqual(response.message, b'OK')
-        except Exception as err:
-            print(err)
-            self.fail("Server failed to start")
-        finally:
-            self.tearDown()
-
-    def test_load_model(self):
-        """
-        This method tests if the model is loaded successfully
-        """
-        try:
-            self.setUp()
-            with grpc.insecure_channel("localhost:50051") as channel:
-                stub = backend_pb2_grpc.BackendStub(channel)
-                response = stub.LoadModel(backend_pb2.ModelOptions(Model="moonshine/tiny"))
-                self.assertTrue(response.success)
-                self.assertEqual(response.message, "Model loaded successfully")
-        except Exception as err:
-            print(err)
-            self.fail("LoadModel service failed")
-        finally:
-            self.tearDown()
-
-    def test_audio_transcription(self):
-        """
-        This method tests if audio transcription works successfully
-        """
-        # Create a temporary directory for the audio file
-        temp_dir = tempfile.mkdtemp()
-        audio_file = os.path.join(temp_dir, 'audio.wav')
-        
-        try:
-            # Download the audio file to the temporary directory
-            print(f"Downloading audio file to {audio_file}...")
-            url = "https://cdn.openai.com/whisper/draft-20220913a/micro-machines.wav"
-            result = subprocess.run(
-                ["wget", "-q", url, "-O", audio_file],
-                capture_output=True,
-                text=True
-            )
-            if result.returncode != 0:
-                self.fail(f"Failed to download audio file: {result.stderr}")
-            
-            # Verify the file was downloaded
-            if not os.path.exists(audio_file):
-                self.fail(f"Audio file was not downloaded to {audio_file}")
-            
-            self.setUp()
-            with grpc.insecure_channel("localhost:50051") as channel:
-                stub = backend_pb2_grpc.BackendStub(channel)
-                # Load the model first
-                load_response = stub.LoadModel(backend_pb2.ModelOptions(Model="moonshine/tiny"))
-                self.assertTrue(load_response.success)
-                
-                # Perform transcription
-                transcript_request = backend_pb2.TranscriptRequest(dst=audio_file)
-                transcript_response = stub.AudioTranscription(transcript_request)
-                
-                # Print the transcribed text for debugging
-                print(f"Transcribed text: {transcript_response.text}")
-                print(f"Number of segments: {len(transcript_response.segments)}")
-                
-                # Verify response structure
-                self.assertIsNotNone(transcript_response)
-                self.assertIsNotNone(transcript_response.text)
-                # Protobuf repeated fields return a sequence, not a list
-                self.assertIsNotNone(transcript_response.segments)
-                # Check if segments is iterable (has length)
-                self.assertGreaterEqual(len(transcript_response.segments), 0)
-                
-                # Verify the transcription contains the expected text
-                expected_text = "This is the micro machine man presenting the most midget miniature"
-                self.assertIn(
-                    expected_text.lower(),
-                    transcript_response.text.lower(),
-                    f"Expected text '{expected_text}' not found in transcription: '{transcript_response.text}'"
-                )
-                
-                # If we got segments, verify they have the expected structure
-                if len(transcript_response.segments) > 0:
-                    segment = transcript_response.segments[0]
-                    self.assertIsNotNone(segment.text)
-                    self.assertIsInstance(segment.id, int)
-                else:
-                    # Even if no segments, we should have text
-                    self.assertIsNotNone(transcript_response.text)
-                    self.assertGreater(len(transcript_response.text), 0)
-        except Exception as err:
-            print(err)
-            self.fail("AudioTranscription service failed")
-        finally:
-            self.tearDown()
-            # Clean up the temporary directory
-            if os.path.exists(temp_dir):
-                shutil.rmtree(temp_dir)
-
--- a/backend/python/moonshine/test.sh
+++ b/backend/python/moonshine/test.sh
@@ -1,12 +0,0 @@
-#!/bin/bash
-set -e
-
-backend_dir=$(dirname $0)
-if [ -d $backend_dir/common ]; then
-    source $backend_dir/common/libbackend.sh
-else
-    source $backend_dir/../common/libbackend.sh
-fi
-
-runUnittests
-
--- a/backend/python/neutts/requirements-hipblas.txt
+++ b/backend/python/neutts/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+torch==2.8.0+rocm6.3
 transformers==4.56.1
 accelerate
 librosa==0.11.0
--- a/backend/python/rerankers/requirements-cublas11.txt
+++ b/backend/python/rerankers/requirements-cublas11.txt
@@ -0,0 +1,5 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+transformers
+accelerate
+torch==2.4.1+cu118
+rerankers[transformers]
--- a/backend/python/rerankers/requirements-hipblas.txt
+++ b/backend/python/rerankers/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.0
 transformers
 accelerate
-torch==2.8.0+rocm6.4
+torch==2.4.1+rocm6.0
 rerankers[transformers]
--- a/backend/python/rfdetr/requirements-cublas11.txt
+++ b/backend/python/rfdetr/requirements-cublas11.txt
@@ -0,0 +1,8 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.7.1+cu118
+rfdetr
+opencv-python
+accelerate
+inference
+peft
+optimum-quanto
--- a/backend/python/rfdetr/requirements-hipblas.txt
+++ b/backend/python/rfdetr/requirements-hipblas.txt
@@ -1,6 +1,6 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
-torchvision==0.23.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+torch==2.7.1+rocm6.3
+torchvision==0.22.1+rocm6.3
 rfdetr
 opencv-python
 accelerate
--- a/backend/python/transformers/requirements-cublas11.txt
+++ b/backend/python/transformers/requirements-cublas11.txt
@@ -0,0 +1,10 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+torch==2.7.1+cu118
+llvmlite==0.43.0
+numba==0.60.0
+accelerate
+transformers
+bitsandbytes
+outetts
+sentence-transformers==5.2.0
+protobuf==6.33.2
--- a/backend/python/transformers/requirements-hipblas.txt
+++ b/backend/python/transformers/requirements-hipblas.txt
@@ -1,5 +1,5 @@
--extra-index-url https://download.pytorch.org/whl/rocm6.4
-torch==2.8.0+rocm6.4
+--extra-index-url https://download.pytorch.org/whl/rocm6.3
+torch==2.7.1+rocm6.3
 accelerate
 transformers
 llvmlite==0.43.0
--- a/backend/python/vibevoice/install.sh
+++ b/backend/python/vibevoice/install.sh
@@ -17,7 +17,7 @@ if [ "x${BUILD_PROFILE}" == "xintel" ]; then
 fi

 # Use python 3.12 for l4t
-if [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
+if [ "x${BUILD_PROFILE}" == "xl4t12" ] || [ "x${BUILD_PROFILE}" == "xl4t13" ]; then
  PYTHON_VERSION="3.12"
  PYTHON_PATCH="12"
  PY_STANDALONE_TAG="20251120"
--- a/backend/python/vibevoice/requirements-cublas11.txt
+++ b/backend/python/vibevoice/requirements-cublas11.txt
@@ -0,0 +1,22 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+git+https://github.com/huggingface/diffusers
+opencv-python
+transformers==4.51.3
+torchvision==0.22.1
+accelerate
+compel
+peft
+sentencepiece
+torch==2.7.1
+optimum-quanto
+ftfy
+llvmlite>=0.40.0
+numba>=0.57.0
+tqdm
+numpy
+scipy
+librosa
+ml-collections
+absl-py
+gradio
+av
--- a/backend/python/vllm/install.sh
+++ b/backend/python/vllm/install.sh
@@ -28,7 +28,7 @@ fi

 # We don't embed this into the images as it is a large dependency and not always needed.
 # Besides, the speed inference are not actually usable in the current state for production use-cases.
-if [ "x${BUILD_TYPE}" == "x" ] && [ "x${FROM_SOURCE:-}" == "xtrue" ]; then
+if [ "x${BUILD_TYPE}" == "x" ] && [ "x${FROM_SOURCE}" == "xtrue" ]; then
        ensureVenv
        # https://docs.vllm.ai/en/v0.6.1/getting_started/cpu-installation.html
        if [ ! -d vllm ]; then
--- a/backend/python/vllm/requirements-cublas11-after.txt
+++ b/backend/python/vllm/requirements-cublas11-after.txt
@@ -0,0 +1 @@
+flash-attn
--- a/backend/python/vllm/requirements-cublas11.txt
+++ b/backend/python/vllm/requirements-cublas11.txt
@@ -0,0 +1,5 @@
+--extra-index-url https://download.pytorch.org/whl/cu118
+accelerate
+torch==2.7.0+cu118
+transformers
+bitsandbytes
--- a/backend/python/vllm/requirements-hipblas.txt
+++ b/backend/python/vllm/requirements-hipblas.txt
@@ -1,4 +1,4 @@
--extra-index-url https://download.pytorch.org/whl/nightly/rocm6.4
+--extra-index-url https://download.pytorch.org/whl/nightly/rocm6.3
 accelerate
 torch
 transformers
--- a/core/http/endpoints/openai/chat.go
+++ b/core/http/endpoints/openai/chat.go
@@ -66,111 +66,10 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator
 	}
 	processTools := func(noAction string, prompt string, req *schema.OpenAIRequest, config *config.ModelConfig, loader *model.ModelLoader, responses chan schema.OpenAIResponse, extraUsage bool) error {
 		result := ""
-		lastEmittedCount := 0
 		_, tokenUsage, err := ComputeChoices(req, prompt, config, cl, startupOptions, loader, func(s string, c *[]schema.Choice) {}, func(s string, usage backend.TokenUsage) bool {
 			result += s
-			// Try incremental XML parsing for streaming support using iterative parser
-			// This allows emitting partial tool calls as they're being generated
-			cleanedResult := functions.CleanupLLMResult(result, config.FunctionsConfig)
-
-			// Determine XML format from config
-			var xmlFormat *functions.XMLToolCallFormat
-			if config.FunctionsConfig.XMLFormat != nil {
-				xmlFormat = config.FunctionsConfig.XMLFormat
-			} else if config.FunctionsConfig.XMLFormatPreset != "" {
-				xmlFormat = functions.GetXMLFormatPreset(config.FunctionsConfig.XMLFormatPreset)
-			}
-
-			// Use iterative parser for streaming (partial parsing enabled)
-			// Try XML parsing first
-			partialResults, parseErr := functions.ParseXMLIterative(cleanedResult, xmlFormat, true)
-			if parseErr == nil && len(partialResults) > 0 {
-				// Emit new XML tool calls that weren't emitted before
-				if len(partialResults) > lastEmittedCount {
-					for i := lastEmittedCount; i < len(partialResults); i++ {
-						toolCall := partialResults[i]
-						initialMessage := schema.OpenAIResponse{
-							ID:      id,
-							Created: created,
-							Model:   req.Model,
-							Choices: []schema.Choice{{
-								Delta: &schema.Message{
-									Role: "assistant",
-									ToolCalls: []schema.ToolCall{
-										{
-											Index: i,
-											ID:    id,
-											Type:  "function",
-											FunctionCall: schema.FunctionCall{
-												Name: toolCall.Name,
-											},
-										},
-									},
-								},
-								Index:        0,
-								FinishReason: nil,
-							}},
-							Object: "chat.completion.chunk",
-						}
-						select {
-						case responses <- initialMessage:
-						default:
-						}
-					}
-					lastEmittedCount = len(partialResults)
-				}
-			} else {
-				// Try JSON tool call parsing for streaming
-				// Check if the result looks like JSON tool calls
-				jsonResults, jsonErr := functions.ParseJSONIterative(cleanedResult, true)
-				if jsonErr == nil && len(jsonResults) > 0 {
-					// Check if these are tool calls (have "name" and optionally "arguments")
-					for _, jsonObj := range jsonResults {
-						if name, ok := jsonObj["name"].(string); ok && name != "" {
-							// This looks like a tool call
-							args := "{}"
-							if argsVal, ok := jsonObj["arguments"]; ok {
-								if argsStr, ok := argsVal.(string); ok {
-									args = argsStr
-								} else {
-									argsBytes, _ := json.Marshal(argsVal)
-									args = string(argsBytes)
-								}
-							}
-							// Emit tool call
-							initialMessage := schema.OpenAIResponse{
-								ID:      id,
-								Created: created,
-								Model:   req.Model,
-								Choices: []schema.Choice{{
-									Delta: &schema.Message{
-										Role: "assistant",
-										ToolCalls: []schema.ToolCall{
-											{
-												Index: lastEmittedCount,
-												ID:    id,
-												Type:  "function",
-												FunctionCall: schema.FunctionCall{
-													Name:      name,
-													Arguments: args,
-												},
-											},
-										},
-									},
-									Index:        0,
-									FinishReason: nil,
-								}},
-								Object: "chat.completion.chunk",
-							}
-							select {
-							case responses <- initialMessage:
-							default:
-							}
-							lastEmittedCount++
-						}
-					}
-				}
-			}
+			// TODO: Change generated BNF grammar to be compliant with the schema so we can
+			// stream the result token by token here.
 			return true
 		})
 		if err != nil {
--- a/core/http/static/chat.js
+++ b/core/http/static/chat.js
@@ -750,7 +750,6 @@ function stopRequest() {
  if (!activeChat) return;
  
  const request = activeRequests.get(activeChat.id);
-  const requestModel = request?.model || null; // Get model before deleting request
  if (request) {
    if (request.controller) {
      request.controller.abort();
@@ -780,8 +779,7 @@ function stopRequest() {
    `<span class='error'>Request cancelled by user</span>`,
    null,
    null,
-    activeChat.id,
-    requestModel
+    activeChat.id
  );
 }

@@ -1233,8 +1231,7 @@ async function promptGPT(systemPrompt, input) {
      startTime: requestStartTime,
      tokensReceived: 0,
      interval: null,
-      maxTokensPerSecond: 0,
-      model: model // Store the model used for this request
+      maxTokensPerSecond: 0
    });
    
    // Update reactive tracking for UI indicators
@@ -1274,27 +1271,21 @@ async function promptGPT(systemPrompt, input) {
        return;
      } else {
        // Timeout error (controller was aborted by timeout, not user)
-        const request = activeRequests.get(chatId);
-        const requestModel = request?.model || null;
        chatStore.add(
          "assistant",
          `<span class='error'>Request timeout: MCP processing is taking longer than expected. Please try again.</span>`,
          null,
          null,
-          chatId,
-          requestModel
+          chatId
        );
      }
    } else {
-      const request = activeRequests.get(chatId);
-      const requestModel = request?.model || null;
      chatStore.add(
        "assistant",
        `<span class='error'>Network Error: ${error.message}</span>`,
        null,
        null,
-        chatId,
-        requestModel
+        chatId
      );
    }
    toggleLoader(false, chatId);
@@ -1308,15 +1299,12 @@ async function promptGPT(systemPrompt, input) {
  }

  if (!response.ok) {
-    const request = activeRequests.get(chatId);
-    const requestModel = request?.model || null;
    chatStore.add(
      "assistant",
      `<span class='error'>Error: POST ${endpoint} ${response.status}</span>`,
      null,
      null,
-      chatId,
-      requestModel
+      chatId
    );
    toggleLoader(false, chatId);
    activeRequests.delete(chatId);
@@ -1336,15 +1324,12 @@ async function promptGPT(systemPrompt, input) {
      .getReader();

    if (!reader) {
-      const request = activeRequests.get(chatId);
-      const requestModel = request?.model || null;
      chatStore.add(
        "assistant",
        `<span class='error'>Error: Failed to decode MCP API response</span>`,
        null,
        null,
-        chatId,
-        requestModel
+        chatId
      );
      toggleLoader(false, chatId);
      activeRequests.delete(chatId);
@@ -1613,15 +1598,12 @@ async function promptGPT(systemPrompt, input) {
                  break;
                
                case "error":
-                  const request = activeRequests.get(chatId);
-                  const requestModel = request?.model || null;
                  chatStore.add(
                    "assistant",
                    `<span class='error'>MCP Error: ${eventData.message}</span>`,
                    null,
                    null,
-                    chatId,
-                    requestModel
+                    chatId
                  );
                  break;
              }
@@ -1642,11 +1624,9 @@ async function promptGPT(systemPrompt, input) {
          // Update or create assistant message with processed regular content
          const currentChat = chatStore.getChat(chatId);
          if (!currentChat) break; // Chat was deleted
-          const request = activeRequests.get(chatId);
-          const requestModel = request?.model || null;
          if (lastAssistantMessageIndex === -1) {
            if (processedRegular && processedRegular.trim()) {
-              chatStore.add("assistant", processedRegular, null, null, chatId, requestModel);
+              chatStore.add("assistant", processedRegular, null, null, chatId);
              lastAssistantMessageIndex = targetHistory.length - 1;
            }
          } else {
@@ -1726,9 +1706,7 @@ async function promptGPT(systemPrompt, input) {
            lastMessage.html = DOMPurify.sanitize(marked.parse(lastMessage.content));
          }
        } else if (processedRegular && processedRegular.trim()) {
-          const request = activeRequests.get(chatId);
-          const requestModel = request?.model || null;
-          chatStore.add("assistant", processedRegular, null, null, chatId, requestModel);
+          chatStore.add("assistant", processedRegular, null, null, chatId);
          lastAssistantMessageIndex = targetHistory.length - 1;
        }
      }
@@ -1776,9 +1754,7 @@ async function promptGPT(systemPrompt, input) {
              lastMessage.html = DOMPurify.sanitize(marked.parse(lastMessage.content));
            }
          } else {
-            const request = activeRequests.get(chatId);
-            const requestModel = request?.model || null;
-            chatStore.add("assistant", finalRegular, null, null, chatId, requestModel);
+            chatStore.add("assistant", finalRegular, null, null, chatId);
          }
        }
        
@@ -1836,15 +1812,12 @@ async function promptGPT(systemPrompt, input) {
      .getReader();

    if (!reader) {
-      const request = activeRequests.get(chatId);
-      const requestModel = request?.model || null;
      chatStore.add(
        "assistant",
        `<span class='error'>Error: Failed to decode API response</span>`,
        null,
        null,
-        chatId,
-        requestModel
+        chatId
      );
      toggleLoader(false, chatId);
      activeRequests.delete(chatId);
@@ -1875,11 +1848,9 @@ async function promptGPT(systemPrompt, input) {
    const addToChat = (token) => {
      const currentChat = chatStore.getChat(chatId);
      if (!currentChat) return; // Chat was deleted
-      // Get model from request for this chat
-      const request = activeRequests.get(chatId);
-      const requestModel = request?.model || null;
-      chatStore.add("assistant", token, null, null, chatId, requestModel);
+      chatStore.add("assistant", token, null, null, chatId);
      // Count tokens for rate calculation (per chat)
+      const request = activeRequests.get(chatId);
      if (request) {
        const tokenCount = Math.ceil(token.length / 4);
        request.tokensReceived += tokenCount;
@@ -2037,15 +2008,12 @@ async function promptGPT(systemPrompt, input) {
      if (error.name !== 'AbortError' || !currentAbortController) {
        const currentChat = chatStore.getChat(chatId);
        if (currentChat) {
-          const request = activeRequests.get(chatId);
-          const requestModel = request?.model || null;
          chatStore.add(
            "assistant",
            `<span class='error'>Error: Failed to process stream</span>`,
            null,
            null,
-            chatId,
-            requestModel
+            chatId
          );
        }
      }
--- a/core/http/static/image.js
+++ b/core/http/static/image.js
@@ -154,7 +154,7 @@ async function promptDallE() {
    if (json.data && json.data.length > 0) {
      json.data.forEach((item, index) => {
        const imageContainer = document.createElement("div");
-        imageContainer.className = "flex flex-col";
+        imageContainer.className = "mb-4 bg-[var(--color-bg-primary)]/50 border border-[#1E293B] rounded-lg p-2";

        // Create image element
        const img = document.createElement("img");
@@ -166,23 +166,23 @@ async function promptDallE() {
          return; // Skip invalid items
        }
        img.alt = prompt;
-        img.className = "w-full h-auto rounded-lg";
+        img.className = "w-full h-auto rounded-lg mb-2";
        imageContainer.appendChild(img);

-        // Create caption container (optional, can be collapsed or shown on hover)
+        // Create caption container
        const captionDiv = document.createElement("div");
-        captionDiv.className = "mt-2 p-2 bg-[var(--color-bg-secondary)] rounded-lg text-xs";
+        captionDiv.className = "mt-2 p-2 bg-[var(--color-bg-secondary)] rounded-lg";

        // Prompt caption
        const promptCaption = document.createElement("p");
-        promptCaption.className = "text-[var(--color-text-primary)] mb-1.5 break-words";
+        promptCaption.className = "text-xs text-[var(--color-text-primary)] mb-1.5";
        promptCaption.innerHTML = '<strong>Prompt:</strong> ' + escapeHtml(prompt);
        captionDiv.appendChild(promptCaption);

        // Negative prompt if provided
        if (negativePrompt) {
          const negativeCaption = document.createElement("p");
-          negativeCaption.className = "text-[var(--color-text-secondary)] mb-1.5 break-words";
+          negativeCaption.className = "text-xs text-[var(--color-text-secondary)] mb-1.5";
          negativeCaption.innerHTML = '<strong>Negative Prompt:</strong> ' + escapeHtml(negativePrompt);
          captionDiv.appendChild(negativeCaption);
        }
--- a/core/http/views/chat.html
+++ b/core/http/views/chat.html
@@ -276,31 +276,12 @@ SOFTWARE.
          }
        },
        
-        add(role, content, image, audio, targetChatId = null, model = null) {
+        add(role, content, image, audio, targetChatId = null) {
          // If targetChatId is provided, add to that chat, otherwise use active chat
          // This allows streaming to continue to the correct chat even if user switches
          const chat = targetChatId ? this.getChat(targetChatId) : this.activeChat();
          if (!chat) return;
          
-          // Determine model for this message:
-          // - If model is explicitly provided, use it (for assistant messages with specific model)
-          // - For user messages, use the current chat's model
-          // - For other messages (thinking, tool_call, etc.), inherit from previous message or use chat model
-          let messageModel = model;
-          if (!messageModel) {
-            if (role === "user") {
-              // User messages always use the current chat's model
-              messageModel = chat.model || "";
-            } else if (role === "assistant") {
-              // Assistant messages use the chat's model (should be set when request is made)
-              messageModel = chat.model || "";
-            } else {
-              // For thinking, tool_call, etc., try to inherit from last assistant message, or use chat model
-              const lastAssistant = chat.history.slice().reverse().find(m => m.role === "assistant");
-              messageModel = lastAssistant?.model || chat.model || "";
-            }
-          }
-          
          const N = chat.history.length - 1;
          // For thinking, reasoning, tool_call, and tool_result messages, always create a new message
          if (role === "thinking" || role === "reasoning" || role === "tool_call" || role === "tool_result") {
@@ -330,7 +311,7 @@ SOFTWARE.
            // Reasoning, tool_call, and tool_result are always collapsed by default
            const isMCPMode = chat.mcpMode || false;
            const shouldExpand = (role === "thinking" && !isMCPMode) || false;
-            chat.history.push({ role, content, html: c, image, audio, expanded: shouldExpand, model: messageModel });
+            chat.history.push({ role, content, html: c, image, audio, expanded: shouldExpand });
            
            // Auto-name chat from first user message
            if (role === "user" && chat.name === "New Chat" && content.trim()) {
@@ -351,10 +332,6 @@ SOFTWARE.
            if (audio && audio.length > 0) {
              chat.history[N].audio = [...(chat.history[N].audio || []), ...audio];
            }
-            // Preserve model if merging (don't overwrite)
-            if (!chat.history[N].model && messageModel) {
-              chat.history[N].model = messageModel;
-            }
          } else {
            let c = "";
            const lines = content.split("\n");
@@ -366,8 +343,7 @@ SOFTWARE.
              content, 
              html: c, 
              image: image || [], 
-              audio: audio || [],
-              model: messageModel
+              audio: audio || [] 
            });
            
            // Auto-name chat from first user message
@@ -1272,20 +1248,11 @@ SOFTWARE.
                </template>
                <template x-if="message.role != 'user' && message.role != 'thinking' && message.role != 'reasoning' && message.role != 'tool_call' && message.role != 'tool_result'">
                  <div class="flex items-center space-x-2">
-                    <!-- Model icon - from message history, fallback to active chat -->
-                    <template x-if="message.model && window.__galleryConfigs && window.__galleryConfigs[message.model] && window.__galleryConfigs[message.model].Icon">
-                      <img :src="window.__galleryConfigs[message.model].Icon" class="rounded-lg mt-2 max-w-8 max-h-8 border border-[var(--color-primary-border)]/20">
-                    </template>
-                    <!-- Fallback: use active chat model if message doesn't have one -->
-                    <template x-if="!message.model && $store.chat.activeChat() && $store.chat.activeChat().model && window.__galleryConfigs && window.__galleryConfigs[$store.chat.activeChat().model] && window.__galleryConfigs[$store.chat.activeChat().model].Icon">
-                      <img :src="window.__galleryConfigs[$store.chat.activeChat().model].Icon" class="rounded-lg mt-2 max-w-8 max-h-8 border border-[var(--color-primary-border)]/20">
-                    </template>
-                    <!-- Final fallback: initial model from server -->
-                    <template x-if="!message.model && (!$store.chat.activeChat() || !$store.chat.activeChat().model) && window.__galleryConfigs && window.__galleryConfigs['{{$model}}'] && window.__galleryConfigs['{{$model}}'].Icon">
-                      <img :src="window.__galleryConfigs['{{$model}}'].Icon" class="rounded-lg mt-2 max-w-8 max-h-8 border border-[var(--color-primary-border)]/20">
-                    </template>
+                    {{ if $galleryConfig }}
+                    {{ if $galleryConfig.Icon }}<img src="{{$galleryConfig.Icon}}" class="rounded-lg mt-2 max-w-8 max-h-8 border border-[var(--color-primary-border)]/20">{{end}}
+                    {{ end }}
                    <div class="flex flex-col flex-1">
-                      <span class="text-xs font-semibold text-[var(--color-text-secondary)] mb-1" x-text="message.model || $store.chat.activeChat()?.model || '{{if .Model}}{{.Model}}{{else}}Assistant{{end}}'"></span>
+                      <span class="text-xs font-semibold text-[var(--color-text-secondary)] mb-1">{{if .Model}}{{.Model}}{{else}}Assistant{{end}}</span>
                      <div class="flex-1 text-[var(--color-text-primary)] flex items-center space-x-2 min-w-0">
                        <div class="p-3 rounded-lg bg-[var(--color-bg-secondary)] border border-[var(--color-accent-border)]/20 shadow-lg max-w-full overflow-x-auto overflow-wrap-anywhere" x-html="message.html"></div>
                        <button @click="copyToClipboard(message.html)" title="Copy to clipboard" class="text-[var(--color-text-secondary)] hover:text-[var(--color-primary)] transition-colors p-1 flex-shrink-0">
--- a/core/http/views/partials/navbar.html
+++ b/core/http/views/partials/navbar.html
@@ -40,7 +40,7 @@
                <a href="traces/" class="text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] px-2 py-2 rounded-lg transition duration-300 ease-in-out hover:bg-[var(--color-bg-secondary)] flex items-center group text-sm">
                    <i class="fas fa-chart-line text-[var(--color-primary)] mr-1.5 text-sm group-hover:scale-110 transition-transform"></i>Traces
                </a>
-                <a href="swagger/index.html" class="text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] px-2 py-2 rounded-lg transition duration-300 ease-in-out hover:bg-[var(--color-bg-secondary)] flex items-center group text-sm">
+                <a href="swagger/" class="text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] px-2 py-2 rounded-lg transition duration-300 ease-in-out hover:bg-[var(--color-bg-secondary)] flex items-center group text-sm">
                    <i class="fas fa-code text-[var(--color-primary)] mr-1.5 text-sm group-hover:scale-110 transition-transform"></i>API
                </a>
                
@@ -100,7 +100,7 @@
                <a href="traces/" class="block text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] hover:bg-[var(--color-bg-secondary)] px-3 py-2 rounded-lg transition duration-300 ease-in-out flex items-center text-sm">
                    <i class="fas fa-chart-line text-[var(--color-primary)] mr-3 w-5 text-center text-sm"></i>Traces
                </a>
-                <a href="swagger/index.html" class="block text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] hover:bg-[var(--color-bg-secondary)] px-3 py-2 rounded-lg transition duration-300 ease-in-out flex items-center text-sm">
+                <a href="swagger/" class="block text-[var(--color-text-secondary)] hover:text-[var(--color-text-primary)] hover:bg-[var(--color-bg-secondary)] px-3 py-2 rounded-lg transition duration-300 ease-in-out flex items-center text-sm">
                    <i class="fas fa-code text-[var(--color-primary)] mr-3 w-5 text-center text-sm"></i>API
                </a>
                
--- a/core/http/views/text2image.html
+++ b/core/http/views/text2image.html
@@ -215,23 +215,26 @@

            <!-- Right Column: Image Preview -->
            <div class="flex-grow lg:w-3/4 flex flex-col min-h-0">
-                <div class="relative flex-1 min-h-0 overflow-y-auto">
-                    <!-- Loading Animation -->
-                    <div id="loader" class="hidden absolute inset-0 flex items-center justify-center bg-[var(--color-bg-primary)]/80 rounded-xl z-10">
-                        <div class="text-center">
-                            <svg class="animate-spin h-10 w-10 text-[var(--color-primary)] mx-auto mb-3" xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24">
-                                <circle class="opacity-25" cx="12" cy="12" r="10" stroke="currentColor" stroke-width="4"></circle>
-                                <path class="opacity-75" fill="currentColor" d="M4 12a8 8 0 018-8V0C5.373 0 0 5.373 0 12h4zm2 5.291A7.962 7.962 0 014 12H0c0 3.042 1.135 5.824 3 7.938l3-2.647z"></path>
-                            </svg>
-                            <p class="text-xs text-[var(--color-text-secondary)]">Generating image...</p>
+                <div class="card p-3 flex flex-col flex-1 min-h-0">
+                    <h3 class="text-sm font-semibold text-[var(--color-text-primary)] mb-3 flex-shrink-0">Generated Images</h3>
+                    <div class="relative flex-1 min-h-0 overflow-y-auto">
+                        <!-- Loading Animation -->
+                        <div id="loader" class="hidden absolute inset-0 flex items-center justify-center bg-[var(--color-bg-primary)]/80 rounded-xl z-10">
+                            <div class="text-center">
+                                <svg class="animate-spin h-10 w-10 text-[var(--color-primary)] mx-auto mb-3" xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24">
+                                    <circle class="opacity-25" cx="12" cy="12" r="10" stroke="currentColor" stroke-width="4"></circle>
+                                    <path class="opacity-75" fill="currentColor" d="M4 12a8 8 0 018-8V0C5.373 0 0 5.373 0 12h4zm2 5.291A7.962 7.962 0 014 12H0c0 3.042 1.135 5.824 3 7.938l3-2.647z"></path>
+                                </svg>
+                                <p class="text-xs text-[var(--color-text-secondary)]">Generating image...</p>
+                            </div>
                        </div>
+                        <!-- Placeholder when no images -->
+                        <div id="result-placeholder" class="bg-[var(--color-bg-primary)]/50 border border-[#1E293B] rounded-xl p-6 min-h-[400px] flex items-center justify-center flex-shrink-0">
+                            <p class="text-xs text-[var(--color-text-secondary)] italic text-center">Your generated images will appear here</p>
+                        </div>
+                        <!-- Results container -->
+                        <div id="result" class="space-y-4 pb-4"></div>
                    </div>
-                    <!-- Placeholder when no images -->
-                    <div id="result-placeholder" class="min-h-[400px] flex items-center justify-center flex-shrink-0">
-                        <p class="text-xs text-[var(--color-text-secondary)] italic text-center">Your generated images will appear here</p>
-                    </div>
-                    <!-- Results container -->
-                    <div id="result" class="grid grid-cols-1 sm:grid-cols-2 gap-4 pb-4"></div>
                </div>
            </div>
        </div>
--- a/core/schema/gallery-model.schema.json
+++ b/core/schema/gallery-model.schema.json
@@ -1,79 +0,0 @@
-{
-  "$schema": "https://json-schema.org/draft/2020-12/schema",
-  "$id": "https://raw.githubusercontent.com/mudler/LocalAI/main/schemas/gallery.model.schema.json",
-  "title": "LocalAI Gallery Model Spec",
-  "description": "Schema for LocalAI gallery model YAML files",
-  "type": "object",
-
-  "properties": {
-    "name": {
-      "type": "string",
-      "description": "Model name"
-    },
-    "description": {
-      "type": "string",
-      "description": "Human-readable description of the model"
-    },
-    "icon": {
-      "type": "string",
-      "description": "Optional icon reference or URL"
-    },
-    "license": {
-      "type": "string",
-      "description": "Model license identifier or text"
-    },
-    "urls": {
-      "type": "array",
-      "description": "URLs pointing to remote model configuration",
-      "items": {
-        "type": "string",
-        "format": "uri"
-      }
-    },
-    "config_file": {
-      "type": "string",
-      "description": "Inline YAML configuration that will be written to the model config file"
-    },
-    "files": {
-      "type": "array",
-      "description": "Files to download and install for this model",
-      "items": {
-        "type": "object",
-        "required": ["filename", "uri"],
-        "properties": {
-          "filename": {
-            "type": "string"
-          },
-          "sha256": {
-            "type": "string",
-            "description": "Optional SHA256 checksum for file verification"
-          },
-          "uri": {
-            "type": "string",
-            "format": "uri"
-          }
-        },
-        "additionalProperties": false
-      }
-    },
-    "prompt_templates": {
-      "type": "array",
-      "description": "Prompt templates written as .tmpl files",
-      "items": {
-        "type": "object",
-        "required": ["name", "content"],
-        "properties": {
-          "name": {
-            "type": "string"
-          },
-          "content": {
-            "type": "string"
-          }
-        },
-        "additionalProperties": false
-      }
-    }
-  },
-
-  "additionalProperties": false
-}
--- a/docker-compose.yaml
+++ b/docker-compose.yaml
@@ -11,7 +11,7 @@ services:
      dockerfile: Dockerfile
      args:
      - IMAGE_TYPE=core
-      - BASE_IMAGE=ubuntu:24.04
+      - BASE_IMAGE=ubuntu:22.04
    ports:
      - 8080:8080
    env_file:
--- a/docs/content/getting-started/container-images.md
+++ b/docs/content/getting-started/container-images.md
@@ -50,6 +50,16 @@ Standard container images do not have pre-installed models. Use these if you wan

 {{% /tab %}}

+{{% tab title="GPU Images CUDA 11" %}}
+
+| Description | Quay | Docker Hub                                                  |
+| --- | --- |-------------------------------------------------------------|
+| Latest images from the branch (development) | `quay.io/go-skynet/local-ai:master-gpu-nvidia-cuda-11` | `localai/localai:master-gpu-nvidia-cuda-11`                      |
+| Latest tag | `quay.io/go-skynet/local-ai:latest-gpu-nvidia-cuda-11` | `localai/localai:latest-gpu-nvidia-cuda-11`                      |
+| Versioned image | `quay.io/go-skynet/local-ai:{{< version >}}-gpu-nvidia-cuda-11` | `localai/localai:{{< version >}}-gpu-nvidia-cuda-11`             |
+
+{{% /tab %}}
+
 {{% tab title="GPU Images CUDA 12" %}}

 | Description | Quay | Docker Hub                                                  |
@@ -159,9 +169,11 @@ services:
    image: localai/localai:latest-aio-cpu
    # For a specific version:
    # image: localai/localai:{{< version >}}-aio-cpu
-    # For Nvidia GPUs decomment one of the following (cuda12 or cuda13):
+    # For Nvidia GPUs decomment one of the following (cuda11, cuda12, or cuda13):
+    # image: localai/localai:{{< version >}}-aio-gpu-nvidia-cuda-11
    # image: localai/localai:{{< version >}}-aio-gpu-nvidia-cuda-12
    # image: localai/localai:{{< version >}}-aio-gpu-nvidia-cuda-13
+    # image: localai/localai:latest-aio-gpu-nvidia-cuda-11
    # image: localai/localai:latest-aio-gpu-nvidia-cuda-12
    # image: localai/localai:latest-aio-gpu-nvidia-cuda-13
    healthcheck:
@@ -213,6 +225,7 @@ docker run -p 8080:8080 --name local-ai -ti -v localai-models:/models localai/lo
 | --- | --- |-----------------------------------------------|
 | Latest images for CPU | `quay.io/go-skynet/local-ai:latest-aio-cpu` | `localai/localai:latest-aio-cpu`                      |
 | Versioned image (e.g. for CPU) | `quay.io/go-skynet/local-ai:{{< version >}}-aio-cpu` | `localai/localai:{{< version >}}-aio-cpu`             |
+| Latest images for Nvidia GPU (CUDA11) | `quay.io/go-skynet/local-ai:latest-aio-gpu-nvidia-cuda-11` | `localai/localai:latest-aio-gpu-nvidia-cuda-11`                      |
 | Latest images for Nvidia GPU (CUDA12) | `quay.io/go-skynet/local-ai:latest-aio-gpu-nvidia-cuda-12` | `localai/localai:latest-aio-gpu-nvidia-cuda-12`                      |
 | Latest images for Nvidia GPU (CUDA13) | `quay.io/go-skynet/local-ai:latest-aio-gpu-nvidia-cuda-13` | `localai/localai:latest-aio-gpu-nvidia-cuda-13`                      |
 | Latest images for AMD GPU | `quay.io/go-skynet/local-ai:latest-aio-gpu-hipblas` | `localai/localai:latest-aio-gpu-hipblas`                      |
--- a/docs/content/installation/docker.md
+++ b/docs/content/installation/docker.md
@@ -68,6 +68,11 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gp
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-12
 ```

+**NVIDIA CUDA 11:**
+```bash
+docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-gpu-nvidia-cuda-11
+```
+
 **AMD GPU (ROCm):**
 ```bash
 docker run -ti --name local-ai -p 8080:8080 --device=/dev/kfd --device=/dev/dri --group-add=video localai/localai:latest-gpu-hipblas
@@ -117,6 +122,11 @@ docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-ai
 docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-12
 ```

+**NVIDIA CUDA 11:**
+```bash
+docker run -ti --name local-ai -p 8080:8080 --gpus all localai/localai:latest-aio-gpu-nvidia-cuda-11
+```
+
 **AMD GPU (ROCm):**
 ```bash
 docker run -ti --name local-ai -p 8080:8080 --device=/dev/kfd --device=/dev/dri --group-add=video localai/localai:latest-aio-gpu-hipblas
--- a/docs/content/integrations.md
+++ b/docs/content/integrations.md
@@ -32,311 +32,5 @@ The list below is a list of software that integrates with LocalAI.
 - [GPTLocalhost (Word Add-in)](https://gptlocalhost.com/demo#LocalAI) - run LocalAI in Microsoft Word locally
 - use LocalAI from Nextcloud with the [integration plugin](https://apps.nextcloud.com/apps/integration_openai) and [AI assistant](https://apps.nextcloud.com/apps/assistant)
 - [Langchain](https://docs.langchain.com/oss/python/integrations/providers/localai) integration package [pypi](https://pypi.org/project/langchain-localai/)
- [VoxInput](https://github.com/richiejp/VoxInput) - Use voice to control your desktop

 Feel free to open up a Pull request (by clicking at the "Edit page" below) to get a page for your project made or if you see a error on one of the pages!
-
-## Configuration Guides
-
-This section provides step-by-step instructions for configuring specific software to work with LocalAI.
-
-### OpenCode
-
-[OpenCode](https://opencode.ai) is an AI-powered code editor that can be configured to use LocalAI as its backend provider.
-
-#### Prerequisites
-
- LocalAI must be running and accessible (either locally or on a network)
- You need to know your LocalAI server's IP address/hostname and port (default is `8080`)
-
-#### Configuration Steps
-
-1. **Edit the OpenCode configuration file**
-
-   Open the OpenCode configuration file located at `~/.config/opencode/opencode.json` in your editor.
-
-2. **Add LocalAI provider configuration**
-
-   Add the following configuration to your `opencode.json` file, replacing the values with your own:
-
-   ```json
-   {
-     "$schema": "https://opencode.ai/config.json",
-     "provider": {
-       "LocalAI": {
-         "npm": "@ai-sdk/openai-compatible",
-         "name": "LocalAI (local)",
-         "options": {
-           "baseURL": "http://127.0.0.1:8080/v1"
-         },
-         "models": {
-           "Qwen3-Coder-30B-A3B-Instruct-i1-GGUF": {
-             "name": "Qwen3-Coder-30B-A3B-Instruct-i1-GGUF",
-             "limit": {
-               "context": 38000,
-               "output": 65536
-             }
-           },
-           "qwen_qwen3-30b-a3b-instruct-2507": {
-             "name": "qwen_qwen3-30b-a3b-instruct-2507",
-             "limit": {
-               "context": 38000,
-               "output": 65536
-             }
-           }
-         }
-       }
-     }
-   }
-   ```
-
-3. **Customize the configuration**
-
-   - **baseURL**: Replace `http://127.0.0.1:8080/v1` with your LocalAI server's address and port.
-   - **name**: Change "LocalAI (local)" to a descriptive name for your setup.
-   - **models**: Replace the model names with the actual model names available in your LocalAI instance. You can find available models by checking your LocalAI models directory or using the LocalAI API.
-   - **limit**: Adjust the `context` and `output` token limits based on your model's capabilities and available resources.
-
-4. **Verify your models**
-
-   Ensure that the model names in the configuration match exactly with the model names configured in your LocalAI instance. You can verify available models by checking your LocalAI configuration or using the `/v1/models` endpoint.
-
-5. **Restart OpenCode**
-
-   After saving the configuration file, restart OpenCode for the changes to take effect.
-
-
-### Charm Crush
-
-You can ask [Charm Crush](https://charm.land/crush) to generate your config by giving it this documentation's URL and your LocalAI instance URL. The configuration will look something like the following and goes in `~/.config/crush/crush.json`:
-```json
-{
-  "$schema": "https://charm.land/crush.json",
-  "providers": {
-    "localai": {
-      "name": "LocalAI",
-      "base_url": "http://localai.lan:8081/v1",
-      "type": "openai-compat",
-      "models": [
-        {
-          "id": "qwen3-coder-480b-a35b-instruct",
-          "name": "Qwen 3 Coder 480b",
-          "context_window": 256000
-        },
-        {
-          "id": "qwen3-30b-a3b",
-          "name": "Qwen 3 30b a3b",
-          "context_window": 32000
-        }
-      ]
-    }
-  }
-}
-```
-
-A list of models can be fetched with `https://<server_address>/v1/models` by crush itself and appropriate models added to the provider list. Crush does not appear to be optimized for smaller models.
-
-### GitHub Actions
-
-You can use LocalAI in GitHub Actions workflows to perform AI-powered tasks like code review, diff summarization, or automated analysis. The [LocalAI GitHub Action](https://github.com/mudler/localai-github-action) makes it easy to spin up a LocalAI instance in your CI/CD pipeline.
-
-#### Prerequisites
-
- A GitHub repository with Actions enabled
- A model name from [models.localai.io](https://models.localai.io) or a Hugging Face model reference
-
-#### Example Workflow
-
-This example workflow demonstrates how to use LocalAI to summarize pull request diffs and send notifications:
-
-1. **Create a workflow file**
-
-   Create a new file in your repository at `.github/workflows/localai.yml`:
-
-```yaml
-name: Use LocalAI in GHA
-on:
-  pull_request:
-     types:
-       - closed
-
-jobs:
-  notify-discord:
-    if: ${{ (github.event.pull_request.merged == true) && (contains(github.event.pull_request.labels.*.name, 'area/ai-model')) }}
-    env:
-        MODEL_NAME: qwen_qwen3-4b-instruct-2507
-    runs-on: ubuntu-latest
-    steps:
-    - uses: actions/checkout@v4
-      with:
-        fetch-depth: 0 # needed to checkout all branches for this Action to work
-    # Starts the LocalAI container
-    - id: foo
-      uses: mudler/localai-github-action@v1.1
-      with:
-        model: 'qwen_qwen3-4b-instruct-2507' # Any from models.localai.io, or from huggingface.com with: "huggingface://<repository>/file"
-    # Check the PR diff using the current branch and the base branch of the PR
-    - uses: GrantBirki/git-diff-action@v2.7.0
-      id: git-diff-action
-      with:
-            json_diff_file_output: diff.json
-            raw_diff_file_output: diff.txt
-            file_output_only: "true"
-    # Ask to explain the diff to LocalAI
-    - name: Summarize
-      env:
-        DIFF: ${{ steps.git-diff-action.outputs.raw-diff-path }}
-      id: summarize
-      run: |
-            input="$(cat $DIFF)"
-
-            # Define the LocalAI API endpoint
-            API_URL="http://localhost:8080/chat/completions"
-
-            # Create a JSON payload using jq to handle special characters
-            json_payload=$(jq -n --arg input "$input" '{
-            model: "'$MODEL_NAME'",
-            messages: [
-                {
-                role: "system",
-                content: "Write a message summarizing the change diffs"
-                },
-                {
-                role: "user",
-                content: $input
-                }
-            ]
-            }')
-
-            # Send the request to LocalAI
-            response=$(curl -s -X POST $API_URL \
-            -H "Content-Type: application/json" \
-            -d "$json_payload")
-
-            # Extract the summary from the response
-            summary="$(echo $response | jq -r '.choices[0].message.content')"
-
-            # Print the summary
-            echo "Summary:"
-            echo "$summary"
-            echo "payload sent"
-            echo "$json_payload"
-            {
-                echo 'message<<EOF'
-                echo "$summary"
-                echo EOF
-              } >> "$GITHUB_OUTPUT"
-    # Send the summary somewhere (e.g. Discord)
-    - name: Discord notification
-      env:
-        DISCORD_WEBHOOK: ${{ secrets.DISCORD_WEBHOOK_URL }}
-        DISCORD_USERNAME: "discord-bot"
-        DISCORD_AVATAR: ""
-      uses: Ilshidur/action-discord@master
-      with:
-        args: ${{ steps.summarize.outputs.message }}
-```
-
-#### Configuration Options
-
- **Model selection**: Replace `qwen_qwen3-4b-instruct-2507` with any model from [models.localai.io](https://models.localai.io). You can also use Hugging Face models by using the full huggingface model url`.
- **Trigger conditions**: Customize the `if` condition to control when the workflow runs. The example only runs when a PR is merged and has a specific label.
- **API endpoint**: The LocalAI container runs on `http://localhost:8080` by default. The action exposes the service on the standard port.
- **Custom prompts**: Modify the system message in the JSON payload to change what LocalAI is asked to do with the diff.
-
-#### Use Cases
-
- **Code review automation**: Automatically review code changes and provide feedback
- **Diff summarization**: Generate human-readable summaries of code changes
- **Documentation generation**: Create documentation from code changes
- **Security scanning**: Analyze code for potential security issues
- **Test generation**: Generate test cases based on code changes
-
-#### Additional Resources
-
- [LocalAI GitHub Action repository](https://github.com/mudler/localai-github-action)
- [Available models](https://models.localai.io)
- [LocalAI API documentation](/reference/)
-
-### Realtime Voice Assistant
-
-LocalAI supports realtime voice interactions , enabling voice assistant applications with real-time speech-to-speech communication. A complete example implementation is available in the [LocalAI-examples repository](https://github.com/mudler/LocalAI-examples/tree/main/realtime).
-
-#### Overview
-
-The realtime voice assistant example demonstrates how to build a voice assistant that:
- Captures audio input from the user in real-time
- Transcribes speech to text using LocalAI's transcription capabilities
- Processes the text with a language model
- Generates audio responses using text-to-speech
- Streams audio back to the user in real-time
-
-#### Prerequisites
-
- A transcription model (e.g., Whisper) configured in LocalAI
- A text-to-speech model configured in LocalAI
- A language model for generating responses
-
-#### Getting Started
-
-1. **Clone the example repository**
-
-   ```bash
-   git clone https://github.com/mudler/LocalAI-examples.git
-   cd LocalAI-examples/realtime
-   ```
-
-2. **Start LocalAI with Docker Compose**
-
-   ```bash
-   docker compose up -d
-   ```
-
-   The first time you start docker compose, it will take a while to download the available models. You can follow the model downloads in real-time:
-
-   ```bash
-   docker logs -f realtime-localai-1
-   ```
-
-3. **Install host dependencies**
-
-   Install the required host dependencies (sudo is required):
-
-   ```bash
-   sudo bash setup.sh
-   ```
-
-4. **Run the voice assistant**
-
-   Start the voice assistant application:
-
-   ```bash
-   bash run.sh
-   ```
-
-#### Configuration Notes
-
- **CPU vs GPU**: The example is optimized for CPU usage. However, you can run LocalAI with a GPU for better performance and to use bigger/better models.
- **Python client**: The Python part downloads PyTorch for CPU, but this is fine as computation is offloaded to LocalAI. The Python client only runs Silero VAD (Voice Activity Detection), which is fast, and handles audio recording.
- **Thin client architecture**: The Python client is designed to run on thin clients such as Raspberry PIs, while LocalAI handles the heavier computational workload on a more powerful machine.
-
-#### Key Features
-
- **Real-time processing**: Low-latency audio streaming for natural conversations
- **Voice Activity Detection (VAD)**: Automatic detection of when the user is speaking
- **Turn-taking**: Handles conversation flow with proper turn detection
- **OpenAI-compatible API**: Uses LocalAI's OpenAI-compatible realtime API endpoints
-
-#### Use Cases
-
- **Voice assistants**: Build custom voice assistants for home automation or productivity
- **Accessibility tools**: Create voice interfaces for accessibility applications
- **Interactive applications**: Add voice interaction to games, educational software, or entertainment apps
- **Customer service**: Implement voice-based customer support systems
-
-#### Additional Resources
-
- [Realtime Voice Assistant Example](https://github.com/mudler/LocalAI-examples/tree/main/realtime)
- [LocalAI Realtime API documentation](/features/)
- [Audio features documentation](/features/text-to-audio/)
- [Transcription features documentation](/features/audio-to-text/)
--- a/docs/content/reference/compatibility-table.md
+++ b/docs/content/reference/compatibility-table.md
@@ -18,9 +18,9 @@ LocalAI will attempt to automatically load models which are not explicitly confi

 | Backend and Bindings                                                             | Compatible models     | Completion/Chat endpoint | Capability | Embeddings support                | Token stream support | Acceleration |
 |----------------------------------------------------------------------------------|-----------------------|--------------------------|---------------------------|-----------------------------------|----------------------|--------------|
-| [llama.cpp]({{%relref "features/text-generation#llama.cpp" %}})        | LLama, Mamba, RWKV, Falcon, Starcoder, GPT-2, [and many others](https://github.com/ggerganov/llama.cpp?tab=readme-ov-file#description) | yes                      | GPT and Functions                        | yes | yes                  | CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
+| [llama.cpp]({{%relref "features/text-generation#llama.cpp" %}})        | LLama, Mamba, RWKV, Falcon, Starcoder, GPT-2, [and many others](https://github.com/ggerganov/llama.cpp?tab=readme-ov-file#description) | yes                      | GPT and Functions                        | yes | yes                  | CUDA 11/12/13, ROCm, Intel SYCL, Vulkan, Metal, CPU |
 | [vLLM](https://github.com/vllm-project/vllm)        | Various GPTs and quantization formats | yes                      | GPT             | no | no                  | CUDA 12/13, ROCm, Intel |
-| [transformers](https://github.com/huggingface/transformers) | Various GPTs and quantization formats  | yes                      | GPT, embeddings, Audio generation            | yes | yes*                  | CUDA 12/13, ROCm, Intel, CPU |
+| [transformers](https://github.com/huggingface/transformers) | Various GPTs and quantization formats  | yes                      | GPT, embeddings, Audio generation            | yes | yes*                  | CUDA 11/12/13, ROCm, Intel, CPU |
 | [exllama2](https://github.com/turboderp-org/exllamav2)  | GPTQ                   | yes                       | GPT only                  | no                               | no                   | CUDA 12/13 |
 | [MLX](https://github.com/ml-explore/mlx-lm)        | Various LLMs               | yes                       | GPT                        | no                                | no                   | Metal (Apple Silicon) |
 | [MLX-VLM](https://github.com/Blaizzy/mlx-vlm)        | Vision-Language Models               | yes                       | Multimodal GPT                        | no                                | no                   | Metal (Apple Silicon) |
@@ -37,7 +37,7 @@ LocalAI will attempt to automatically load models which are not explicitly confi
 | [bark-cpp](https://github.com/PABannier/bark.cpp)        | bark               | no                       | Audio-Only                 | no                                | no                   | CUDA, Metal, CPU |
 | [coqui](https://github.com/idiap/coqui-ai-TTS) | Coqui TTS    | no                       | Audio generation and Voice cloning    | no                               | no                   | CUDA 12/13, ROCm, Intel, CPU |
 | [kokoro](https://github.com/hexgrad/kokoro) | Kokoro TTS    | no                       | Text-to-speech    | no                               | no                   | CUDA 12/13, ROCm, Intel, CPU |
-| [chatterbox](https://github.com/resemble-ai/chatterbox) | Chatterbox TTS    | no                       | Text-to-speech    | no                               | no                   | CUDA 12/13, CPU |
+| [chatterbox](https://github.com/resemble-ai/chatterbox) | Chatterbox TTS    | no                       | Text-to-speech    | no                               | no                   | CUDA 11/12/13, CPU |
 | [kitten-tts](https://github.com/KittenML/KittenTTS) | Kitten TTS    | no                       | Text-to-speech    | no                               | no                   | CPU |
 | [silero-vad](https://github.com/snakers4/silero-vad) with [Golang bindings](https://github.com/streamer45/silero-vad-go) | Silero VAD    | no                       | Voice Activity Detection    | no                               | no                   | CPU |
 | [neutts](https://github.com/neuphonic/neuttsair) | NeuTTSAir    | no                       | Text-to-speech with voice cloning    | no                               | no                   | CUDA 12/13, ROCm, CPU |
@@ -49,7 +49,7 @@ LocalAI will attempt to automatically load models which are not explicitly confi
 | Backend and Bindings                                                             | Compatible models     | Completion/Chat endpoint | Capability | Embeddings support                | Token stream support | Acceleration |
 |----------------------------------------------------------------------------------|-----------------------|--------------------------|---------------------------|-----------------------------------|----------------------|--------------|
 | [stablediffusion.cpp](https://github.com/leejet/stable-diffusion.cpp)         | stablediffusion-1, stablediffusion-2, stablediffusion-3, flux, PhotoMaker               | no                       | Image                 | no                                | no                   | CUDA 12/13, Intel SYCL, Vulkan, CPU |
-| [diffusers](https://github.com/huggingface/diffusers)  | SD, various diffusion models,...                   | no                       | Image/Video generation    | no                               | no                   | CUDA 12/13, ROCm, Intel, Metal, CPU |
+| [diffusers](https://github.com/huggingface/diffusers)  | SD, various diffusion models,...                   | no                       | Image/Video generation    | no                               | no                   | CUDA 11/12/13, ROCm, Intel, Metal, CPU |
 | [transformers-musicgen](https://github.com/huggingface/transformers)  | MusicGen                    | no                       | Audio generation                | no                               | no                   | CUDA, CPU |

 ## Specialized AI Tasks
@@ -57,14 +57,14 @@ LocalAI will attempt to automatically load models which are not explicitly confi
 | Backend and Bindings                                                             | Compatible models     | Completion/Chat endpoint | Capability | Embeddings support                | Token stream support | Acceleration |
 |----------------------------------------------------------------------------------|-----------------------|--------------------------|---------------------------|-----------------------------------|----------------------|--------------|
 | [rfdetr](https://github.com/roboflow/rf-detr) | RF-DETR    | no                       | Object Detection    | no                               | no                   | CUDA 12/13, Intel, CPU |
-| [rerankers](https://github.com/AnswerDotAI/rerankers) | Reranking API    | no                       | Reranking   | no                               | no                   | CUDA 12/13, ROCm, Intel, CPU |
+| [rerankers](https://github.com/AnswerDotAI/rerankers) | Reranking API    | no                       | Reranking   | no                               | no                   | CUDA 11/12/13, ROCm, Intel, CPU |
 | [local-store](https://github.com/mudler/LocalAI) | Vector database    | no                       | Vector storage   | yes                               | no                   | CPU |
 | [huggingface](https://huggingface.co/docs/hub/en/api) | HuggingFace API models    | yes                       | Various AI tasks   | yes                               | yes                   | API-based |

 ## Acceleration Support Summary

 ### GPU Acceleration
- **NVIDIA CUDA**: CUDA 12.0, CUDA 13.0 support across most backends
+- **NVIDIA CUDA**: CUDA 11.7, CUDA 12.0, CUDA 13.0 support across most backends
 - **AMD ROCm**: HIP-based acceleration for AMD GPUs
 - **Intel oneAPI**: SYCL-based acceleration for Intel GPUs (F16/F32 precision)
 - **Vulkan**: Cross-platform GPU acceleration
--- a/gallery/index.yaml
+++ b/gallery/index.yaml
@@ -1,104 +1,4 @@
 ---
- name: "liquidai.lfm2-2.6b-transcript"
-  url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
-  urls:
-    - https://huggingface.co/DevQuasar/LiquidAI.LFM2-2.6B-Transcript-GGUF
-  description: |
-    This is a large language model (2.6B parameters) designed for text-generation tasks. It is a quantized version of the original model `LiquidAI/LFM2-2.6B-Transcript`, optimized for efficiency while retaining strong performance. The model is built on the foundation of the base model, with additional optimizations for deployment and use cases like transcription or language modeling. It is trained on large-scale text data and supports multiple languages.
-  overrides:
-    parameters:
-      model: llama-cpp/models/LiquidAI.LFM2-2.6B-Transcript.Q4_K_M.gguf
-    name: LiquidAI.LFM2-2.6B-Transcript-GGUF
-    backend: llama-cpp
-    template:
-      use_tokenizer_template: true
-    known_usecases:
-      - chat
-    function:
-      grammar:
-        disable: true
-    description: Imported from https://huggingface.co/DevQuasar/LiquidAI.LFM2-2.6B-Transcript-GGUF
-    options:
-      - use_jinja:true
-  files:
-    - filename: llama-cpp/models/LiquidAI.LFM2-2.6B-Transcript.Q4_K_M.gguf
-      sha256: 301a8467531781909dc7a6263318103a3d8673a375afc4641e358d4174bd15d4
-      uri: https://huggingface.co/DevQuasar/LiquidAI.LFM2-2.6B-Transcript-GGUF/resolve/main/LiquidAI.LFM2-2.6B-Transcript.Q4_K_M.gguf
- name: "lfm2.5-1.2b-nova-function-calling"
-  url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
-  urls:
-    - https://huggingface.co/NovachronoAI/LFM2.5-1.2B-Nova-Function-Calling-GGUF
-  description: |
-    The **LFM2.5-1.2B-Nova-Function-Calling-GGUF** is a quantized version of the original model, optimized for efficiency with **Unsloth**. It supports text and multimodal tasks, using different quantization levels (e.g., Q2_K, Q3_K, Q4_K, etc.) to balance performance and memory usage. The model is designed for function calling and is faster than the original version, making it suitable for tasks like code generation, reasoning, and multi-modal input processing.
-  overrides:
-    parameters:
-      model: llama-cpp/models/LFM2.5-1.2B-Nova-Function-Calling.Q4_K_M.gguf
-    name: LFM2.5-1.2B-Nova-Function-Calling-GGUF
-    backend: llama-cpp
-    template:
-      use_tokenizer_template: true
-    known_usecases:
-      - chat
-    function:
-      grammar:
-        disable: true
-    description: Imported from https://huggingface.co/NovachronoAI/LFM2.5-1.2B-Nova-Function-Calling-GGUF
-    options:
-      - use_jinja:true
-  files:
-    - filename: llama-cpp/models/LFM2.5-1.2B-Nova-Function-Calling.Q4_K_M.gguf
-      sha256: 5d039ad4195447cf4b6dbee8f7fe11f985c01d671a18153084c869077e431fbf
-      uri: https://huggingface.co/NovachronoAI/LFM2.5-1.2B-Nova-Function-Calling-GGUF/resolve/main/LFM2.5-1.2B-Nova-Function-Calling.Q4_K_M.gguf
- name: "mistral-nemo-instruct-2407-12b-thinking-m-claude-opus-high-reasoning-i1"
-  url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
-  urls:
-    - https://huggingface.co/mradermacher/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning-i1-GGUF
-  description: |
-    The model described in this repository is the **Mistral-Nemo-Instruct-2407-12B** (12 billion parameters), a large language model optimized for instruction tuning and high-level reasoning tasks. It is a **quantized version** of the original model, compressed for efficiency while retaining key capabilities. The model is designed to generate human-like text, perform complex reasoning, and support multi-modal tasks, making it suitable for applications requiring strong language understanding and output.
-  overrides:
-    parameters:
-      model: llama-cpp/models/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning.i1-Q4_K_M.gguf
-    name: Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning-i1-GGUF
-    backend: llama-cpp
-    template:
-      use_tokenizer_template: true
-    known_usecases:
-      - chat
-    function:
-      grammar:
-        disable: true
-    description: Imported from https://huggingface.co/mradermacher/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning-i1-GGUF
-    options:
-      - use_jinja:true
-  files:
-    - filename: llama-cpp/models/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning.i1-Q4_K_M.gguf
-      sha256: 7337216f6d42b0771344328da00d454c0fdc91743ced0a4f5a1c6632f4f4b063
-      uri: https://huggingface.co/mradermacher/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning-i1-GGUF/resolve/main/Mistral-Nemo-Instruct-2407-12B-Thinking-M-Claude-Opus-High-Reasoning.i1-Q4_K_M.gguf
- name: "rwkv7-g1c-13.3b"
-  url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
-  urls:
-    - https://huggingface.co/NaomiBTW/rwkv7-g1c-13.3b-gguf
-  description: |
-    The model is **RWKV7 g1c 13B**, a large language model optimized for efficiency. It is quantized using **Bartowski's calibrationv5 for imatrix** to reduce memory usage while maintaining performance. The base model is **BlinkDL/rwkv7-g1**, and this version is tailored for text-generation tasks. It balances accuracy and efficiency, making it suitable for deployment in various applications.
-  overrides:
-    parameters:
-      model: llama-cpp/models/rwkv7-g1c-13.3b-20251231-Q8_0.gguf
-    name: rwkv7-g1c-13.3b-gguf
-    backend: llama-cpp
-    template:
-      use_tokenizer_template: true
-    known_usecases:
-      - chat
-    function:
-      grammar:
-        disable: true
-    description: Imported from https://huggingface.co/NaomiBTW/rwkv7-g1c-13.3b-gguf
-    options:
-      - use_jinja:true
-  files:
-    - filename: llama-cpp/models/rwkv7-g1c-13.3b-20251231-Q8_0.gguf
-      sha256: e06b3b31cee207723be00425cfc25ae09b7fa1abbd7d97eda4e62a7ef254f877
-      uri: https://huggingface.co/NaomiBTW/rwkv7-g1c-13.3b-gguf/resolve/main/rwkv7-g1c-13.3b-20251231-Q8_0.gguf
 - name: "iquest-coder-v1-40b-instruct-i1"
  url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
  urls:
@@ -6111,7 +6011,6 @@
  tags:
    - embeddings
  overrides:
-    backend: llama-cpp
    embeddings: true
    parameters:
      model: granite-embedding-107m-multilingual-f16.gguf
--- a/go.mod
+++ b/go.mod
@@ -25,7 +25,7 @@ require (
 	github.com/jaypipes/ghw v0.21.2
 	github.com/joho/godotenv v1.5.1
 	github.com/klauspost/cpuid/v2 v2.3.0
-	github.com/labstack/echo/v4 v4.15.0
+	github.com/labstack/echo/v4 v4.14.0
 	github.com/libp2p/go-libp2p v0.43.0
 	github.com/lithammer/fuzzysearch v1.1.8
 	github.com/mholt/archiver/v3 v3.5.1
--- a/go.sum
+++ b/go.sum
@@ -386,8 +386,8 @@ github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
 github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE=
 github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
 github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw=
-github.com/labstack/echo/v4 v4.15.0 h1:hoRTKWcnR5STXZFe9BmYun9AMTNeSbjHi2vtDuADJ24=
-github.com/labstack/echo/v4 v4.15.0/go.mod h1:xmw1clThob0BSVRX1CRQkGQ/vjwcpOMjQZSZa9fKA/c=
+github.com/labstack/echo/v4 v4.14.0 h1:+tiMrDLxwv6u0oKtD03mv+V1vXXB3wCqPHJqPuIe+7M=
+github.com/labstack/echo/v4 v4.14.0/go.mod h1:xmw1clThob0BSVRX1CRQkGQ/vjwcpOMjQZSZa9fKA/c=
 github.com/labstack/gommon v0.4.2 h1:F8qTUNXgG1+6WQmqoUWnz8WiEU60mXVVw0P4ht1WRA0=
 github.com/labstack/gommon v0.4.2/go.mod h1:QlUFxVM+SNXhDL/Z7YhocGIBYOiwB0mXm1+1bAPHPyU=
 github.com/libp2p/go-buffer-pool v0.1.0 h1:oK4mSFcQz7cTQIfqbe4MIj9gLW+mnanjyFtc6cdF0Y8=
--- a/pkg/functions/iterative_parser.go
+++ b/pkg/functions/iterative_parser.go
--- a/pkg/functions/json_stack_parser.go
+++ b/pkg/functions/json_stack_parser.go
@@ -1,431 +0,0 @@
-package functions
-
-import (
-	"encoding/json"
-	"errors"
-	"regexp"
-	"strings"
-	"unicode"
-)
-
-// JSONStackElementType represents the type of JSON stack element
-type JSONStackElementType int
-
-const (
-	JSONStackElementObject JSONStackElementType = iota
-	JSONStackElementKey
-	JSONStackElementArray
-)
-
-// JSONStackElement represents an element in the JSON parsing stack
-type JSONStackElement struct {
-	Type JSONStackElementType
-	Key  string
-}
-
-// JSONErrorLocator tracks JSON parsing state and errors
-type JSONErrorLocator struct {
-	position         int
-	foundError       bool
-	lastToken        string
-	exceptionMessage string
-	stack            []JSONStackElement
-}
-
-// parseJSONWithStack parses JSON with stack tracking, matching llama.cpp's common_json_parse
-// Returns the parsed JSON value, whether it was healed, and any error
-func parseJSONWithStack(input string, healingMarker string) (any, bool, string, error) {
-	if healingMarker == "" {
-		// No healing marker, just try to parse normally
-		var result any
-		if err := json.Unmarshal([]byte(input), &result); err != nil {
-			return nil, false, "", err
-		}
-		return result, false, "", nil
-	}
-
-	// Try to parse complete JSON first
-	var result any
-	if err := json.Unmarshal([]byte(input), &result); err == nil {
-		return result, false, "", nil
-	}
-
-	// Parsing failed, need to track stack and heal
-	errLoc := &JSONErrorLocator{
-		position:   0,
-		foundError: false,
-		stack:      make([]JSONStackElement, 0),
-	}
-
-	// Parse with stack tracking to find where error occurs
-	errorPos, err := parseJSONWithStackTracking(input, errLoc)
-	if err == nil && !errLoc.foundError {
-		// No error found, should have parsed successfully
-		var result any
-		if err := json.Unmarshal([]byte(input), &result); err != nil {
-			return nil, false, "", err
-		}
-		return result, false, "", nil
-	}
-
-	if !errLoc.foundError || len(errLoc.stack) == 0 {
-		// Can't heal without stack information
-		return nil, false, "", errors.New("incomplete JSON")
-	}
-
-	// Build closing braces/brackets from stack
-	closing := ""
-	for i := len(errLoc.stack) - 1; i >= 0; i-- {
-		el := errLoc.stack[i]
-		if el.Type == JSONStackElementObject {
-			closing += "}"
-		} else if el.Type == JSONStackElementArray {
-			closing += "]"
-		}
-		// Keys don't add closing characters
-	}
-
-	// Get the partial input up to error position
-	partialInput := input
-	if errorPos > 0 && errorPos < len(input) {
-		partialInput = input[:errorPos]
-	}
-
-	// Find last non-space character
-	lastNonSpacePos := strings.LastIndexFunc(partialInput, func(r rune) bool {
-		return !unicode.IsSpace(r)
-	})
-	if lastNonSpacePos == -1 {
-		return nil, false, "", errors.New("cannot heal a truncated JSON that stopped in an unknown location")
-	}
-	lastNonSpaceChar := rune(partialInput[lastNonSpacePos])
-
-	// Check if we stopped on a number
-	wasMaybeNumber := func() bool {
-		if len(partialInput) > 0 && unicode.IsSpace(rune(partialInput[len(partialInput)-1])) {
-			return false
-		}
-		return unicode.IsDigit(lastNonSpaceChar) ||
-			lastNonSpaceChar == '.' ||
-			lastNonSpaceChar == 'e' ||
-			lastNonSpaceChar == 'E' ||
-			lastNonSpaceChar == '-'
-	}
-
-	// Check for partial unicode escape sequences
-	partialUnicodeRegex := regexp.MustCompile(`\\u(?:[0-9a-fA-F](?:[0-9a-fA-F](?:[0-9a-fA-F](?:[0-9a-fA-F])?)?)?)?$`)
-	unicodeMarkerPadding := "udc00"
-	lastUnicodeMatch := partialUnicodeRegex.FindStringSubmatch(partialInput)
-	if lastUnicodeMatch != nil {
-		// Pad the escape sequence
-		unicodeMarkerPadding = strings.Repeat("0", 6-len(lastUnicodeMatch[0]))
-		// Check if it's a high surrogate
-		if len(lastUnicodeMatch[0]) >= 4 {
-			seq := lastUnicodeMatch[0]
-			if seq[0] == '\\' && seq[1] == 'u' {
-				third := strings.ToLower(string(seq[2]))
-				if third == "d" {
-					fourth := strings.ToLower(string(seq[3]))
-					if fourth == "8" || fourth == "9" || fourth == "a" || fourth == "b" {
-						// High surrogate, add low surrogate
-						unicodeMarkerPadding += "\\udc00"
-					}
-				}
-			}
-		}
-	}
-
-	canParse := func(str string) bool {
-		var test any
-		return json.Unmarshal([]byte(str), &test) == nil
-	}
-
-	// Heal based on stack top element type
-	healedJSON := partialInput
-	jsonDumpMarker := ""
-	topElement := errLoc.stack[len(errLoc.stack)-1]
-
-	if topElement.Type == JSONStackElementKey {
-		// We're inside an object value
-		if lastNonSpaceChar == ':' && canParse(healedJSON+"1"+closing) {
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if canParse(healedJSON + ": 1" + closing) {
-			jsonDumpMarker = ":\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if lastNonSpaceChar == '{' && canParse(healedJSON+closing) {
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else if canParse(healedJSON + "\"" + closing) {
-			jsonDumpMarker = healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if len(healedJSON) > 0 && healedJSON[len(healedJSON)-1] == '\\' && canParse(healedJSON+"\\\""+closing) {
-			jsonDumpMarker = "\\" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if canParse(healedJSON + unicodeMarkerPadding + "\"" + closing) {
-			jsonDumpMarker = unicodeMarkerPadding + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else {
-			// Find last colon and cut back
-			lastColon := strings.LastIndex(healedJSON, ":")
-			if lastColon == -1 {
-				return nil, false, "", errors.New("cannot heal a truncated JSON that stopped in an unknown location")
-			}
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON = healedJSON[:lastColon+1] + jsonDumpMarker + "\"" + closing
-		}
-	} else if topElement.Type == JSONStackElementArray {
-		// We're inside an array
-		if (lastNonSpaceChar == ',' || lastNonSpaceChar == '[') && canParse(healedJSON+"1"+closing) {
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if canParse(healedJSON + "\"" + closing) {
-			jsonDumpMarker = healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if len(healedJSON) > 0 && healedJSON[len(healedJSON)-1] == '\\' && canParse(healedJSON+"\\\""+closing) {
-			jsonDumpMarker = "\\" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if canParse(healedJSON + unicodeMarkerPadding + "\"" + closing) {
-			jsonDumpMarker = unicodeMarkerPadding + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else if !wasMaybeNumber() && canParse(healedJSON+", 1"+closing) {
-			jsonDumpMarker = ",\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\"" + closing
-		} else {
-			lastBracketOrComma := strings.LastIndexAny(healedJSON, "[,")
-			if lastBracketOrComma == -1 {
-				return nil, false, "", errors.New("cannot heal a truncated JSON array stopped in an unknown location")
-			}
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON = healedJSON[:lastBracketOrComma+1] + jsonDumpMarker + "\"" + closing
-		}
-	} else if topElement.Type == JSONStackElementObject {
-		// We're inside an object (expecting a key)
-		if (lastNonSpaceChar == '{' && canParse(healedJSON+closing)) ||
-			(lastNonSpaceChar == ',' && canParse(healedJSON+"\"\": 1"+closing)) {
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else if !wasMaybeNumber() && canParse(healedJSON+",\"\": 1"+closing) {
-			jsonDumpMarker = ",\"" + healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else if canParse(healedJSON + "\": 1" + closing) {
-			jsonDumpMarker = healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else if len(healedJSON) > 0 && healedJSON[len(healedJSON)-1] == '\\' && canParse(healedJSON+"\\\": 1"+closing) {
-			jsonDumpMarker = "\\" + healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else if canParse(healedJSON + unicodeMarkerPadding + "\": 1" + closing) {
-			jsonDumpMarker = unicodeMarkerPadding + healingMarker
-			healedJSON += jsonDumpMarker + "\": 1" + closing
-		} else {
-			lastColon := strings.LastIndex(healedJSON, ":")
-			if lastColon == -1 {
-				return nil, false, "", errors.New("cannot heal a truncated JSON object stopped in an unknown location")
-			}
-			jsonDumpMarker = "\"" + healingMarker
-			healedJSON = healedJSON[:lastColon+1] + jsonDumpMarker + "\"" + closing
-		}
-	} else {
-		return nil, false, "", errors.New("cannot heal a truncated JSON object stopped in an unknown location")
-	}
-
-	// Try to parse the healed JSON
-	var healedValue any
-	if err := json.Unmarshal([]byte(healedJSON), &healedValue); err != nil {
-		return nil, false, "", err
-	}
-
-	// Remove healing marker from result
-	cleaned := removeHealingMarkerFromJSONAny(healedValue, healingMarker)
-	return cleaned, true, jsonDumpMarker, nil
-}
-
-// parseJSONWithStackTracking parses JSON while tracking the stack structure
-// Returns the error position and any error encountered
-// This implements stack tracking similar to llama.cpp's json_error_locator
-func parseJSONWithStackTracking(input string, errLoc *JSONErrorLocator) (int, error) {
-	// First, try to parse to get exact error position
-	decoder := json.NewDecoder(strings.NewReader(input))
-	var test any
-	err := decoder.Decode(&test)
-	if err != nil {
-		errLoc.foundError = true
-		errLoc.exceptionMessage = err.Error()
-
-		var errorPos int
-		if syntaxErr, ok := err.(*json.SyntaxError); ok {
-			errorPos = int(syntaxErr.Offset)
-			errLoc.position = errorPos
-		} else {
-			// Fallback: use end of input
-			errorPos = len(input)
-			errLoc.position = errorPos
-		}
-
-		// Now build the stack by parsing up to the error position
-		// This matches llama.cpp's approach of tracking stack during SAX parsing
-		partialInput := input
-		if errorPos > 0 && errorPos < len(input) {
-			partialInput = input[:errorPos]
-		}
-
-		// Track stack by parsing character by character up to error
-		pos := 0
-		inString := false
-		escape := false
-		keyStart := -1
-		keyEnd := -1
-
-		for pos < len(partialInput) {
-			ch := partialInput[pos]
-
-			if escape {
-				escape = false
-				pos++
-				continue
-			}
-
-			if ch == '\\' {
-				escape = true
-				pos++
-				continue
-			}
-
-			if ch == '"' {
-				if !inString {
-					// Starting a string
-					inString = true
-					// Check if we're in an object context (expecting a key)
-					if len(errLoc.stack) > 0 {
-						top := errLoc.stack[len(errLoc.stack)-1]
-						if top.Type == JSONStackElementObject {
-							// This could be a key
-							keyStart = pos + 1 // Start after quote
-						}
-					}
-				} else {
-					// Ending a string
-					inString = false
-					if keyStart != -1 {
-						// This was potentially a key, extract it
-						keyEnd = pos
-						key := partialInput[keyStart:keyEnd]
-
-						// Look ahead to see if next non-whitespace is ':'
-						nextPos := pos + 1
-						for nextPos < len(partialInput) && unicode.IsSpace(rune(partialInput[nextPos])) {
-							nextPos++
-						}
-						if nextPos < len(partialInput) && partialInput[nextPos] == ':' {
-							// This is a key, add it to stack
-							errLoc.stack = append(errLoc.stack, JSONStackElement{Type: JSONStackElementKey, Key: key})
-						}
-						keyStart = -1
-						keyEnd = -1
-					}
-				}
-				pos++
-				continue
-			}
-
-			if inString {
-				pos++
-				continue
-			}
-
-			// Handle stack operations (outside strings)
-			if ch == '{' {
-				errLoc.stack = append(errLoc.stack, JSONStackElement{Type: JSONStackElementObject})
-			} else if ch == '}' {
-				// Pop object and any key on top (keys are popped when value starts, but handle here too)
-				for len(errLoc.stack) > 0 {
-					top := errLoc.stack[len(errLoc.stack)-1]
-					errLoc.stack = errLoc.stack[:len(errLoc.stack)-1]
-					if top.Type == JSONStackElementObject {
-						break
-					}
-				}
-			} else if ch == '[' {
-				errLoc.stack = append(errLoc.stack, JSONStackElement{Type: JSONStackElementArray})
-			} else if ch == ']' {
-				// Pop array
-				for len(errLoc.stack) > 0 {
-					top := errLoc.stack[len(errLoc.stack)-1]
-					errLoc.stack = errLoc.stack[:len(errLoc.stack)-1]
-					if top.Type == JSONStackElementArray {
-						break
-					}
-				}
-			} else if ch == ':' {
-				// Colon means we're starting a value, pop the key if it's on stack
-				if len(errLoc.stack) > 0 && errLoc.stack[len(errLoc.stack)-1].Type == JSONStackElementKey {
-					errLoc.stack = errLoc.stack[:len(errLoc.stack)-1]
-				}
-			}
-			// Note: commas and whitespace don't affect stack structure
-
-			pos++
-		}
-
-		return errorPos, err
-	}
-
-	// No error, parse was successful - build stack anyway for completeness
-	// (though we shouldn't need healing in this case)
-	pos := 0
-	inString := false
-	escape := false
-
-	for pos < len(input) {
-		ch := input[pos]
-
-		if escape {
-			escape = false
-			pos++
-			continue
-		}
-
-		if ch == '\\' {
-			escape = true
-			pos++
-			continue
-		}
-
-		if ch == '"' {
-			inString = !inString
-			pos++
-			continue
-		}
-
-		if inString {
-			pos++
-			continue
-		}
-
-		if ch == '{' {
-			errLoc.stack = append(errLoc.stack, JSONStackElement{Type: JSONStackElementObject})
-		} else if ch == '}' {
-			for len(errLoc.stack) > 0 {
-				top := errLoc.stack[len(errLoc.stack)-1]
-				errLoc.stack = errLoc.stack[:len(errLoc.stack)-1]
-				if top.Type == JSONStackElementObject {
-					break
-				}
-			}
-		} else if ch == '[' {
-			errLoc.stack = append(errLoc.stack, JSONStackElement{Type: JSONStackElementArray})
-		} else if ch == ']' {
-			for len(errLoc.stack) > 0 {
-				top := errLoc.stack[len(errLoc.stack)-1]
-				errLoc.stack = errLoc.stack[:len(errLoc.stack)-1]
-				if top.Type == JSONStackElementArray {
-					break
-				}
-			}
-		}
-
-		pos++
-	}
-
-	return len(input), nil
-}
--- a/pkg/functions/parse.go
+++ b/pkg/functions/parse.go
--- a/pkg/functions/parse_test.go
+++ b/pkg/functions/parse_test.go
--- a/pkg/system/capabilities.go
+++ b/pkg/system/capabilities.go
@@ -8,6 +8,7 @@ import (
 	"runtime"
 	"strings"

+	"github.com/jaypipes/ghw/pkg/gpu"
 	"github.com/mudler/xlog"
 )

@@ -18,9 +19,8 @@ const (
 	metal             = "metal"
 	nvidia            = "nvidia"

-	amd    = "amd"
-	intel  = "intel"
-	vulkan = "vulkan"
+	amd   = "amd"
+	intel = "intel"

 	nvidiaCuda13    = "nvidia-cuda-13"
 	nvidiaCuda12    = "nvidia-cuda-12"
@@ -131,6 +131,26 @@ func (s *SystemState) getSystemCapabilities() string {
 	return s.GPUVendor
 }

+func detectGPUVendor(gpus []*gpu.GraphicsCard) (string, error) {
+	for _, gpu := range gpus {
+		if gpu.DeviceInfo != nil {
+			if gpu.DeviceInfo.Vendor != nil {
+				gpuVendorName := strings.ToUpper(gpu.DeviceInfo.Vendor.Name)
+				if strings.Contains(gpuVendorName, strings.ToUpper(nvidia)) {
+					return nvidia, nil
+				}
+				if strings.Contains(gpuVendorName, strings.ToUpper(amd)) {
+					return amd, nil
+				}
+				if strings.Contains(gpuVendorName, strings.ToUpper(intel)) {
+					return intel, nil
+				}
+			}
+		}
+	}
+
+	return "", nil
+}

 // BackendPreferenceTokens returns a list of substrings that represent the preferred
 // backend implementation order for the current system capability. Callers can use
@@ -149,8 +169,6 @@ func (s *SystemState) BackendPreferenceTokens() []string {
 		return []string{"metal", "cpu"}
 	case strings.HasPrefix(capStr, darwinX86):
 		return []string{"darwin-x86", "cpu"}
-	case strings.HasPrefix(capStr, vulkan):
-		return []string{"vulkan", "cpu"}
 	default:
 		return []string{"cpu"}
 	}
--- a/pkg/system/state.go
+++ b/pkg/system/state.go
@@ -1,6 +1,7 @@
 package system

 import (
+	"github.com/jaypipes/ghw/pkg/gpu"
 	"github.com/mudler/LocalAI/pkg/xsysinfo"
 	"github.com/mudler/xlog"
 )
@@ -18,6 +19,7 @@ type SystemState struct {
 	GPUVendor string
 	Backend   Backend
 	Model     Model
+	gpus      []*gpu.GraphicsCard
 	VRAM      uint64
 }

@@ -48,7 +50,9 @@ func GetSystemState(opts ...SystemStateOptions) (*SystemState, error) {
 	}

 	// Detection is best-effort here, we don't want to fail if it fails
-	state.GPUVendor, _ = xsysinfo.DetectGPUVendor()
+	state.gpus, _ = xsysinfo.GPUs()
+	xlog.Debug("GPUs", "gpus", state.gpus)
+	state.GPUVendor, _ = detectGPUVendor(state.gpus)
 	xlog.Debug("GPU vendor", "gpuVendor", state.GPUVendor)
 	state.VRAM, _ = xsysinfo.TotalAvailableVRAM()
 	xlog.Debug("Total available VRAM", "vram", state.VRAM)
--- a/pkg/xsysinfo/gpu.go
+++ b/pkg/xsysinfo/gpu.go
@@ -89,39 +89,21 @@ func GPUs() ([]*gpu.GraphicsCard, error) {
 }

 func TotalAvailableVRAM() (uint64, error) {
-	// First, try ghw library detection
 	gpus, err := GPUs()
-	if err == nil {
-		var totalVRAM uint64
-		for _, gpu := range gpus {
-			if gpu != nil && gpu.Node != nil && gpu.Node.Memory != nil {
-				if gpu.Node.Memory.TotalUsableBytes > 0 {
-					totalVRAM += uint64(gpu.Node.Memory.TotalUsableBytes)
-				}
+	if err != nil {
+		return 0, err
+	}
+
+	var totalVRAM uint64
+	for _, gpu := range gpus {
+		if gpu != nil && gpu.Node != nil && gpu.Node.Memory != nil {
+			if gpu.Node.Memory.TotalUsableBytes > 0 {
+				totalVRAM += uint64(gpu.Node.Memory.TotalUsableBytes)
 			}
 		}
-		// If we got valid VRAM from ghw, return it
-		if totalVRAM > 0 {
-			return totalVRAM, nil
-		}
 	}

-	// Fallback to binary-based detection via GetGPUMemoryUsage()
-	// This works even when ghw dependencies are missing from the base image
-	gpuMemoryInfo := GetGPUMemoryUsage()
-	if len(gpuMemoryInfo) > 0 {
-		var totalVRAM uint64
-		for _, gpu := range gpuMemoryInfo {
-			totalVRAM += gpu.TotalVRAM
-		}
-		if totalVRAM > 0 {
-			xlog.Debug("VRAM detected via binary tools", "total_vram", totalVRAM)
-			return totalVRAM, nil
-		}
-	}
-
-	// No VRAM detected
-	return 0, nil
+	return totalVRAM, nil
 }

 func HasGPU(vendor string) bool {
@@ -140,66 +122,6 @@ func HasGPU(vendor string) bool {
 	return false
 }

-// DetectGPUVendor detects the GPU vendor using multiple methods with fallbacks.
-// First tries ghw library, then falls back to binary detection.
-// Returns vendor string (VendorNVIDIA, VendorAMD, VendorIntel, VendorVulkan) or empty string if not detected.
-// Priority order: NVIDIA > AMD > Intel > Vulkan
-func DetectGPUVendor() (string, error) {
-	// First, try ghw library detection
-	gpus, err := GPUs()
-	if err == nil && len(gpus) > 0 {
-		for _, gpu := range gpus {
-			if gpu.DeviceInfo != nil && gpu.DeviceInfo.Vendor != nil {
-				vendorName := strings.ToUpper(gpu.DeviceInfo.Vendor.Name)
-				if strings.Contains(vendorName, strings.ToUpper(VendorNVIDIA)) {
-					xlog.Debug("GPU vendor detected via ghw", "vendor", VendorNVIDIA)
-					return VendorNVIDIA, nil
-				}
-				if strings.Contains(vendorName, strings.ToUpper(VendorAMD)) {
-					xlog.Debug("GPU vendor detected via ghw", "vendor", VendorAMD)
-					return VendorAMD, nil
-				}
-				if strings.Contains(vendorName, strings.ToUpper(VendorIntel)) {
-					xlog.Debug("GPU vendor detected via ghw", "vendor", VendorIntel)
-					return VendorIntel, nil
-				}
-			}
-		}
-	}
-
-	// Fallback to binary detection (priority: NVIDIA > AMD > Intel > Vulkan)
-	// Check for nvidia-smi
-	if _, err := exec.LookPath("nvidia-smi"); err == nil {
-		xlog.Debug("GPU vendor detected via binary", "vendor", VendorNVIDIA, "binary", "nvidia-smi")
-		return VendorNVIDIA, nil
-	}
-
-	// Check for rocm-smi (AMD)
-	if _, err := exec.LookPath("rocm-smi"); err == nil {
-		xlog.Debug("GPU vendor detected via binary", "vendor", VendorAMD, "binary", "rocm-smi")
-		return VendorAMD, nil
-	}
-
-	// Check for xpu-smi or intel_gpu_top (Intel)
-	if _, err := exec.LookPath("xpu-smi"); err == nil {
-		xlog.Debug("GPU vendor detected via binary", "vendor", VendorIntel, "binary", "xpu-smi")
-		return VendorIntel, nil
-	}
-	if _, err := exec.LookPath("intel_gpu_top"); err == nil {
-		xlog.Debug("GPU vendor detected via binary", "vendor", VendorIntel, "binary", "intel_gpu_top")
-		return VendorIntel, nil
-	}
-
-	// Check for vulkaninfo (Vulkan - lowest priority as it can detect any GPU)
-	if _, err := exec.LookPath("vulkaninfo"); err == nil {
-		xlog.Debug("GPU vendor detected via binary", "vendor", VendorVulkan, "binary", "vulkaninfo")
-		return VendorVulkan, nil
-	}
-
-	// No vendor detected
-	return "", nil
-}
-
 // isUnifiedMemoryDevice checks if the given GPU name matches any known unified memory device
 func isUnifiedMemoryDevice(gpuName string) bool {
 	gpuNameUpper := strings.ToUpper(gpuName)
--- a/scripts/build/package-gpu-libs.sh
+++ b/scripts/build/package-gpu-libs.sh
@@ -1,347 +0,0 @@
-#!/bin/bash
-# Script to package GPU libraries based on BUILD_TYPE
-# This script copies GPU-specific runtime libraries to a target lib directory
-# so backends can run in isolation with their own GPU libraries.
-#
-# Usage: source package-gpu-libs.sh TARGET_LIB_DIR
-#        package_gpu_libs
-#
-# Environment variables:
-#   BUILD_TYPE - The GPU build type (cublas, l4t, hipblas, sycl_f16, sycl_f32, intel, vulkan)
-#   CUDA_MAJOR_VERSION - CUDA major version (for cublas/l4t builds)
-#
-# This enables backends to be fully self-contained and run on a unified base image
-# without requiring GPU drivers to be pre-installed in the host image.
-
-set -e
-
-TARGET_LIB_DIR="${1:-./lib}"
-
-# Create target directory if it doesn't exist
-mkdir -p "$TARGET_LIB_DIR"
-
-# Associative array to track copied files by basename
-# Note: We use basename for deduplication because the target is a flat directory.
-# If the same library exists in multiple source paths, we only copy it once.
-declare -A COPIED_FILES
-
-# Helper function to copy library preserving symlinks structure
-# Instead of following symlinks and duplicating files, this function:
-# 1. Resolves symlinks to their real target
-# 2. Copies the real file only once
-# 3. Recreates symlinks pointing to the real file
-copy_lib() {
-    local src="$1"
-
-    # Check if source exists (follows symlinks)
-    if [ ! -e "$src" ]; then
-        return
-    fi
-
-    local src_basename
-    src_basename=$(basename "$src")
-
-    # Skip if we've already processed this filename
-    if [[ -n "${COPIED_FILES[$src_basename]:-}" ]]; then
-        return
-    fi
-
-    if [ -L "$src" ]; then
-        # Source is a symbolic link
-        # Resolve the real file (following all symlinks)
-        local real_file
-        real_file=$(readlink -f "$src")
-
-        if [ ! -e "$real_file" ]; then
-            echo "Warning: symlink target does not exist: $src -> $real_file" >&2
-            return
-        fi
-
-        local real_basename
-        real_basename=$(basename "$real_file")
-
-        # Copy the real file if we haven't already
-        if [[ -z "${COPIED_FILES[$real_basename]:-}" ]]; then
-            cp -v "$real_file" "$TARGET_LIB_DIR/$real_basename" 2>/dev/null || true
-            COPIED_FILES[$real_basename]=1
-        fi
-
-        # Create the symlink if the source name differs from the real file name
-        if [ "$src_basename" != "$real_basename" ]; then
-            # Point directly to the real file for simplicity and reliability
-            ln -sfv "$real_basename" "$TARGET_LIB_DIR/$src_basename" 2>/dev/null || true
-        fi
-        COPIED_FILES[$src_basename]=1
-    else
-        # Source is a regular file - copy if not already copied
-        if [[ -z "${COPIED_FILES[$src_basename]:-}" ]]; then
-            cp -v "$src" "$TARGET_LIB_DIR/$src_basename" 2>/dev/null || true
-        fi
-        COPIED_FILES[$src_basename]=1
-    fi
-}
-
-# Helper function to copy all matching libraries from a glob pattern
-# Files are sorted so that regular files are processed before symlinks
-copy_libs_glob() {
-    local pattern="$1"
-    # Use nullglob option to handle non-matching patterns gracefully
-    local old_nullglob=$(shopt -p nullglob)
-    shopt -s nullglob
-    local matched=($pattern)
-    eval "$old_nullglob"
-
-    # Sort files: regular files first, then symlinks
-    # This ensures real files are copied before we try to create symlinks pointing to them
-    local regular_files=()
-    local symlinks=()
-    for file in "${matched[@]}"; do
-        if [ -L "$file" ]; then
-            symlinks+=("$file")
-        elif [ -e "$file" ]; then
-            regular_files+=("$file")
-        fi
-    done
-
-    # Process regular files first, then symlinks
-    for lib in "${regular_files[@]}" "${symlinks[@]}"; do
-        copy_lib "$lib"
-    done
-}
-
-# Package NVIDIA CUDA libraries
-package_cuda_libs() {
-    echo "Packaging CUDA libraries for BUILD_TYPE=${BUILD_TYPE}..."
-
-    local cuda_lib_paths=(
-        "/usr/local/cuda/lib64"
-        "/usr/local/cuda-${CUDA_MAJOR_VERSION:-}/lib64"
-        "/usr/lib/x86_64-linux-gnu"
-        "/usr/lib/aarch64-linux-gnu"
-    )
-
-    # Core CUDA runtime libraries
-    local cuda_libs=(
-        "libcudart.so*"
-        "libcublas.so*"
-        "libcublasLt.so*"
-        "libcufft.so*"
-        "libcurand.so*"
-        "libcusparse.so*"
-        "libcusolver.so*"
-        "libnvrtc.so*"
-        "libnvrtc-builtins.so*"
-        "libcudnn.so*"
-        "libcudnn_ops.so*"
-        "libcudnn_cnn.so*"
-        "libnvJitLink.so*"
-        "libnvinfer.so*"
-        "libnvonnxparser.so*"
-    )
-
-    for lib_path in "${cuda_lib_paths[@]}"; do
-        if [ -d "$lib_path" ]; then
-            for lib_pattern in "${cuda_libs[@]}"; do
-                copy_libs_glob "${lib_path}/${lib_pattern}"
-            done
-        fi
-    done
-
-    # Copy CUDA target directory for runtime compilation support
-    if [ -d "/usr/local/cuda/targets" ]; then
-        mkdir -p "$TARGET_LIB_DIR/../cuda"
-        cp -arfL /usr/local/cuda/targets "$TARGET_LIB_DIR/../cuda/" 2>/dev/null || true
-    fi
-
-    echo "CUDA libraries packaged successfully"
-}
-
-# Package AMD ROCm/HIPBlas libraries
-package_rocm_libs() {
-    echo "Packaging ROCm/HIPBlas libraries for BUILD_TYPE=${BUILD_TYPE}..."
-
-    local rocm_lib_paths=(
-        "/opt/rocm/lib"
-        "/opt/rocm/lib64"
-        "/opt/rocm/hip/lib"
-    )
-
-    # Find the actual ROCm versioned directory
-    for rocm_dir in /opt/rocm-*; do
-        if [ -d "$rocm_dir/lib" ]; then
-            rocm_lib_paths+=("$rocm_dir/lib")
-        fi
-    done
-
-    # Core ROCm/HIP runtime libraries
-    local rocm_libs=(
-        "libamdhip64.so*"
-        "libhipblas.so*"
-        "librocblas.so*"
-        "librocrand.so*"
-        "librocsparse.so*"
-        "librocsolver.so*"
-        "librocfft.so*"
-        "libMIOpen.so*"
-        "libroctx64.so*"
-        "libhsa-runtime64.so*"
-        "libamd_comgr.so*"
-        "libhip_hcc.so*"
-        "libhiprtc.so*"
-    )
-
-    for lib_path in "${rocm_lib_paths[@]}"; do
-        if [ -d "$lib_path" ]; then
-            for lib_pattern in "${rocm_libs[@]}"; do
-                copy_libs_glob "${lib_path}/${lib_pattern}"
-            done
-        fi
-    done
-
-    # Copy rocblas library data (tuning files, etc.)
-    local old_nullglob=$(shopt -p nullglob)
-    shopt -s nullglob
-    local rocm_dirs=(/opt/rocm /opt/rocm-*)
-    eval "$old_nullglob"
-    for rocm_base in "${rocm_dirs[@]}"; do
-        if [ -d "$rocm_base/lib/rocblas" ]; then
-            mkdir -p "$TARGET_LIB_DIR/rocblas"
-            cp -arfL "$rocm_base/lib/rocblas/"* "$TARGET_LIB_DIR/rocblas/" 2>/dev/null || true
-        fi
-    done
-
-    # Copy libomp from LLVM (required for ROCm)
-    shopt -s nullglob
-    local omp_libs=(/opt/rocm*/lib/llvm/lib/libomp.so*)
-    eval "$old_nullglob"
-    for omp_path in "${omp_libs[@]}"; do
-        if [ -e "$omp_path" ]; then
-            copy_lib "$omp_path"
-        fi
-    done
-
-    echo "ROCm libraries packaged successfully"
-}
-
-# Package Intel oneAPI/SYCL libraries
-package_intel_libs() {
-    echo "Packaging Intel oneAPI/SYCL libraries for BUILD_TYPE=${BUILD_TYPE}..."
-
-    local intel_lib_paths=(
-        "/opt/intel/oneapi/compiler/latest/lib"
-        "/opt/intel/oneapi/mkl/latest/lib/intel64"
-        "/opt/intel/oneapi/tbb/latest/lib/intel64/gcc4.8"
-    )
-
-    # Core Intel oneAPI runtime libraries
-    local intel_libs=(
-        "libsycl.so*"
-        "libOpenCL.so*"
-        "libmkl_core.so*"
-        "libmkl_intel_lp64.so*"
-        "libmkl_intel_thread.so*"
-        "libmkl_sequential.so*"
-        "libmkl_sycl.so*"
-        "libiomp5.so*"
-        "libsvml.so*"
-        "libirng.so*"
-        "libimf.so*"
-        "libintlc.so*"
-        "libtbb.so*"
-        "libtbbmalloc.so*"
-        "libpi_level_zero.so*"
-        "libpi_opencl.so*"
-        "libze_loader.so*"
-    )
-
-    for lib_path in "${intel_lib_paths[@]}"; do
-        if [ -d "$lib_path" ]; then
-            for lib_pattern in "${intel_libs[@]}"; do
-                copy_libs_glob "${lib_path}/${lib_pattern}"
-            done
-        fi
-    done
-
-    echo "Intel oneAPI libraries packaged successfully"
-}
-
-# Package Vulkan libraries
-package_vulkan_libs() {
-    echo "Packaging Vulkan libraries for BUILD_TYPE=${BUILD_TYPE}..."
-
-    local vulkan_lib_paths=(
-        "/usr/lib/x86_64-linux-gnu"
-        "/usr/lib/aarch64-linux-gnu"
-        "/usr/local/lib"
-    )
-
-    # Core Vulkan runtime libraries
-    local vulkan_libs=(
-        "libvulkan.so*"
-        "libshaderc_shared.so*"
-        "libSPIRV.so*"
-        "libSPIRV-Tools.so*"
-        "libglslang.so*"
-    )
-
-    for lib_path in "${vulkan_lib_paths[@]}"; do
-        if [ -d "$lib_path" ]; then
-            for lib_pattern in "${vulkan_libs[@]}"; do
-                copy_libs_glob "${lib_path}/${lib_pattern}"
-            done
-        fi
-    done
-
-    # Copy Vulkan ICD files
-    if [ -d "/usr/share/vulkan/icd.d" ]; then
-        mkdir -p "$TARGET_LIB_DIR/../vulkan/icd.d"
-        cp -arfL /usr/share/vulkan/icd.d/* "$TARGET_LIB_DIR/../vulkan/icd.d/" 2>/dev/null || true
-    fi
-
-    echo "Vulkan libraries packaged successfully"
-}
-
-# Main function to package GPU libraries based on BUILD_TYPE
-package_gpu_libs() {
-    local build_type="${BUILD_TYPE:-}"
-
-    echo "Packaging GPU libraries for BUILD_TYPE=${build_type}..."
-
-    case "$build_type" in
-        cublas|l4t)
-            package_cuda_libs
-            ;;
-        hipblas)
-            package_rocm_libs
-            ;;
-        sycl_f16|sycl_f32|intel)
-            package_intel_libs
-            ;;
-        vulkan)
-            package_vulkan_libs
-            ;;
-        ""|cpu)
-            echo "No GPU libraries to package for BUILD_TYPE=${build_type}"
-            ;;
-        *)
-            echo "Unknown BUILD_TYPE: ${build_type}, skipping GPU library packaging"
-            ;;
-    esac
-
-    echo "GPU library packaging complete. Contents of ${TARGET_LIB_DIR}:"
-    ls -la "$TARGET_LIB_DIR/" 2>/dev/null || echo "  (empty or not created)"
-}
-
-# Export the function so it can be sourced and called
-export -f package_gpu_libs
-export -f copy_lib
-export -f copy_libs_glob
-export -f package_cuda_libs
-export -f package_rocm_libs
-export -f package_intel_libs
-export -f package_vulkan_libs
-
-# If script is run directly (not sourced), execute the packaging
-if [[ "${BASH_SOURCE[0]}" == "${0}" ]]; then
-    package_gpu_libs
-fi
--- a/swagger/docs.go
+++ b/swagger/docs.go
@@ -702,6 +702,30 @@ const docTemplate = `{
                }
            }
        },
+        "/mcp/v1/completions": {
+            "post": {
+                "summary": "Generate completions for a given prompt and model.",
+                "parameters": [
+                    {
+                        "description": "query params",
+                        "name": "request",
+                        "in": "body",
+                        "required": true,
+                        "schema": {
+                            "$ref": "#/definitions/schema.OpenAIRequest"
+                        }
+                    }
+                ],
+                "responses": {
+                    "200": {
+                        "description": "Response",
+                        "schema": {
+                            "$ref": "#/definitions/schema.OpenAIResponse"
+                        }
+                    }
+                }
+            }
+        },
        "/metrics": {
            "get": {
                "summary": "Prometheus metrics endpoint",
--- a/swagger/swagger.json
+++ b/swagger/swagger.json
@@ -695,6 +695,30 @@
                }
            }
        },
+        "/mcp/v1/completions": {
+            "post": {
+                "summary": "Generate completions for a given prompt and model.",
+                "parameters": [
+                    {
+                        "description": "query params",
+                        "name": "request",
+                        "in": "body",
+                        "required": true,
+                        "schema": {
+                            "$ref": "#/definitions/schema.OpenAIRequest"
+                        }
+                    }
+                ],
+                "responses": {
+                    "200": {
+                        "description": "Response",
+                        "schema": {
+                            "$ref": "#/definitions/schema.OpenAIResponse"
+                        }
+                    }
+                }
+            }
+        },
        "/metrics": {
            "get": {
                "summary": "Prometheus metrics endpoint",
--- a/swagger/swagger.yaml
+++ b/swagger/swagger.yaml
@@ -1495,6 +1495,21 @@ paths:
          schema:
            $ref: '#/definitions/services.GalleryOpStatus'
      summary: Returns the job status
+  /mcp/v1/completions:
+    post:
+      parameters:
+      - description: query params
+        in: body
+        name: request
+        required: true
+        schema:
+          $ref: '#/definitions/schema.OpenAIRequest'
+      responses:
+        "200":
+          description: Response
+          schema:
+            $ref: '#/definitions/schema.OpenAIResponse'
+      summary: Generate completions for a given prompt and model.
  /metrics:
    get:
      parameters: