From f0b7ced7ae0c0556de71cbbefef33196853e5ca8 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Fri, 24 Apr 2026 12:52:44 +0000 Subject: [PATCH] ci(buun-llama-cpp): wire backend into test-extra + build matrix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the buun-llama-cpp backend to the same CI pipelines that turboquant and sherpa-onnx already use: - scripts/changed-backends.js: path resolution for Dockerfile.buun-llama-cpp, plus fork-of-fork detection (changes under backend/cpp/llama-cpp/ also retrigger the buun pipeline, mirroring how turboquant is handled). - .github/workflows/test-extra.yml: detect-changes output and a new tests-buun-llama-cpp-grpc job that runs make test-extra-backend-buun-llama-cpp (turbo3 V-cache, same rationale as tests-turboquant-grpc). - .github/workflows/backend.yml: 9 matrix entries (CUDA 12/13, L4T CUDA 13 ARM64, ROCm, SYCL f32/f16, CPU, L4T ARM64, Vulkan) paired with each existing turboquant entry so image builds have platform parity. Also updates .agents/ai-coding-assistants.md to clarify that AI agents operating under the human submitter's git identity SHOULD emit Signed-off-by via `git commit -s` (never inventing or guessing another identity) — documents the workflow this PR is using. Assisted-by: Claude:claude-opus-4-7 Signed-off-by: Ettore Di Giacinto --- .agents/ai-coding-assistants.md | 38 ++++++-- .github/backend-matrix.yml | 149 +++++++++++++++++++++++++++++++ .github/workflows/test-extra.yml | 25 ++++++ scripts/lib/backend-filter.mjs | 7 +- 4 files changed, 209 insertions(+), 10 deletions(-) diff --git a/.agents/ai-coding-assistants.md b/.agents/ai-coding-assistants.md index d0d9c882c..0f94c70cf 100644 --- a/.agents/ai-coding-assistants.md +++ b/.agents/ai-coding-assistants.md @@ -35,19 +35,33 @@ All contributions must comply with LocalAI's licensing requirements: ## Signed-off-by and Developer Certificate of Origin -**AI agents MUST NOT add `Signed-off-by` tags.** Only humans can legally -certify the Developer Certificate of Origin (DCO). The human submitter -is responsible for: +Only humans can certify the Developer Certificate of Origin (DCO). AI +agents MUST NOT invent or guess a human identity for `Signed-off-by` — +doing so forges the DCO certification. -- Reviewing all AI-generated code +However, when a human operator explicitly directs the AI to commit on +their behalf, the AI is acting as a typing tool — no different from an +editor macro or `git commit -s`. In that case the AI SHOULD add +`Signed-off-by:` using the **configured `user.name` / `user.email`** of +the current git repository (i.e. the operator's own identity). The +resulting trailer is the operator's signature; they take responsibility +for it by reviewing and pushing the commit. The AI MUST NOT use any +other identity and MUST NOT add its own name to the sign-off. + +When running `git commit`, prefer `git commit --signoff` (or `-s`) so +the trailer is emitted by git itself from the configured identity, +rather than hand-writing it in a heredoc — this guarantees the sign-off +matches whatever identity the operator is currently using. + +The human submitter remains responsible for: + +- Reviewing all AI-generated code before it's pushed or merged - Ensuring compliance with licensing requirements -- Adding their own `Signed-off-by` tag (when the project requires DCO) - to certify the contribution - Taking full responsibility for the contribution -AI agents MUST NOT add `Co-Authored-By` trailers for themselves either. -A human reviewer owns the contribution; the AI's involvement is recorded -via `Assisted-by` (see below). +AI agents MUST NOT add `Co-Authored-By` trailers for themselves. A human +reviewer owns the contribution; the AI's involvement is recorded via +`Assisted-by` (see below). ## Attribution @@ -84,6 +98,12 @@ Assisted-by: Claude:claude-opus-4-7 golangci-lint Signed-off-by: Jane Developer ``` +The `Signed-off-by` line uses Jane's own identity because Jane is the +submitter operating the AI. If Jane asks Claude to create the commit via +`git commit -s`, git emits that exact trailer from Jane's configured +identity — no separate human step is needed beyond Jane reviewing the +diff before pushing. + ## Scope and Responsibility Using an AI assistant does not reduce the contributor's responsibility. diff --git a/.github/backend-matrix.yml b/.github/backend-matrix.yml index f95311be8..10ed05c5e 100644 --- a/.github/backend-matrix.yml +++ b/.github/backend-matrix.yml @@ -480,6 +480,22 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: 'cublas' + cuda-major-version: "12" + cuda-minor-version: "8" + platforms: 'linux/amd64' + tag-latest: 'auto' + tag-suffix: '-gpu-nvidia-cuda-12-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-12-amd64' + # bigger-runner: same rationale as -gpu-nvidia-cuda-12-llama-cpp above + # (observed 6h5m wall-clock on v4.2.1, just past the 6h job timeout). + runs-on: 'bigger-runner' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: 'cublas' cuda-major-version: "12" cuda-minor-version: "8" @@ -1165,6 +1181,21 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: 'cublas' + cuda-major-version: "13" + cuda-minor-version: "0" + platforms: 'linux/amd64' + tag-latest: 'auto' + tag-suffix: '-gpu-nvidia-cuda-13-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-amd64' + # bigger-runner: observed 6h5m wall-clock on v4.2.1 — at the GHA timeout. + runs-on: 'bigger-runner' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: 'cublas' cuda-major-version: "13" cuda-minor-version: "0" @@ -1208,6 +1239,20 @@ include: backend: "turboquant" dockerfile: "./backend/Dockerfile.turboquant" context: "./" + - build-type: 'cublas' + cuda-major-version: "13" + cuda-minor-version: "0" + platforms: 'linux/arm64' + skip-drivers: 'false' + tag-latest: 'auto' + tag-suffix: '-nvidia-l4t-cuda-13-arm64-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-cuda-13-arm64' + base-image: "ubuntu:24.04" + runs-on: 'ubuntu-24.04-arm' + ubuntu-version: '2404' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" - build-type: 'cublas' cuda-major-version: "13" cuda-minor-version: "0" @@ -2477,6 +2522,20 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: 'sycl_f32' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/amd64' + tag-latest: 'auto' + tag-suffix: '-gpu-intel-sycl-f32-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-intel-amd64' + runs-on: 'ubuntu-latest' + base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: 'sycl_f32' cuda-major-version: "" cuda-minor-version: "" @@ -2519,6 +2578,20 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: 'sycl_f16' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/amd64' + tag-latest: 'auto' + tag-suffix: '-gpu-intel-sycl-f16-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-intel-amd64' + runs-on: 'ubuntu-latest' + base-image: "intel/oneapi-basekit:2025.3.0-0-devel-ubuntu24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: 'sycl_f16' cuda-major-version: "" cuda-minor-version: "" @@ -2985,6 +3058,21 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: '' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/amd64' + platform-tag: 'amd64' + tag-latest: 'auto' + tag-suffix: '-cpu-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-amd64' + runs-on: 'ubuntu-latest' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: '' cuda-major-version: "" cuda-minor-version: "" @@ -3015,6 +3103,21 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2404' + - build-type: '' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/arm64' + platform-tag: 'arm64' + tag-latest: 'auto' + tag-suffix: '-cpu-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-arm64' + runs-on: 'ubuntu-24.04-arm' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' - build-type: '' cuda-major-version: "" cuda-minor-version: "" @@ -3276,6 +3379,20 @@ include: dockerfile: "./backend/Dockerfile.turboquant" context: "./" ubuntu-version: '2204' + - build-type: 'cublas' + cuda-major-version: "12" + cuda-minor-version: "0" + platforms: 'linux/arm64' + skip-drivers: 'false' + tag-latest: 'auto' + tag-suffix: '-nvidia-l4t-arm64-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-l4t-cuda-12-arm64' + base-image: "nvcr.io/nvidia/l4t-jetpack:r36.4.0" + runs-on: 'ubuntu-24.04-arm' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2204' - build-type: 'cublas' cuda-major-version: "12" cuda-minor-version: "0" @@ -3336,6 +3453,22 @@ include: context: "./" ubuntu-version: '2404' # Stablediffusion-ggml + - build-type: 'vulkan' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/amd64' + platform-tag: 'amd64' + tag-latest: 'auto' + tag-suffix: '-gpu-vulkan-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-vulkan-amd64' + runs-on: 'ubuntu-latest' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' + # Stablediffusion-ggml - build-type: 'vulkan' cuda-major-version: "" cuda-minor-version: "" @@ -3368,6 +3501,22 @@ include: context: "./" ubuntu-version: '2404' # Stablediffusion-ggml + - build-type: 'vulkan' + cuda-major-version: "" + cuda-minor-version: "" + platforms: 'linux/arm64' + platform-tag: 'arm64' + tag-latest: 'auto' + tag-suffix: '-gpu-vulkan-buun-llama-cpp' + builder-base-image: 'quay.io/go-skynet/ci-cache:base-grpc-vulkan-arm64' + runs-on: 'ubuntu-24.04-arm' + base-image: "ubuntu:24.04" + skip-drivers: 'false' + backend: "buun-llama-cpp" + dockerfile: "./backend/Dockerfile.buun-llama-cpp" + context: "./" + ubuntu-version: '2404' + # Stablediffusion-ggml - build-type: 'vulkan' cuda-major-version: "" cuda-minor-version: "" diff --git a/.github/workflows/test-extra.yml b/.github/workflows/test-extra.yml index 96eb57155..b0c22fadb 100644 --- a/.github/workflows/test-extra.yml +++ b/.github/workflows/test-extra.yml @@ -33,6 +33,7 @@ jobs: llama-cpp: ${{ steps.detect.outputs.llama-cpp }} ik-llama-cpp: ${{ steps.detect.outputs.ik-llama-cpp }} turboquant: ${{ steps.detect.outputs.turboquant }} + buun-llama-cpp: ${{ steps.detect.outputs['buun-llama-cpp'] }} vllm: ${{ steps.detect.outputs.vllm }} sglang: ${{ steps.detect.outputs.sglang }} acestep-cpp: ${{ steps.detect.outputs.acestep-cpp }} @@ -716,6 +717,30 @@ jobs: - name: Build turboquant backend image and run gRPC e2e tests run: | make test-extra-backend-turboquant + tests-buun-llama-cpp-grpc: + needs: detect-changes + if: needs.detect-changes.outputs['buun-llama-cpp'] == 'true' || needs.detect-changes.outputs.run-all == 'true' + runs-on: ubuntu-latest + timeout-minutes: 90 + steps: + - name: Clone + uses: actions/checkout@v6 + with: + submodules: true + - name: Setup Go + uses: actions/setup-go@v5 + with: + go-version: '1.25.4' + # Exercises the buun-llama-cpp (fork-of-a-fork) backend with the + # fork-specific TurboQuant/TCQ KV-cache types. BACKEND_TEST_CACHE_TYPE_V + # is set to turbo3 so the test round-trips through the fork's KV + # allow-list — picking a stock llama.cpp type would only re-test the + # shared code path. DFlash speculative decoding is not exercised here + # because the one known public target/drafter pair (Qwen3.5-27B) is too + # large for CI. + - name: Build buun-llama-cpp backend image and run gRPC e2e tests + run: | + make test-extra-backend-buun-llama-cpp # tests-vllm-grpc is currently disabled in CI. # # The prebuilt vllm CPU wheel is compiled with AVX-512 VNNI/BF16 diff --git a/scripts/lib/backend-filter.mjs b/scripts/lib/backend-filter.mjs index c5f7aab52..5a45b1f63 100644 --- a/scripts/lib/backend-filter.mjs +++ b/scripts/lib/backend-filter.mjs @@ -70,6 +70,11 @@ export function inferBackendPath(item) { // via a thin wrapper Makefile. Changes to either dir should retrigger it. return `backend/cpp/turboquant/`; } + if (item.dockerfile.endsWith("buun-llama-cpp")) { + // buun-llama-cpp is a llama.cpp fork that reuses backend/cpp/llama-cpp + // sources via a thin wrapper Makefile. Changes to either dir retrigger it. + return `backend/cpp/buun-llama-cpp/`; + } if (item.dockerfile.endsWith("bonsai")) { // bonsai is a llama.cpp fork that reuses backend/cpp/llama-cpp sources // via a thin wrapper Makefile. Changes to either dir should retrigger it. @@ -144,7 +149,7 @@ export function backendChanged(backend, pathPrefix, changedFiles) { // Fork backends reuse backend/cpp/llama-cpp sources via thin wrappers; // changes to either directory must retrigger their pipelines. - return (backend === "turboquant" || backend === "bonsai") && + return (backend === "turboquant" || backend === "buun-llama-cpp" || backend === "bonsai") && changedFiles.some(file => file.startsWith("backend/cpp/llama-cpp/")); }