feat(backends): add CachyLLaMA support

Add CachyLLaMA as a GGUF-compatible llama.cpp fork backend with CPU and Vulkan builds on Linux and Metal on Apple silicon. Expose it through model import, document persistent SSD cache options, and wire CI path filtering and dependency updates.

Assisted-by: Codex:gpt-5
This commit is contained in:
localai-org-maint-bot
2026-07-29 18:15:33 +00:00
parent b13c429b3b
commit 955fed341c
21 changed files with 778 additions and 13 deletions

View File

@@ -70,6 +70,10 @@ export function inferBackendPath(item) {
// via a thin wrapper Makefile. Changes to either dir should retrigger it.
return `backend/cpp/turboquant/`;
}
if (item.dockerfile.endsWith("cachyllama")) {
// CachyLLaMA is a llama.cpp fork that reuses the LocalAI gRPC sources.
return `backend/cpp/cachyllama/`;
}
if (item.dockerfile.endsWith("bonsai")) {
// bonsai is a llama.cpp fork that reuses backend/cpp/llama-cpp sources
// via a thin wrapper Makefile. Changes to either dir should retrigger it.
@@ -97,6 +101,9 @@ export function inferBackendPathDarwin(item) {
if (item.backend === "llama-cpp") {
return `backend/cpp/llama-cpp/`;
}
if (item.backend === "cachyllama") {
return `backend/cpp/cachyllama/`;
}
// ds4 is C++ too (built via `make backends/ds4-darwin`); the matrix entry
// carries lang=go for runner/toolchain selection, but the source is C++.
if (item.backend === "ds4") {
@@ -144,7 +151,7 @@ export function backendChanged(backend, pathPrefix, changedFiles) {
// Fork backends reuse backend/cpp/llama-cpp sources via thin wrappers;
// changes to either directory must retrigger their pipelines.
return (backend === "turboquant" || backend === "bonsai") &&
return (backend === "turboquant" || backend === "cachyllama" || backend === "bonsai") &&
changedFiles.some(file => file.startsWith("backend/cpp/llama-cpp/"));
}
@@ -152,12 +159,13 @@ export function backendChanged(backend, pathPrefix, changedFiles) {
// without it is a Python backend (see .github/backend-matrix.yml).
const isDarwinPython = item => !item.lang;
// backend_build_darwin.yml routes llama-cpp, ds4, privacy-filter and audio-cpp
// backend_build_darwin.yml routes llama-cpp, cachyllama, ds4, privacy-filter and audio-cpp
// to their own bespoke make targets; every other lang=go entry goes through
// `make build-darwin-go-backend` -> scripts/build/golang-darwin.sh. Keep this
// set in sync with the `if:` conditions in that workflow.
const DARWIN_BESPOKE_BUILDERS = new Set([
"llama-cpp",
"cachyllama",
"ds4",
"privacy-filter",
"audio-cpp",
@@ -410,6 +418,16 @@ export const SHARED_BUILD_INPUTS = [
linux: never,
darwin: item => item.backend === "llama-cpp",
},
{
matches: file => file === "scripts/build/cachyllama-darwin.sh",
linux: never,
darwin: item => item.backend === "cachyllama",
},
{
matches: file => file === ".docker/cachyllama-compile.sh",
linux: item => item.backend === "cachyllama",
darwin: never,
},
{
matches: file => file === "scripts/build/ds4-darwin.sh",
linux: never,

View File

@@ -71,6 +71,12 @@ const includes = [
"tag-suffix": "-cpu-audio-cpp",
"base-image": "ubuntu:24.04",
},
{
backend: "cachyllama",
dockerfile: "./backend/Dockerfile.cachyllama",
"tag-suffix": "-cachyllama",
"base-image": "ubuntu:24.04",
},
];
const includesDarwin = [
@@ -78,6 +84,7 @@ const includesDarwin = [
{ backend: "mlx", "tag-suffix": "-metal-darwin-arm64-mlx", "build-type": "mps" },
{ backend: "whisper", lang: "go", "tag-suffix": "-metal-darwin-arm64-whisper", "build-type": "metal" },
{ backend: "llama-cpp", lang: "go", "tag-suffix": "-metal-darwin-arm64-llama-cpp", "build-type": "metal" },
{ backend: "cachyllama", lang: "go", "tag-suffix": "-metal-darwin-arm64-cachyllama", "build-type": "metal" },
{ backend: "ds4", lang: "go", "tag-suffix": "-metal-darwin-arm64-ds4", "build-type": "metal" },
];
@@ -190,6 +197,19 @@ test("a bespoke Darwin build script rebuilds only its own backend", () => {
assert.deepEqual(names(filteredDarwin), ["ds4"]);
});
test("the CachyLLaMA Darwin build script rebuilds only CachyLLaMA", () => {
const { filteredDarwin } = run(["scripts/build/cachyllama-darwin.sh"]);
assert.deepEqual(names(filteredDarwin), ["cachyllama"]);
});
test("the CachyLLaMA compile helper rebuilds only Linux CachyLLaMA", () => {
const { filtered, filteredDarwin } = run([".docker/cachyllama-compile.sh"]);
assert.deepEqual(names(filtered), ["cachyllama"]);
assert.deepEqual(filteredDarwin, []);
});
test("an unclassified scripts/build/ file conservatively rebuilds everything", () => {
const { filtered, filteredDarwin } = run([
"scripts/build/package-something-new.sh",
@@ -211,7 +231,17 @@ test("tests for the packaging scripts do not rebuild anything", () => {
test("turboquant still retriggers on llama-cpp source changes", () => {
const { filtered } = run(["backend/cpp/llama-cpp/grpc-server.cpp"]);
assert.deepEqual(names(filtered), ["llama-cpp", "turboquant"]);
assert.deepEqual(names(filtered), ["cachyllama", "llama-cpp", "turboquant"]);
});
test("cachyllama maps to its wrapper source directory", () => {
assert.equal(
inferBackendPath({
backend: "cachyllama",
dockerfile: "./backend/Dockerfile.cachyllama",
}),
"backend/cpp/cachyllama/"
);
});
// ---------------------------------------------------------------------------