Files
LocalAI/backend/cpp/llama-cpp/prepare.sh
localai-org-maint-bot 40bae01e40 fix(llama-cpp): preserve GPU layers during option passthrough
Stage the negative GPU-layer sentinels expected by the upstream argument parser, then restore LocalAI resolved values unless a passthrough flag explicitly overrides them. This avoids the parser assertion that terminated the backend for any generic option.

Assisted-by: Codex:gpt-5
2026-07-29 14:10:03 +00:00

65 lines
2.7 KiB
Bash

#!/bin/bash
set -e
## Patches
## Apply patches from the `patches` directory. Runs under set -e so a
## rejected patch aborts the build here, loudly, instead of surfacing later
## as a confusing compile error. A missing or empty patches dir is a no-op.
if [ -d "patches" ]; then
for patch in $(ls patches); do
echo "Applying patch $patch"
patch -d llama.cpp/ -p1 < patches/$patch
done
fi
for file in $(ls llama.cpp/tools/server/); do
cp -rfv llama.cpp/tools/server/$file llama.cpp/tools/grpc-server/
done
cp -r CMakeLists.txt llama.cpp/tools/grpc-server/
cp -r grpc-server.cpp llama.cpp/tools/grpc-server/
# Shared message-reconstruction helpers (included by grpc-server.cpp) and their
# unit test (compiled only when -DLLAMA_GRPC_BUILD_TESTS=ON).
cp -r message_content.h llama.cpp/tools/grpc-server/
cp -r message_content_test.cpp llama.cpp/tools/grpc-server/
# Generic passthrough parser staging and its standalone regression test.
cp -r passthrough_options.h llama.cpp/tools/grpc-server/
cp -r passthrough_options_test.cpp llama.cpp/tools/grpc-server/
# Parent-death watcher (included by grpc-server.cpp) and its standalone unit
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
cp -r parent_watch.h llama.cpp/tools/grpc-server/
cp -r parent_watch_test.cpp llama.cpp/tools/grpc-server/
cp -rfv llama.cpp/vendor/nlohmann/json.hpp llama.cpp/tools/grpc-server/
cp -rfv llama.cpp/vendor/cpp-httplib/httplib.h llama.cpp/tools/grpc-server/
## Fork-skew probe. Upstream folded common_params::use_mmap / use_mlock /
## use_direct_io into a single `load_mode` enum (ggml-org/llama.cpp#20834).
## turboquant and bonsai compile this very same grpc-server.cpp against forks
## that branched before that change, so the field set is decided from the
## checkout in front of us rather than from a per-fork build flag: the flavor
## targets disagree on whether they forward CMAKE_ARGS or EXTRA_CMAKE_ARGS, and
## probing heals itself the moment a fork rebases past the refactor.
if grep -q "LLAMA_LOAD_MODE_MMAP" llama.cpp/include/llama.h; then
echo "==> llama.cpp carries the load-mode enum, using common_params::load_mode"
LEGACY_LOAD_MODE=0
else
echo "==> llama.cpp predates the load-mode enum, using the legacy mmap/mlock/direct-io booleans"
LEGACY_LOAD_MODE=1
fi
cat > llama.cpp/tools/grpc-server/llama_compat.h <<EOF
// Generated by backend/cpp/llama-cpp/prepare.sh. Do not edit.
#pragma once
#define LOCALAI_LEGACY_LOAD_MODE ${LEGACY_LOAD_MODE}
EOF
set +e
if grep -q "grpc-server" llama.cpp/tools/CMakeLists.txt; then
echo "grpc-server already added"
else
echo "add_subdirectory(grpc-server)" >> llama.cpp/tools/CMakeLists.txt
fi
set -e