mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-13 14:56:11 -04:00
* ⬆️ Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(llama-cpp): follow upstream MTMD APIs The dependency update adds MTMD initialization options to prompt and bitmap helpers. The gRPC adapter now passes the server options through each affected path. The update also replaces the per-layer MoE regex helper. Preparation probes both APIs because older forks still reuse this adapter. Assisted-by: Codex:gpt-5 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
90 lines
3.6 KiB
Bash
90 lines
3.6 KiB
Bash
#!/bin/bash
|
|
|
|
set -e
|
|
|
|
|
|
## Patches
|
|
|
|
## Apply patches from the `patches` directory. Runs under set -e so a
|
|
## rejected patch aborts the build here, loudly, instead of surfacing later
|
|
## as a confusing compile error. A missing or empty patches dir is a no-op.
|
|
if [ -d "patches" ]; then
|
|
for patch in $(ls patches); do
|
|
echo "Applying patch $patch"
|
|
patch -d llama.cpp/ -p1 < patches/$patch
|
|
done
|
|
fi
|
|
|
|
for file in $(ls llama.cpp/tools/server/); do
|
|
cp -rfv llama.cpp/tools/server/$file llama.cpp/tools/grpc-server/
|
|
done
|
|
|
|
cp -r CMakeLists.txt llama.cpp/tools/grpc-server/
|
|
cp -r grpc-server.cpp llama.cpp/tools/grpc-server/
|
|
# Shared message-reconstruction helpers (included by grpc-server.cpp) and their
|
|
# unit test (compiled only when -DLLAMA_GRPC_BUILD_TESTS=ON).
|
|
cp -r message_content.h llama.cpp/tools/grpc-server/
|
|
cp -r message_content_test.cpp llama.cpp/tools/grpc-server/
|
|
# Generic passthrough parser staging and its standalone regression test.
|
|
cp -r passthrough_options.h llama.cpp/tools/grpc-server/
|
|
cp -r passthrough_options_test.cpp llama.cpp/tools/grpc-server/
|
|
# TTS request validation (included by grpc-server.cpp) and its standalone
|
|
# regression test.
|
|
cp -r tts_request_options.h llama.cpp/tools/grpc-server/
|
|
cp -r tts_request_options_test.cpp llama.cpp/tools/grpc-server/
|
|
# Thread-count default normalization and its standalone regression test.
|
|
cp -r thread_params.h llama.cpp/tools/grpc-server/
|
|
cp -r thread_params_test.cpp llama.cpp/tools/grpc-server/
|
|
# Parent-death watcher (included by grpc-server.cpp) and its standalone unit
|
|
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
|
|
cp -r parent_watch.h llama.cpp/tools/grpc-server/
|
|
cp -r parent_watch_test.cpp llama.cpp/tools/grpc-server/
|
|
cp -rfv llama.cpp/vendor/nlohmann/json.hpp llama.cpp/tools/grpc-server/
|
|
cp -rfv llama.cpp/vendor/cpp-httplib/httplib.h llama.cpp/tools/grpc-server/
|
|
|
|
## Fork-skew probe. Upstream folded common_params::use_mmap / use_mlock /
|
|
## use_direct_io into a single `load_mode` enum (ggml-org/llama.cpp#20834).
|
|
## turboquant and bonsai compile this very same grpc-server.cpp against forks
|
|
## that branched before that change, so the field set is decided from the
|
|
## checkout in front of us rather than from a per-fork build flag: the flavor
|
|
## targets disagree on whether they forward CMAKE_ARGS or EXTRA_CMAKE_ARGS, and
|
|
## probing heals itself the moment a fork rebases past the refactor.
|
|
if grep -q "LLAMA_LOAD_MODE_MMAP" llama.cpp/include/llama.h; then
|
|
echo "==> llama.cpp carries the load-mode enum, using common_params::load_mode"
|
|
LEGACY_LOAD_MODE=0
|
|
else
|
|
echo "==> llama.cpp predates the load-mode enum, using the legacy mmap/mlock/direct-io booleans"
|
|
LEGACY_LOAD_MODE=1
|
|
fi
|
|
if grep -q "server_metrics metrics;" llama.cpp/tools/server/server-task.h; then
|
|
HAS_SERVER_METRICS=1
|
|
else
|
|
HAS_SERVER_METRICS=0
|
|
fi
|
|
if grep -q "mtmd_helper_init_opt" llama.cpp/tools/mtmd/mtmd-helper.h; then
|
|
HAS_MTMD_INIT_OPT=1
|
|
else
|
|
HAS_MTMD_INIT_OPT=0
|
|
fi
|
|
if grep -q "llm_add_n_cpu_ffn_overrides" llama.cpp/common/common.h; then
|
|
HAS_N_CPU_FFN_HELPER=1
|
|
else
|
|
HAS_N_CPU_FFN_HELPER=0
|
|
fi
|
|
cat > llama.cpp/tools/grpc-server/llama_compat.h <<EOF
|
|
// Generated by backend/cpp/llama-cpp/prepare.sh. Do not edit.
|
|
#pragma once
|
|
#define LOCALAI_LEGACY_LOAD_MODE ${LEGACY_LOAD_MODE}
|
|
#define LOCALAI_HAS_SERVER_METRICS ${HAS_SERVER_METRICS}
|
|
#define LOCALAI_HAS_MTMD_INIT_OPT ${HAS_MTMD_INIT_OPT}
|
|
#define LOCALAI_HAS_N_CPU_FFN_HELPER ${HAS_N_CPU_FFN_HELPER}
|
|
EOF
|
|
|
|
set +e
|
|
if grep -q "grpc-server" llama.cpp/tools/CMakeLists.txt; then
|
|
echo "grpc-server already added"
|
|
else
|
|
echo "add_subdirectory(grpc-server)" >> llama.cpp/tools/CMakeLists.txt
|
|
fi
|
|
set -e
|