mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-13 14:56:11 -04:00
grpc::ServerWriter::Write() returns false once the peer is gone, and PredictStream ignored that result at every call site. The handler kept pulling decoded tokens and writing them into a dead stream, so the llama.cpp slot stayed busy until the generation ended on its own terms. A model configured with max_tokens 0 and a large context ends on its own terms only at the context limit. On a 35B model at ~41 t/s a 120k context is about fifty minutes, and a slot held that long is a slot every other request for that model queues behind. Two abandoned requests were enough to make a node with free VRAM and a healthy control plane serve nothing: new requests timed out waiting for a slot, each timeout abandoned another generation, and the node fell further behind the longer it ran. Track the peer instead. The first failed write retires it for good, since a stream never recovers, and the RPC's own cancellation flag folds into the same predicate so the loop has one condition to test. Returning early is what frees the slot: ~server_response_reader() posts SERVER_TASK_TYPE_CANCEL for whatever is still decoding. TTSStream already checked Write(); this brings PredictStream in line. Cancellation stays cooperative and is checked between decoded results, so a batch already in flight may finish before the request stops. Assisted-by: Claude:claude-opus-5
122 lines
5.3 KiB
Bash
122 lines
5.3 KiB
Bash
#!/bin/bash
|
|
|
|
set -e
|
|
|
|
|
|
## Patches
|
|
|
|
## Apply patches from the `patches` directory. Runs under set -e so a
|
|
## rejected patch aborts the build here, loudly, instead of surfacing later
|
|
## as a confusing compile error. A missing or empty patches dir is a no-op.
|
|
if [ -d "patches" ]; then
|
|
for patch in $(ls patches); do
|
|
echo "Applying patch $patch"
|
|
patch -d llama.cpp/ -p1 < patches/$patch
|
|
done
|
|
fi
|
|
|
|
## Apple RDMA link fixup.
|
|
|
|
## ggml-rpc hands Apple's librdma to the linker with
|
|
## target_link_options(ggml-rpc PRIVATE "LINKER:-weak_library,..."). Link options are not
|
|
## a usage requirement of a static library, so in our BUILD_SHARED_LIBS=OFF build the flag
|
|
## dies with libggml-rpc.a and every ibv_* symbol transport-apple.cpp reaches for comes out
|
|
## undefined when grpc-server and ggml-rpc-server link. Re-declare the same weak link as
|
|
## INTERFACE so it travels to whoever links the static library.
|
|
##
|
|
## Guarded on the marker so a second prepare.sh over the same checkout is a no-op, and on
|
|
## GGML_RPC_RDMA_APPLE so forks that branched before the Apple RDMA transport (turboquant,
|
|
## bonsai) are left alone.
|
|
RPC_CMAKE=llama.cpp/ggml/src/ggml-rpc/CMakeLists.txt
|
|
if [ -f "$RPC_CMAKE" ] && grep -q "GGML_RPC_RDMA_APPLE" "$RPC_CMAKE" && ! grep -q "LOCALAI_RDMA_IFACE" "$RPC_CMAKE"; then
|
|
echo "==> ggml-rpc carries the Apple RDMA transport, re-declaring its weak librdma link as INTERFACE"
|
|
cat >> "$RPC_CMAKE" <<'EOF'
|
|
|
|
# LOCALAI_RDMA_IFACE: added by backend/cpp/llama-cpp/prepare.sh
|
|
if (GGML_RPC_RDMA AND APPLE AND NOT BUILD_SHARED_LIBS)
|
|
target_link_options(ggml-rpc INTERFACE "LINKER:-weak_library,${RDMA_LIB}")
|
|
endif()
|
|
EOF
|
|
fi
|
|
|
|
for file in $(ls llama.cpp/tools/server/); do
|
|
cp -rfv llama.cpp/tools/server/$file llama.cpp/tools/grpc-server/
|
|
done
|
|
|
|
cp -r CMakeLists.txt llama.cpp/tools/grpc-server/
|
|
cp -r grpc-server.cpp llama.cpp/tools/grpc-server/
|
|
# Model-load diagnostics (included by grpc-server.cpp) and their standalone
|
|
# regression test.
|
|
cp -r model_load_error.h llama.cpp/tools/grpc-server/
|
|
cp -r model_load_error_test.cpp llama.cpp/tools/grpc-server/
|
|
# Shared message-reconstruction helpers (included by grpc-server.cpp) and their
|
|
# unit test (compiled only when -DLLAMA_GRPC_BUILD_TESTS=ON).
|
|
cp -r message_content.h llama.cpp/tools/grpc-server/
|
|
cp -r message_content_test.cpp llama.cpp/tools/grpc-server/
|
|
# Generic passthrough parser staging and its standalone regression test.
|
|
cp -r passthrough_options.h llama.cpp/tools/grpc-server/
|
|
cp -r passthrough_options_test.cpp llama.cpp/tools/grpc-server/
|
|
# TTS request validation (included by grpc-server.cpp) and its standalone
|
|
# regression test.
|
|
cp -r tts_request_options.h llama.cpp/tools/grpc-server/
|
|
cp -r tts_request_options_test.cpp llama.cpp/tools/grpc-server/
|
|
# Thread-count default normalization and its standalone regression test.
|
|
cp -r thread_params.h llama.cpp/tools/grpc-server/
|
|
cp -r thread_params_test.cpp llama.cpp/tools/grpc-server/
|
|
# Parent-death watcher (included by grpc-server.cpp) and its standalone unit
|
|
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
|
|
cp -r parent_watch.h llama.cpp/tools/grpc-server/
|
|
cp -r parent_watch_test.cpp llama.cpp/tools/grpc-server/
|
|
# Dead-stream tracker (included by grpc-server.cpp) and its standalone unit
|
|
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
|
|
cp -r stream_peer.h llama.cpp/tools/grpc-server/
|
|
cp -r stream_peer_test.cpp llama.cpp/tools/grpc-server/
|
|
cp -rfv llama.cpp/vendor/nlohmann/json.hpp llama.cpp/tools/grpc-server/
|
|
cp -rfv llama.cpp/vendor/cpp-httplib/httplib.h llama.cpp/tools/grpc-server/
|
|
|
|
## Fork-skew probe. Upstream folded common_params::use_mmap / use_mlock /
|
|
## use_direct_io into a single `load_mode` enum (ggml-org/llama.cpp#20834).
|
|
## turboquant and bonsai compile this very same grpc-server.cpp against forks
|
|
## that branched before that change, so the field set is decided from the
|
|
## checkout in front of us rather than from a per-fork build flag: the flavor
|
|
## targets disagree on whether they forward CMAKE_ARGS or EXTRA_CMAKE_ARGS, and
|
|
## probing heals itself the moment a fork rebases past the refactor.
|
|
if grep -q "LLAMA_LOAD_MODE_MMAP" llama.cpp/include/llama.h; then
|
|
echo "==> llama.cpp carries the load-mode enum, using common_params::load_mode"
|
|
LEGACY_LOAD_MODE=0
|
|
else
|
|
echo "==> llama.cpp predates the load-mode enum, using the legacy mmap/mlock/direct-io booleans"
|
|
LEGACY_LOAD_MODE=1
|
|
fi
|
|
if grep -q "server_metrics metrics;" llama.cpp/tools/server/server-task.h; then
|
|
HAS_SERVER_METRICS=1
|
|
else
|
|
HAS_SERVER_METRICS=0
|
|
fi
|
|
if grep -q "mtmd_helper_init_opt" llama.cpp/tools/mtmd/mtmd-helper.h; then
|
|
HAS_MTMD_INIT_OPT=1
|
|
else
|
|
HAS_MTMD_INIT_OPT=0
|
|
fi
|
|
if grep -q "llm_add_n_cpu_ffn_overrides" llama.cpp/common/common.h; then
|
|
HAS_N_CPU_FFN_HELPER=1
|
|
else
|
|
HAS_N_CPU_FFN_HELPER=0
|
|
fi
|
|
cat > llama.cpp/tools/grpc-server/llama_compat.h <<EOF
|
|
// Generated by backend/cpp/llama-cpp/prepare.sh. Do not edit.
|
|
#pragma once
|
|
#define LOCALAI_LEGACY_LOAD_MODE ${LEGACY_LOAD_MODE}
|
|
#define LOCALAI_HAS_SERVER_METRICS ${HAS_SERVER_METRICS}
|
|
#define LOCALAI_HAS_MTMD_INIT_OPT ${HAS_MTMD_INIT_OPT}
|
|
#define LOCALAI_HAS_N_CPU_FFN_HELPER ${HAS_N_CPU_FFN_HELPER}
|
|
EOF
|
|
|
|
set +e
|
|
if grep -q "grpc-server" llama.cpp/tools/CMakeLists.txt; then
|
|
echo "grpc-server already added"
|
|
else
|
|
echo "add_subdirectory(grpc-server)" >> llama.cpp/tools/CMakeLists.txt
|
|
fi
|
|
set -e
|