diff --git a/backend/cpp/llama-cpp/CMakeLists.txt b/backend/cpp/llama-cpp/CMakeLists.txt index 50982b45d..3c195c37d 100644 --- a/backend/cpp/llama-cpp/CMakeLists.txt +++ b/backend/cpp/llama-cpp/CMakeLists.txt @@ -126,6 +126,11 @@ if(LLAMA_GRPC_BUILD_TESTS) target_compile_features(thread_params_test PRIVATE cxx_std_17) add_test(NAME thread_params_test COMMAND thread_params_test) + add_executable(parallel_params_test parallel_params_test.cpp parallel_params.h) + target_include_directories(parallel_params_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) + target_compile_features(parallel_params_test PRIVATE cxx_std_17) + add_test(NAME parallel_params_test COMMAND parallel_params_test) + add_executable(model_load_error_test model_load_error_test.cpp model_load_error.h) target_include_directories(model_load_error_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) target_compile_features(model_load_error_test PRIVATE cxx_std_17) diff --git a/backend/cpp/llama-cpp/grpc-server.cpp b/backend/cpp/llama-cpp/grpc-server.cpp index 8b206612c..bf1c43d39 100644 --- a/backend/cpp/llama-cpp/grpc-server.cpp +++ b/backend/cpp/llama-cpp/grpc-server.cpp @@ -55,6 +55,7 @@ #include "llama_compat.h" // fork-skew switches, generated by prepare.sh #include "model_load_error.h" #include "thread_params.h" +#include "parallel_params.h" #include "message_content.h" #include "passthrough_options.h" #include "stream_peer.h" @@ -545,6 +546,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt // slot state in host RAM and the backend grows without bound. // Initialize n_parallel to 1 by default (can be overridden by options) params.n_parallel = 1; + // Set only when the model options carry parallel/n_parallel, so an + // explicit 1 can be told apart from the default. + std::optional parallel_option; // Initialize grpc_servers to empty (can be overridden by options) std::string grpc_servers_option = ""; @@ -664,12 +668,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt } else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) { if (optval != NULL) { try { - params.n_parallel = std::stoi(optval_str); - if (params.n_parallel > 1) { - params.cont_batching = true; - } + parallel_option = std::stoi(optval_str); } catch (const std::exception& e) { - // If conversion fails, keep default value (1) + // If conversion fails, fall back to the environment/default } } } else if (!strcmp(optname, "grpc_servers") || !strcmp(optname, "rpc_servers")) { @@ -1238,19 +1239,12 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt } } - // Set params.n_parallel from environment variable if not set via options (fallback) - if (params.n_parallel == 1) { - const char *env_parallel = std::getenv("LLAMACPP_PARALLEL"); - if (env_parallel != NULL) { - try { - params.n_parallel = std::stoi(env_parallel); - if (params.n_parallel > 1) { - params.cont_batching = true; - } - } catch (const std::exception& e) { - // If conversion fails, keep default value (1) - } - } + // The model options win over LLAMACPP_PARALLEL, including an explicit + // parallel:1 (previously indistinguishable from the default and replaced + // by the environment value). + params.n_parallel = llama_grpc::resolve_n_parallel(parallel_option, std::getenv("LLAMACPP_PARALLEL")); + if (params.n_parallel > 1) { + params.cont_batching = true; } // Add RPC devices from option or environment variable (fallback) diff --git a/backend/cpp/llama-cpp/parallel_params.h b/backend/cpp/llama-cpp/parallel_params.h new file mode 100644 index 000000000..de88ba386 --- /dev/null +++ b/backend/cpp/llama-cpp/parallel_params.h @@ -0,0 +1,25 @@ +#pragma once + +#include +#include + +namespace llama_grpc { + +// resolve_n_parallel picks the slot count. A value from the model options +// always wins, including an explicit 1: the YAML takes precedence over the +// environment, as documented. LLAMACPP_PARALLEL is only the fallback when +// the options do not set it; a value that does not parse is ignored. +inline int resolve_n_parallel(const std::optional& from_options, const char* env, int fallback = 1) { + if (from_options) { + return *from_options; + } + if (env != nullptr) { + try { + return std::stoi(env); + } catch (const std::exception&) { + } + } + return fallback; +} + +} // namespace llama_grpc diff --git a/backend/cpp/llama-cpp/parallel_params_test.cpp b/backend/cpp/llama-cpp/parallel_params_test.cpp new file mode 100644 index 000000000..27b55b2fb --- /dev/null +++ b/backend/cpp/llama-cpp/parallel_params_test.cpp @@ -0,0 +1,29 @@ +#include "parallel_params.h" + +#include + +int main() { + // parallel:1 in the model options must not be replaced by the environment. + if (llama_grpc::resolve_n_parallel(1, "4") != 1) { + std::fprintf(stderr, "explicit parallel:1 was overwritten by LLAMACPP_PARALLEL\n"); + return 1; + } + if (llama_grpc::resolve_n_parallel(8, "4") != 8) { + std::fprintf(stderr, "explicit parallel was overwritten by LLAMACPP_PARALLEL\n"); + return 1; + } + // Without an option the environment applies, otherwise the default. + if (llama_grpc::resolve_n_parallel(std::nullopt, "4") != 4) { + std::fprintf(stderr, "LLAMACPP_PARALLEL was not used as fallback\n"); + return 1; + } + if (llama_grpc::resolve_n_parallel(std::nullopt, nullptr) != 1) { + std::fprintf(stderr, "default slot count is not 1\n"); + return 1; + } + if (llama_grpc::resolve_n_parallel(std::nullopt, "many") != 1) { + std::fprintf(stderr, "unparsable LLAMACPP_PARALLEL was not ignored\n"); + return 1; + } + return 0; +} diff --git a/backend/cpp/llama-cpp/prepare.sh b/backend/cpp/llama-cpp/prepare.sh index 8bae396cc..a7fe9ab0f 100644 --- a/backend/cpp/llama-cpp/prepare.sh +++ b/backend/cpp/llama-cpp/prepare.sh @@ -63,6 +63,10 @@ cp -r tts_request_options_test.cpp llama.cpp/tools/grpc-server/ # Thread-count default normalization and its standalone regression test. cp -r thread_params.h llama.cpp/tools/grpc-server/ cp -r thread_params_test.cpp llama.cpp/tools/grpc-server/ +# Slot-count resolution (option over LLAMACPP_PARALLEL) and its standalone +# regression test. +cp -r parallel_params.h llama.cpp/tools/grpc-server/ +cp -r parallel_params_test.cpp llama.cpp/tools/grpc-server/ # Parent-death watcher (included by grpc-server.cpp) and its standalone unit # test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest). cp -r parent_watch.h llama.cpp/tools/grpc-server/ diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 2b4fad656..d1af3b7fb 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -626,6 +626,8 @@ options: **Note:** The `parallel` option can also be set via the `LLAMACPP_PARALLEL` environment variable, and `grpc_servers` can be set via the `LLAMACPP_GRPC_SERVERS` environment variable. Options specified in the YAML file take precedence over environment variables. +An explicit `parallel: 1` (or `n_parallel: 1`) in the model options takes precedence over `LLAMACPP_PARALLEL`, like any other value. The environment variable is only used when neither option is set; if it is missing or not a number, the backend uses one slot. + ##### Hardware auto-tuning (and how to override it) On a detected GPU, LocalAI fills a few performance-relevant defaults the model config leaves unset - a larger physical batch on NVIDIA Blackwell, and a VRAM-scaled `parallel` slot count for concurrent serving. Both are gated on **per-device** VRAM at the model's context: when a large context already fills a single card (e.g. a 27B model with a 200k context across 2×16 GiB), the batch boost and the extra parallel slots are suppressed so they can't tip the tighter GPU into CUDA out-of-memory.