fix(llama-cpp): let parallel:1 in the model options win over LLAMACPP_PARALLEL (#12426)

The environment fallback was applied whenever n_parallel was still 1
after option parsing. An explicit `parallel: 1` in the model YAML is
indistinguishable from the default that way, so it was replaced by
LLAMACPP_PARALLEL. The docs say options in the YAML take precedence
over environment variables; a single model could not be forced to one
slot while the global variable was set.

Track whether the options set the slot count and resolve it in a small
helper (parallel_params.h): option first, then LLAMACPP_PARALLEL, then
1. The helper gets a standalone unit test picked up by
`make test-backend-cpp`.

Assisted-by: Claude:claude-opus-5-5

Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
Stefan Walcz authored and GitHub committed 2026-10-02 09:17:44 +02:00
1 parent dd19ee8912
commit 6aa7b9b871
6 files changed
+77 -18

No files matched your search

+5
View File
@@ -126,6 +126,11 @@ if(LLAMA_GRPC_BUILD_TESTS)
target_compile_features(thread_params_test PRIVATE cxx_std_17)
add_test(NAME thread_params_test COMMAND thread_params_test)
add_executable(parallel_params_test parallel_params_test.cpp parallel_params.h)
target_include_directories(parallel_params_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR})
target_compile_features(parallel_params_test PRIVATE cxx_std_17)
add_test(NAME parallel_params_test COMMAND parallel_params_test)
add_executable(model_load_error_test model_load_error_test.cpp model_load_error.h)
target_include_directories(model_load_error_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR})
target_compile_features(model_load_error_test PRIVATE cxx_std_17)
+12 -18
View File
@@ -55,6 +55,7 @@
#include "llama_compat.h" // fork-skew switches, generated by prepare.sh
#include "model_load_error.h"
#include "thread_params.h"
#include "parallel_params.h"
#include "message_content.h"
#include "passthrough_options.h"
#include "stream_peer.h"
@@ -545,6 +546,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
// slot state in host RAM and the backend grows without bound.
// Initialize n_parallel to 1 by default (can be overridden by options)
params.n_parallel = 1;
// Set only when the model options carry parallel/n_parallel, so an
// explicit 1 can be told apart from the default.
std::optional<int> parallel_option;
// Initialize grpc_servers to empty (can be overridden by options)
std::string grpc_servers_option = "";
@@ -664,12 +668,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
if (optval != NULL) {
try {
params.n_parallel = std::stoi(optval_str);
if (params.n_parallel > 1) {
params.cont_batching = true;
}
parallel_option = std::stoi(optval_str);
} catch (const std::exception& e) {
// If conversion fails, keep default value (1)
// If conversion fails, fall back to the environment/default
}
}
} else if (!strcmp(optname, "grpc_servers") || !strcmp(optname, "rpc_servers")) {
@@ -1238,19 +1239,12 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
}
}
// Set params.n_parallel from environment variable if not set via options (fallback)
if (params.n_parallel == 1) {
const char *env_parallel = std::getenv("LLAMACPP_PARALLEL");
if (env_parallel != NULL) {
try {
params.n_parallel = std::stoi(env_parallel);
if (params.n_parallel > 1) {
params.cont_batching = true;
}
} catch (const std::exception& e) {
// If conversion fails, keep default value (1)
}
}
// The model options win over LLAMACPP_PARALLEL, including an explicit
// parallel:1 (previously indistinguishable from the default and replaced
// by the environment value).
params.n_parallel = llama_grpc::resolve_n_parallel(parallel_option, std::getenv("LLAMACPP_PARALLEL"));
if (params.n_parallel > 1) {
params.cont_batching = true;
}
// Add RPC devices from option or environment variable (fallback)
+25
View File
@@ -0,0 +1,25 @@
#pragma once
#include <optional>
#include <string>
namespace llama_grpc {
// resolve_n_parallel picks the slot count. A value from the model options
// always wins, including an explicit 1: the YAML takes precedence over the
// environment, as documented. LLAMACPP_PARALLEL is only the fallback when
// the options do not set it; a value that does not parse is ignored.
inline int resolve_n_parallel(const std::optional<int>& from_options, const char* env, int fallback = 1) {
if (from_options) {
return *from_options;
}
if (env != nullptr) {
try {
return std::stoi(env);
} catch (const std::exception&) {
}
}
return fallback;
}
} // namespace llama_grpc
@@ -0,0 +1,29 @@
#include "parallel_params.h"
#include <cstdio>
int main() {
// parallel:1 in the model options must not be replaced by the environment.
if (llama_grpc::resolve_n_parallel(1, "4") != 1) {
std::fprintf(stderr, "explicit parallel:1 was overwritten by LLAMACPP_PARALLEL\n");
return 1;
}
if (llama_grpc::resolve_n_parallel(8, "4") != 8) {
std::fprintf(stderr, "explicit parallel was overwritten by LLAMACPP_PARALLEL\n");
return 1;
}
// Without an option the environment applies, otherwise the default.
if (llama_grpc::resolve_n_parallel(std::nullopt, "4") != 4) {
std::fprintf(stderr, "LLAMACPP_PARALLEL was not used as fallback\n");
return 1;
}
if (llama_grpc::resolve_n_parallel(std::nullopt, nullptr) != 1) {
std::fprintf(stderr, "default slot count is not 1\n");
return 1;
}
if (llama_grpc::resolve_n_parallel(std::nullopt, "many") != 1) {
std::fprintf(stderr, "unparsable LLAMACPP_PARALLEL was not ignored\n");
return 1;
}
return 0;
}
+4
View File
@@ -63,6 +63,10 @@ cp -r tts_request_options_test.cpp llama.cpp/tools/grpc-server/
# Thread-count default normalization and its standalone regression test.
cp -r thread_params.h llama.cpp/tools/grpc-server/
cp -r thread_params_test.cpp llama.cpp/tools/grpc-server/
# Slot-count resolution (option over LLAMACPP_PARALLEL) and its standalone
# regression test.
cp -r parallel_params.h llama.cpp/tools/grpc-server/
cp -r parallel_params_test.cpp llama.cpp/tools/grpc-server/
# Parent-death watcher (included by grpc-server.cpp) and its standalone unit
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
cp -r parent_watch.h llama.cpp/tools/grpc-server/
+2
View File
@@ -626,6 +626,8 @@ options:
**Note:** The `parallel` option can also be set via the `LLAMACPP_PARALLEL` environment variable, and `grpc_servers` can be set via the `LLAMACPP_GRPC_SERVERS` environment variable. Options specified in the YAML file take precedence over environment variables.
An explicit `parallel: 1` (or `n_parallel: 1`) in the model options takes precedence over `LLAMACPP_PARALLEL`, like any other value. The environment variable is only used when neither option is set; if it is missing or not a number, the backend uses one slot.
##### Hardware auto-tuning (and how to override it)
On a detected GPU, LocalAI fills a few performance-relevant defaults the model config leaves unset - a larger physical batch on NVIDIA Blackwell, and a VRAM-scaled `parallel` slot count for concurrent serving. Both are gated on **per-device** VRAM at the model's context: when a large context already fills a single card (e.g. a 27B model with a 200k context across 2×16 GiB), the batch boost and the extra parallel slots are suppressed so they can't tip the tighter GPU into CUDA out-of-memory.