fix(llama-cpp): let parallel:1 in the model options win over LLAMACPP_PARALLEL (#12426)

The environment fallback was applied whenever n_parallel was still 1
after option parsing. An explicit `parallel: 1` in the model YAML is
indistinguishable from the default that way, so it was replaced by
LLAMACPP_PARALLEL. The docs say options in the YAML take precedence
over environment variables; a single model could not be forced to one
slot while the global variable was set.

Track whether the options set the slot count and resolve it in a small
helper (parallel_params.h): option first, then LLAMACPP_PARALLEL, then
1. The helper gets a standalone unit test picked up by
`make test-backend-cpp`.

Assisted-by: Claude:claude-opus-5-5

Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
Stefan Walcz authored and GitHub committed 2026-10-02 09:17:44 +02:00
1 parent dd19ee8912
commit 6aa7b9b871
6 files changed
+77 -18

No files matched your search

+12 -18
View File
@@ -55,6 +55,7 @@
#include "llama_compat.h" // fork-skew switches, generated by prepare.sh
#include "model_load_error.h"
#include "thread_params.h"
#include "parallel_params.h"
#include "message_content.h"
#include "passthrough_options.h"
#include "stream_peer.h"
@@ -545,6 +546,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
// slot state in host RAM and the backend grows without bound.
// Initialize n_parallel to 1 by default (can be overridden by options)
params.n_parallel = 1;
// Set only when the model options carry parallel/n_parallel, so an
// explicit 1 can be told apart from the default.
std::optional<int> parallel_option;
// Initialize grpc_servers to empty (can be overridden by options)
std::string grpc_servers_option = "";
@@ -664,12 +668,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
if (optval != NULL) {
try {
params.n_parallel = std::stoi(optval_str);
if (params.n_parallel > 1) {
params.cont_batching = true;
}
parallel_option = std::stoi(optval_str);
} catch (const std::exception& e) {
// If conversion fails, keep default value (1)
// If conversion fails, fall back to the environment/default
}
}
} else if (!strcmp(optname, "grpc_servers") || !strcmp(optname, "rpc_servers")) {
@@ -1238,19 +1239,12 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
}
}
// Set params.n_parallel from environment variable if not set via options (fallback)
if (params.n_parallel == 1) {
const char *env_parallel = std::getenv("LLAMACPP_PARALLEL");
if (env_parallel != NULL) {
try {
params.n_parallel = std::stoi(env_parallel);
if (params.n_parallel > 1) {
params.cont_batching = true;
}
} catch (const std::exception& e) {
// If conversion fails, keep default value (1)
}
}
// The model options win over LLAMACPP_PARALLEL, including an explicit
// parallel:1 (previously indistinguishable from the default and replaced
// by the environment value).
params.n_parallel = llama_grpc::resolve_n_parallel(parallel_option, std::getenv("LLAMACPP_PARALLEL"));
if (params.n_parallel > 1) {
params.cont_batching = true;
}
// Add RPC devices from option or environment variable (fallback)