mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 12:34:43 -04:00
fix(llama-cpp): let parallel:1 in the model options win over LLAMACPP_PARALLEL (#12426)
The environment fallback was applied whenever n_parallel was still 1 after option parsing. An explicit `parallel: 1` in the model YAML is indistinguishable from the default that way, so it was replaced by LLAMACPP_PARALLEL. The docs say options in the YAML take precedence over environment variables; a single model could not be forced to one slot while the global variable was set. Track whether the options set the slot count and resolve it in a small helper (parallel_params.h): option first, then LLAMACPP_PARALLEL, then 1. The helper gets a standalone unit test picked up by `make test-backend-cpp`. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
1 parent
dd19ee8912
commit
6aa7b9b871
6 files changed
+77
-18
No files matched your search
@@ -55,6 +55,7 @@
|
||||
#include "llama_compat.h" // fork-skew switches, generated by prepare.sh
|
||||
#include "model_load_error.h"
|
||||
#include "thread_params.h"
|
||||
#include "parallel_params.h"
|
||||
#include "message_content.h"
|
||||
#include "passthrough_options.h"
|
||||
#include "stream_peer.h"
|
||||
@@ -545,6 +546,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
// slot state in host RAM and the backend grows without bound.
|
||||
// Initialize n_parallel to 1 by default (can be overridden by options)
|
||||
params.n_parallel = 1;
|
||||
// Set only when the model options carry parallel/n_parallel, so an
|
||||
// explicit 1 can be told apart from the default.
|
||||
std::optional<int> parallel_option;
|
||||
// Initialize grpc_servers to empty (can be overridden by options)
|
||||
std::string grpc_servers_option = "";
|
||||
|
||||
@@ -664,12 +668,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
|
||||
if (optval != NULL) {
|
||||
try {
|
||||
params.n_parallel = std::stoi(optval_str);
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
parallel_option = std::stoi(optval_str);
|
||||
} catch (const std::exception& e) {
|
||||
// If conversion fails, keep default value (1)
|
||||
// If conversion fails, fall back to the environment/default
|
||||
}
|
||||
}
|
||||
} else if (!strcmp(optname, "grpc_servers") || !strcmp(optname, "rpc_servers")) {
|
||||
@@ -1238,19 +1239,12 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
}
|
||||
}
|
||||
|
||||
// Set params.n_parallel from environment variable if not set via options (fallback)
|
||||
if (params.n_parallel == 1) {
|
||||
const char *env_parallel = std::getenv("LLAMACPP_PARALLEL");
|
||||
if (env_parallel != NULL) {
|
||||
try {
|
||||
params.n_parallel = std::stoi(env_parallel);
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
} catch (const std::exception& e) {
|
||||
// If conversion fails, keep default value (1)
|
||||
}
|
||||
}
|
||||
// The model options win over LLAMACPP_PARALLEL, including an explicit
|
||||
// parallel:1 (previously indistinguishable from the default and replaced
|
||||
// by the environment value).
|
||||
params.n_parallel = llama_grpc::resolve_n_parallel(parallel_option, std::getenv("LLAMACPP_PARALLEL"));
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
|
||||
// Add RPC devices from option or environment variable (fallback)
|
||||
|
||||
Reference in new issue
Block a user