mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 12:34:43 -04:00
fix(llama-cpp): let parallel:1 in the model options win over LLAMACPP_PARALLEL (#12426)
The environment fallback was applied whenever n_parallel was still 1 after option parsing. An explicit `parallel: 1` in the model YAML is indistinguishable from the default that way, so it was replaced by LLAMACPP_PARALLEL. The docs say options in the YAML take precedence over environment variables; a single model could not be forced to one slot while the global variable was set. Track whether the options set the slot count and resolve it in a small helper (parallel_params.h): option first, then LLAMACPP_PARALLEL, then 1. The helper gets a standalone unit test picked up by `make test-backend-cpp`. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
1 parent
dd19ee8912
commit
6aa7b9b871
6 files changed
+77
-18
No files matched your search
@@ -126,6 +126,11 @@ if(LLAMA_GRPC_BUILD_TESTS)
|
||||
target_compile_features(thread_params_test PRIVATE cxx_std_17)
|
||||
add_test(NAME thread_params_test COMMAND thread_params_test)
|
||||
|
||||
add_executable(parallel_params_test parallel_params_test.cpp parallel_params.h)
|
||||
target_include_directories(parallel_params_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
target_compile_features(parallel_params_test PRIVATE cxx_std_17)
|
||||
add_test(NAME parallel_params_test COMMAND parallel_params_test)
|
||||
|
||||
add_executable(model_load_error_test model_load_error_test.cpp model_load_error.h)
|
||||
target_include_directories(model_load_error_test PRIVATE ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
target_compile_features(model_load_error_test PRIVATE cxx_std_17)
|
||||
|
||||
@@ -55,6 +55,7 @@
|
||||
#include "llama_compat.h" // fork-skew switches, generated by prepare.sh
|
||||
#include "model_load_error.h"
|
||||
#include "thread_params.h"
|
||||
#include "parallel_params.h"
|
||||
#include "message_content.h"
|
||||
#include "passthrough_options.h"
|
||||
#include "stream_peer.h"
|
||||
@@ -545,6 +546,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
// slot state in host RAM and the backend grows without bound.
|
||||
// Initialize n_parallel to 1 by default (can be overridden by options)
|
||||
params.n_parallel = 1;
|
||||
// Set only when the model options carry parallel/n_parallel, so an
|
||||
// explicit 1 can be told apart from the default.
|
||||
std::optional<int> parallel_option;
|
||||
// Initialize grpc_servers to empty (can be overridden by options)
|
||||
std::string grpc_servers_option = "";
|
||||
|
||||
@@ -664,12 +668,9 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
|
||||
if (optval != NULL) {
|
||||
try {
|
||||
params.n_parallel = std::stoi(optval_str);
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
parallel_option = std::stoi(optval_str);
|
||||
} catch (const std::exception& e) {
|
||||
// If conversion fails, keep default value (1)
|
||||
// If conversion fails, fall back to the environment/default
|
||||
}
|
||||
}
|
||||
} else if (!strcmp(optname, "grpc_servers") || !strcmp(optname, "rpc_servers")) {
|
||||
@@ -1238,19 +1239,12 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
}
|
||||
}
|
||||
|
||||
// Set params.n_parallel from environment variable if not set via options (fallback)
|
||||
if (params.n_parallel == 1) {
|
||||
const char *env_parallel = std::getenv("LLAMACPP_PARALLEL");
|
||||
if (env_parallel != NULL) {
|
||||
try {
|
||||
params.n_parallel = std::stoi(env_parallel);
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
} catch (const std::exception& e) {
|
||||
// If conversion fails, keep default value (1)
|
||||
}
|
||||
}
|
||||
// The model options win over LLAMACPP_PARALLEL, including an explicit
|
||||
// parallel:1 (previously indistinguishable from the default and replaced
|
||||
// by the environment value).
|
||||
params.n_parallel = llama_grpc::resolve_n_parallel(parallel_option, std::getenv("LLAMACPP_PARALLEL"));
|
||||
if (params.n_parallel > 1) {
|
||||
params.cont_batching = true;
|
||||
}
|
||||
|
||||
// Add RPC devices from option or environment variable (fallback)
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
#pragma once
|
||||
|
||||
#include <optional>
|
||||
#include <string>
|
||||
|
||||
namespace llama_grpc {
|
||||
|
||||
// resolve_n_parallel picks the slot count. A value from the model options
|
||||
// always wins, including an explicit 1: the YAML takes precedence over the
|
||||
// environment, as documented. LLAMACPP_PARALLEL is only the fallback when
|
||||
// the options do not set it; a value that does not parse is ignored.
|
||||
inline int resolve_n_parallel(const std::optional<int>& from_options, const char* env, int fallback = 1) {
|
||||
if (from_options) {
|
||||
return *from_options;
|
||||
}
|
||||
if (env != nullptr) {
|
||||
try {
|
||||
return std::stoi(env);
|
||||
} catch (const std::exception&) {
|
||||
}
|
||||
}
|
||||
return fallback;
|
||||
}
|
||||
|
||||
} // namespace llama_grpc
|
||||
@@ -0,0 +1,29 @@
|
||||
#include "parallel_params.h"
|
||||
|
||||
#include <cstdio>
|
||||
|
||||
int main() {
|
||||
// parallel:1 in the model options must not be replaced by the environment.
|
||||
if (llama_grpc::resolve_n_parallel(1, "4") != 1) {
|
||||
std::fprintf(stderr, "explicit parallel:1 was overwritten by LLAMACPP_PARALLEL\n");
|
||||
return 1;
|
||||
}
|
||||
if (llama_grpc::resolve_n_parallel(8, "4") != 8) {
|
||||
std::fprintf(stderr, "explicit parallel was overwritten by LLAMACPP_PARALLEL\n");
|
||||
return 1;
|
||||
}
|
||||
// Without an option the environment applies, otherwise the default.
|
||||
if (llama_grpc::resolve_n_parallel(std::nullopt, "4") != 4) {
|
||||
std::fprintf(stderr, "LLAMACPP_PARALLEL was not used as fallback\n");
|
||||
return 1;
|
||||
}
|
||||
if (llama_grpc::resolve_n_parallel(std::nullopt, nullptr) != 1) {
|
||||
std::fprintf(stderr, "default slot count is not 1\n");
|
||||
return 1;
|
||||
}
|
||||
if (llama_grpc::resolve_n_parallel(std::nullopt, "many") != 1) {
|
||||
std::fprintf(stderr, "unparsable LLAMACPP_PARALLEL was not ignored\n");
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -63,6 +63,10 @@ cp -r tts_request_options_test.cpp llama.cpp/tools/grpc-server/
|
||||
# Thread-count default normalization and its standalone regression test.
|
||||
cp -r thread_params.h llama.cpp/tools/grpc-server/
|
||||
cp -r thread_params_test.cpp llama.cpp/tools/grpc-server/
|
||||
# Slot-count resolution (option over LLAMACPP_PARALLEL) and its standalone
|
||||
# regression test.
|
||||
cp -r parallel_params.h llama.cpp/tools/grpc-server/
|
||||
cp -r parallel_params_test.cpp llama.cpp/tools/grpc-server/
|
||||
# Parent-death watcher (included by grpc-server.cpp) and its standalone unit
|
||||
# test (run via backend/cpp/run-unit-tests.sh; also buildable under ctest).
|
||||
cp -r parent_watch.h llama.cpp/tools/grpc-server/
|
||||
|
||||
Reference in new issue
Block a user