mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 12:34:43 -04:00
The environment fallback was applied whenever n_parallel was still 1 after option parsing. An explicit `parallel: 1` in the model YAML is indistinguishable from the default that way, so it was replaced by LLAMACPP_PARALLEL. The docs say options in the YAML take precedence over environment variables; a single model could not be forced to one slot while the global variable was set. Track whether the options set the slot count and resolve it in a small helper (parallel_params.h): option first, then LLAMACPP_PARALLEL, then 1. The helper gets a standalone unit test picked up by `make test-backend-cpp`. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
26 lines
723 B
C++
26 lines
723 B
C++
#pragma once
|
|
|
|
#include <optional>
|
|
#include <string>
|
|
|
|
namespace llama_grpc {
|
|
|
|
// resolve_n_parallel picks the slot count. A value from the model options
|
|
// always wins, including an explicit 1: the YAML takes precedence over the
|
|
// environment, as documented. LLAMACPP_PARALLEL is only the fallback when
|
|
// the options do not set it; a value that does not parse is ignored.
|
|
inline int resolve_n_parallel(const std::optional<int>& from_options, const char* env, int fallback = 1) {
|
|
if (from_options) {
|
|
return *from_options;
|
|
}
|
|
if (env != nullptr) {
|
|
try {
|
|
return std::stoi(env);
|
|
} catch (const std::exception&) {
|
|
}
|
|
}
|
|
return fallback;
|
|
}
|
|
|
|
} // namespace llama_grpc
|