Heartbeat
{timeAgo(node.last_heartbeat)}
diff --git a/core/services/modeladmin/config.go b/core/services/modeladmin/config.go
index 2cadfcaed..f85f29ee9 100644
--- a/core/services/modeladmin/config.go
+++ b/core/services/modeladmin/config.go
@@ -93,7 +93,7 @@ func (s *ConfigService) GetConfig(_ context.Context, name string) (*ConfigView,
if configPath == "" {
return nil, ErrConfigFileMissing
}
- if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
+ if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
data, err := os.ReadFile(configPath)
@@ -137,7 +137,7 @@ func (s *ConfigService) patchConfig(ctx context.Context, name string, patch map[
return nil, fmt.Errorf("%w: PATCH cannot rename model %q to %q; use the model edit endpoint", ErrInvalidConfig, name, patchedName)
}
configPath := cfg.GetModelConfigFile()
- if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
+ if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
diskYAML, err := os.ReadFile(configPath)
@@ -289,7 +289,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
configPath := existing.GetModelConfigFile()
modelsPath := s.modelsPath()
- if err := utils.VerifyPath(configPath, modelsPath); err != nil {
+ if err := utils.VerifyResolvedPath(configPath, modelsPath); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
@@ -304,7 +304,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
}
newConfigPath := filepath.Join(modelsPath, req.Name+".yaml")
paths = append(paths, newConfigPath, filepath.Join(modelsPath, gallery.GalleryFileName(name)), filepath.Join(modelsPath, gallery.GalleryFileName(req.Name)))
- if err := utils.VerifyPath(newConfigPath, modelsPath); err != nil {
+ if err := utils.VerifyPath(req.Name+".yaml", modelsPath); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
if _, err := os.Stat(newConfigPath); err == nil {
diff --git a/core/services/modeladmin/config_path_trust_test.go b/core/services/modeladmin/config_path_trust_test.go
new file mode 100644
index 000000000..5b375b699
--- /dev/null
+++ b/core/services/modeladmin/config_path_trust_test.go
@@ -0,0 +1,58 @@
+package modeladmin
+
+import (
+ "context"
+ "os"
+ "path/filepath"
+
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+// A model config can be loaded from outside the models directory (for
+// example with --config-file). The admin mutations write the config file
+// back, so they must refuse a file outside the models directory rather than
+// write wherever the loader found it.
+var _ = Describe("ConfigService config file containment", func() {
+ var (
+ svc *ConfigService
+ ctx context.Context
+ outside string
+ orig []byte
+ )
+
+ BeforeEach(func() {
+ svc, _ = newTestService()
+ ctx = context.Background()
+ outside = filepath.Join(GinkgoT().TempDir(), "external.yaml")
+ orig = []byte("name: external\nbackend: llama-cpp\n")
+ Expect(os.WriteFile(outside, orig, 0o644)).To(Succeed())
+ Expect(svc.Loader.ReadModelConfig(outside, svc.AppConfig.ToConfigLoaderOptions()...)).To(Succeed())
+ cfg, ok := svc.Loader.GetModelConfig("external")
+ Expect(ok).To(BeTrue())
+ Expect(cfg.GetModelConfigFile()).To(Equal(outside))
+ })
+
+ It("refuses to pin a model whose config file is outside the models directory", func() {
+ _, err := svc.TogglePinned(ctx, "external", ActionPin, nil)
+ Expect(err).To(MatchError(ErrPathNotTrusted))
+ Expect(os.ReadFile(outside)).To(Equal(orig))
+ })
+
+ It("refuses to toggle the state of such a model", func() {
+ _, err := svc.ToggleState(ctx, "external", ActionDisable)
+ Expect(err).To(MatchError(ErrPathNotTrusted))
+ Expect(os.ReadFile(outside)).To(Equal(orig))
+ })
+
+ It("refuses to patch such a model", func() {
+ _, err := svc.PatchConfig(ctx, "external", map[string]any{"context_size": 4096})
+ Expect(err).To(MatchError(ErrPathNotTrusted))
+ Expect(os.ReadFile(outside)).To(Equal(orig))
+ })
+
+ It("refuses to read such a model's config", func() {
+ _, err := svc.GetConfig(ctx, "external")
+ Expect(err).To(MatchError(ErrPathNotTrusted))
+ })
+})
diff --git a/core/services/modeladmin/pinned.go b/core/services/modeladmin/pinned.go
index b9ef45724..17c4a2f39 100644
--- a/core/services/modeladmin/pinned.go
+++ b/core/services/modeladmin/pinned.go
@@ -29,7 +29,7 @@ func (s *ConfigService) TogglePinned(_ context.Context, name string, action Acti
if configPath == "" {
return nil, ErrConfigFileMissing
}
- if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
+ if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
if err := mutateYAMLBoolFlag(configPath, "pinned", action == ActionPin); err != nil {
diff --git a/core/services/modeladmin/state.go b/core/services/modeladmin/state.go
index d37368d9d..cee6d7855 100644
--- a/core/services/modeladmin/state.go
+++ b/core/services/modeladmin/state.go
@@ -49,7 +49,7 @@ func (s *ConfigService) toggleState(ctx context.Context, name string, action Act
if configPath == "" {
return nil, ErrConfigFileMissing
}
- if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
+ if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
var result *ToggleResult
diff --git a/core/services/nodes/declared_files.go b/core/services/nodes/declared_files.go
new file mode 100644
index 000000000..4df3b000a
--- /dev/null
+++ b/core/services/nodes/declared_files.go
@@ -0,0 +1,78 @@
+package nodes
+
+import (
+ "os"
+ "path/filepath"
+ "strings"
+
+ pb "github.com/mudler/LocalAI/pkg/grpc/proto"
+ "github.com/mudler/xlog"
+)
+
+// declaredExtraFiles returns the files the model's install declared that the
+// path fields of opts do not already stage: neither named by a field nor
+// inside a directory a field names. It must run on the local paths, before
+// staging rewrites the fields to remote ones.
+func (r *SmartRouter) declaredExtraFiles(trackingKey string, opts *pb.ModelOptions) []string {
+ if r.modelFiles == nil || opts == nil || trackingKey == "" {
+ return nil
+ }
+ covered := append([]string{
+ opts.ModelFile, opts.MMProj, opts.LoraAdapter, opts.DraftModel,
+ opts.CLIPModel, opts.Tokenizer, opts.AudioPath, opts.LoraBase,
+ }, opts.LoraAdapters...)
+
+ seen := map[string]struct{}{}
+ var extra []string
+ for _, p := range r.modelFiles(trackingKey) {
+ p = filepath.Clean(p)
+ if _, dup := seen[p]; dup || coveredByField(p, covered) {
+ continue
+ }
+ seen[p] = struct{}{}
+ extra = append(extra, p)
+ }
+ return extra
+}
+
+func coveredByField(path string, fields []string) bool {
+ for _, f := range fields {
+ if f == "" {
+ continue
+ }
+ f = filepath.Clean(f)
+ if path == f || strings.HasPrefix(path, f+string(filepath.Separator)) {
+ return true
+ }
+ }
+ return false
+}
+
+// existingFiles drops declared files that are not on the frontend. An install
+// can declare files that are gone by load time (an archive unpacked and then
+// removed, say), so a missing one is not a reason to refuse the load; the
+// backend reports it if it really needed it.
+func existingFiles(paths []string, nodeName, trackingKey string) []string {
+ out := paths[:0:0]
+ for _, p := range paths {
+ if _, err := os.Stat(p); err != nil {
+ xlog.Warn("Skipping staging for declared model file that is not on the frontend", "path", p, "node", nodeName, "model", trackingKey, "error", err)
+ continue
+ }
+ out = append(out, p)
+ }
+ return out
+}
+
+// stagingPayloadBytes totals the on-disk size of everything staging uploads
+// for a model: the path fields plus the declared files they do not cover. The
+// first shard of a split GGUF can be a few MB of metadata while the weights
+// sit in the others, so sizing the fields alone starves the load budget and
+// the disk-headroom check.
+func (r *SmartRouter) stagingPayloadBytes(trackingKey string, opts *pb.ModelOptions) int64 {
+ total := modelPayloadBytes(opts)
+ for _, p := range r.declaredExtraFiles(trackingKey, opts) {
+ total += pathBytes(p)
+ }
+ return total
+}
diff --git a/core/services/nodes/registry.go b/core/services/nodes/registry.go
index 4dbda5e2d..b4896a653 100644
--- a/core/services/nodes/registry.go
+++ b/core/services/nodes/registry.go
@@ -124,6 +124,11 @@ type BackendNode struct {
// worker's re-registration value does not clobber it (mirrors
// MaxReplicasPerModelManuallySet).
VRAMBudgetManuallySet bool `gorm:"column:vram_budget_manually_set;default:false" json:"vram_budget_manually_set"`
+ // Version is the LocalAI build version reported by the worker at
+ // registration. Empty for workers registered before this field existed.
+ Version string `gorm:"column:version;size:64" json:"version,omitempty"`
+ // Commit is the git commit hash the worker binary was built from.
+ Commit string `gorm:"column:commit;size:64" json:"commit,omitempty"`
APIKeyID string `gorm:"size:36" json:"-"` // auto-provisioned API key ID (for cleanup)
AuthUserID string `gorm:"size:36" json:"-"` // auto-provisioned user ID (for cleanup)
LastHeartbeat time.Time `gorm:"column:last_heartbeat" json:"last_heartbeat"`
diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go
index c1cc846bd..49065c7b2 100644
--- a/core/services/nodes/router.go
+++ b/core/services/nodes/router.go
@@ -73,6 +73,12 @@ type SmartRouterOptions struct {
// nil disables the exclusion. Deliberate teardown (UnloadModel, admin
// endpoints, node drain) is unaffected.
PinnedResolver PinnedModelResolver
+ // ModelFiles, when set, returns the absolute local paths of every file a
+ // model's install declared (gallery `files:`, config `download_files`).
+ // The path fields of a load request name only what the backend opens
+ // first; this is how staging learns about the rest, such as the other
+ // shards of a split GGUF. nil stages the path fields alone.
+ ModelFiles func(modelName string) []string
// PrefixProvider, when set, enables prefix-cache-aware routing: requests
// carrying a prompt prefix chain (distributedhdr.PrefixChain) are biased
// toward the node that already holds the longest matching prefix, subject
@@ -189,6 +195,9 @@ type SmartRouter struct {
// pinnedResolver feeds the eviction paths the set of pinned model names
// (see SmartRouterOptions.PinnedResolver). nil disables the exclusion.
pinnedResolver PinnedModelResolver
+ // modelFiles resolves a model's declared files (see
+ // SmartRouterOptions.ModelFiles). nil stages the path fields alone.
+ modelFiles func(modelName string) []string
// prefixProvider is the prefix-cache routing seam (nil disables it; see
// SmartRouterOptions.PrefixProvider). prefixConfig holds the global policy
// and thresholds.
@@ -283,6 +292,7 @@ func NewSmartRouter(registry ModelRouter, opts SmartRouterOptions) *SmartRouter
stagingTracker: NewStagingTracker(),
conflictResolver: opts.ConflictResolver,
pinnedResolver: opts.PinnedResolver,
+ modelFiles: opts.ModelFiles,
probeCache: newProbeCache(probeCacheTTL),
prefixProvider: opts.PrefixProvider,
prefixConfig: opts.PrefixConfig,
@@ -425,7 +435,7 @@ func (r *SmartRouter) scheduleAndLoad(ctx context.Context, backendType, tracking
// Size the remote load budget BEFORE staging: stageModelFiles rewrites the
// path fields to their remote equivalents on a clone, and only the local
// paths can be stat'ed here.
- payloadBytes := modelPayloadBytes(modelOpts)
+ payloadBytes := r.stagingPayloadBytes(trackingKey, modelOpts)
loadTimeout := r.loadTimeoutFor(payloadBytes)
// Pre-stage model files via FileStager before loading
@@ -1367,7 +1377,7 @@ func (r *SmartRouter) narrowByDiskHeadroom(ctx context.Context, modelID string,
return candidateNodeIDs, nil
}
- requiredDisk := DiskRequirementFor(modelPayloadBytes(modelOpts))
+ requiredDisk := DiskRequirementFor(r.stagingPayloadBytes(modelID, modelOpts))
diskCandidates, diskErr := r.registry.NarrowByDiskHeadroom(ctx, candidateNodeIDs, requiredDisk)
// The check runs even when disabled. "Disabled" means do not BLOCK, not do
@@ -1568,6 +1578,10 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
localModelDir = filepath.Dir(opts.ModelFile)
}
+ // Resolved before the path fields are rewritten to remote paths below,
+ // since that is what tells which declared files the fields already cover.
+ declared := existingFiles(r.declaredExtraFiles(trackingKey, opts), node.Name, trackingKey)
+
// keyMapper generates storage keys namespaced under trackingKey, preserving
// subdirectory structure relative to frontendModelsDir. This ensures:
// 1. All files for a model land in one directory on the worker for clean deletion
@@ -1614,6 +1628,7 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
totalFiles++
}
}
+ totalFiles += len(declared)
// Start tracking staging progress
r.stagingTracker.Start(trackingKey, node.Name, totalFiles)
@@ -1757,6 +1772,21 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
}
}
+ for _, localPath := range declared {
+ fileIdx++
+ fileName := filepath.Base(localPath)
+ stageCtx := r.withStagingCallback(ctx, trackingKey, fileName, fileIdx, totalFiles)
+
+ xlog.Info("Staging declared model file", "model", trackingKey, "node", node.Name, "file", fileName, "fileIndex", fileIdx, "totalFiles", totalFiles)
+ if _, err := r.fileStager.EnsureRemote(stageCtx, node.ID, localPath, keyMapper.Key(localPath)); err != nil {
+ // The install declared it, so the backend may read it: loading
+ // without it fails later with a less useful error.
+ xlog.Error("Failed to stage declared model file for remote node", "node", node.Name, "path", localPath, "error", err)
+ return nil, fmt.Errorf("staging declared model file %s: %w", localPath, err)
+ }
+ r.stagingTracker.FileComplete(trackingKey, fileIdx, totalFiles)
+ }
+
// Stage file paths referenced in generic Options (key:value pairs where values
// are file paths). Options stay as relative paths — backends resolve them via ModelPath.
for _, options := range [][]string{opts.Options, opts.Overrides} {
diff --git a/core/services/nodes/router_declared_files_stage_test.go b/core/services/nodes/router_declared_files_stage_test.go
new file mode 100644
index 000000000..e1c00e8c5
--- /dev/null
+++ b/core/services/nodes/router_declared_files_stage_test.go
@@ -0,0 +1,126 @@
+package nodes
+
+import (
+ "context"
+ "os"
+ "path/filepath"
+
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+
+ pb "github.com/mudler/LocalAI/pkg/grpc/proto"
+)
+
+// A model's config names only the file the backend opens first, but its
+// install can declare more that the backend reads by itself: llama.cpp opens
+// the "-0000N-of-0000M" shards of a split GGUF from the directory of the first
+// one. The worker has no view of the frontend's models directory, so every
+// declared file must be staged, or the load fails with "failed to load GGUF
+// split".
+var _ = Describe("stageModelFiles declared model files", func() {
+ var (
+ stager *fakeFileStager
+ router *SmartRouter
+ node *BackendNode
+ modelDir string
+ shards []string
+ mmproj string
+ declared map[string][]string
+ )
+
+ BeforeEach(func() {
+ stager = &fakeFileStager{}
+ declared = map[string][]string{}
+ router = &SmartRouter{
+ fileStager: stager,
+ stagingTracker: NewStagingTracker(),
+ modelFiles: func(name string) []string { return declared[name] },
+ }
+ node = &BackendNode{ID: "node-1", Name: "node-1", Address: "10.0.0.1:50051"}
+ root := GinkgoT().TempDir()
+ modelDir = filepath.Join(root, "llama-cpp", "models", "big")
+ Expect(os.MkdirAll(modelDir, 0o755)).To(Succeed())
+
+ shards = nil
+ for _, name := range []string{
+ "Big-Q4_K_M-00001-of-00003.gguf",
+ "Big-Q4_K_M-00002-of-00003.gguf",
+ "Big-Q4_K_M-00003-of-00003.gguf",
+ } {
+ p := filepath.Join(modelDir, name)
+ Expect(os.WriteFile(p, []byte("shard "+name), 0o644)).To(Succeed())
+ shards = append(shards, p)
+ }
+ mmproj = filepath.Join(root, "llama-cpp", "mmproj", "big", "mmproj.gguf")
+ Expect(os.MkdirAll(filepath.Dir(mmproj), 0o755)).To(Succeed())
+ Expect(os.WriteFile(mmproj, []byte("mmproj"), 0o644)).To(Succeed())
+ })
+
+ opts := func() *pb.ModelOptions {
+ return &pb.ModelOptions{
+ Model: "llama-cpp/models/big/Big-Q4_K_M-00001-of-00003.gguf",
+ ModelFile: shards[0],
+ MMProj: mmproj,
+ }
+ }
+
+ stagedPaths := func() []string {
+ out := make([]string, 0, len(stager.ensureCalls))
+ for _, c := range stager.ensureCalls {
+ out = append(out, c.localPath)
+ }
+ return out
+ }
+
+ It("stages every declared file once, beside the ones the config names", func() {
+ declared["big"] = append(append([]string{}, shards...), mmproj)
+
+ staged, err := router.stageModelFiles(context.Background(), node, opts(), "big")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
+
+ // llama.cpp derives the other shards' paths from the first one, so
+ // they must land in the same remote directory.
+ for _, c := range stager.ensureCalls {
+ if c.localPath != mmproj {
+ Expect(filepath.Dir(c.key)).To(Equal(filepath.Dir(stager.ensureCalls[0].key)))
+ }
+ }
+ Expect(staged.ModelFile).To(Equal("/remote/" + stager.ensureCalls[0].key))
+ })
+
+ It("sizes declared files for the load budget and disk check", func() {
+ declared["big"] = append(append([]string{}, shards...), mmproj)
+
+ var want int64
+ for _, p := range append(append([]string{}, shards...), mmproj) {
+ fi, err := os.Stat(p)
+ Expect(err).ToNot(HaveOccurred())
+ want += fi.Size()
+ }
+ Expect(router.stagingPayloadBytes("big", opts())).To(Equal(want))
+ })
+
+ It("skips a declared file that is missing locally instead of failing", func() {
+ declared["big"] = append(append([]string{}, shards...), filepath.Join(modelDir, "gone.bin"))
+
+ _, err := router.stageModelFiles(context.Background(), node, opts(), "big")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
+ })
+
+ It("does not stage a declared file twice when a directory field covers it", func() {
+ declared["dir"] = []string{shards[1]}
+
+ _, err := router.stageModelFiles(context.Background(), node,
+ &pb.ModelOptions{Model: "llama-cpp/models/big", ModelFile: modelDir}, "dir")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(stagedPaths()).To(ConsistOf(shards[0], shards[1], shards[2]))
+ })
+
+ It("stages only the named files for a model that declares none", func() {
+ _, err := router.stageModelFiles(context.Background(), node, opts(), "handwritten")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj))
+ })
+})
diff --git a/core/services/worker/registration.go b/core/services/worker/registration.go
index c3ac109a9..4cb4806ab 100644
--- a/core/services/worker/registration.go
+++ b/core/services/worker/registration.go
@@ -8,6 +8,7 @@ import (
"strconv"
"strings"
+ "github.com/mudler/LocalAI/internal"
"github.com/mudler/LocalAI/pkg/system"
"github.com/mudler/LocalAI/pkg/xsysinfo"
"github.com/mudler/xlog"
@@ -154,6 +155,8 @@ func (cfg *Config) registrationBody() map[string]any {
"gpu_compute_capability": gpuComputeCap,
"capability": capability,
"max_replicas_per_model": maxReplicas,
+ "version": internal.Version,
+ "commit": internal.Commit,
}
// Report free space on the filesystem that backs the MODELS directory.
diff --git a/core/startup/model_preload.go b/core/startup/model_preload.go
index 4f3bb1683..8dd167828 100644
--- a/core/startup/model_preload.go
+++ b/core/startup/model_preload.go
@@ -41,7 +41,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
// Check if it's a model gallery, or print a warning
e, found := installModel(ctx, galleries, backendGalleries, url, systemState, modelLoader, downloadStatus, enforceScan, autoloadBackendGalleries, requireBackendIntegrity, installOptions...)
if e != nil && found {
- xlog.Error("[startup] failed installing model", "error", err, "model", url)
+ xlog.Error("[startup] failed installing model", "error", e, "model", url)
err = errors.Join(err, e)
} else if !found {
xlog.Debug("[startup] model not found in the gallery", "model", url)
@@ -54,7 +54,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
modelConfig, discoverErr := importers.DiscoverModelConfig(url, json.RawMessage{})
if discoverErr != nil {
xlog.Error("[startup] failed to discover model config", "error", discoverErr, "model", url)
- err = errors.Join(discoverErr, fmt.Errorf("failed to discover model config: %w", err))
+ err = errors.Join(err, fmt.Errorf("failed to discover model config: %w", discoverErr))
continue
}
diff --git a/docker-compose.yaml b/docker-compose.yaml
index 82b3c18b6..50eccaaf4 100644
--- a/docker-compose.yaml
+++ b/docker-compose.yaml
@@ -41,7 +41,7 @@ services:
# Here we can specify a list of models to run (see quickstart https://localai.io/basics/getting_started/#running-models )
# or an URL pointing to a YAML configuration file, for example:
# - https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
- - phi-2
+ - phi-2-chat
# For NVIDIA GPU support with CDI (recommended for NVIDIA Container Toolkit 1.14+):
# Uncomment the following deploy section and use driver: nvidia.com/gpu.
# Include `utility` in capabilities so nvidia-smi / NVML are available —
diff --git a/docs/content/advanced/model-configuration.md b/docs/content/advanced/model-configuration.md
index 1c2ade4de..5cfa74ccd 100644
--- a/docs/content/advanced/model-configuration.md
+++ b/docs/content/advanced/model-configuration.md
@@ -74,6 +74,8 @@ When using `--models-config-file`, you can define multiple models as a list:
backend: llama-cpp
```
+LocalAI changes only config files that are inside the models directory. If the file from `--models-config-file` is outside the models directory, you cannot view, edit, pin, enable or disable its models from the web UI or the model admin API. Edit the file directly, then restart LocalAI.
+
## Core Configuration Fields
### Basic Model Settings
diff --git a/docs/content/features/GPU-acceleration.md b/docs/content/features/GPU-acceleration.md
index 207860ec6..fb939a47a 100644
--- a/docs/content/features/GPU-acceleration.md
+++ b/docs/content/features/GPU-acceleration.md
@@ -243,7 +243,7 @@ The devices in the following list have been tested with `hipblas` images.
1. Check your GPU LLVM target is compatible with the version of ROCm. This can be found in the [LLVM Docs](https://llvm.org/docs/AMDGPUUsage.html).
2. Check which ROCm version is compatible with your LLVM target and your chosen OS (pay special attention to supported kernel versions). See the [ROCm compatibility matrix](https://rocm.docs.amd.com/en/latest/compatibility/compatibility-matrix.html).
-3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/how-to/native-install/index.html).
+3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/install/install-methods/package-manager-index.html).
4. Deploy. Yes it's that easy.
#### Setup Example (Docker/containerd)
diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md
index 6fe983de2..12c82fc15 100644
--- a/docs/content/features/distributed-mode.md
+++ b/docs/content/features/distributed-mode.md
@@ -740,6 +740,19 @@ Set `LOCALAI_DISTRIBUTED_SHARED_MODELS=true` (or `--distributed-shared-models`)
This flag is a contract you assert: all nodes must mount identical paths. Leave it off (the default) when workers have independent models directories - the frontend stages files to them over HTTP (or S3) as described above.
+### Which files are staged
+
+The frontend stages the files that the model config names (`parameters.model`, `mmproj`, draft model, LoRA adapters and similar fields). It also stages every other file that the model declares:
+
+- The `files:` of the gallery entry or `/import-model` import that installed the model. LocalAI records these in `._gallery_
.yaml` next to the model config.
+- The `download_files:` of the model config.
+
+A backend can read files that the config does not name. For example, llama.cpp opens all shards of a split GGUF (`-00002-of-00004.gguf` and the rest) from the directory of the first shard. The worker cannot see the frontend's models directory, so it gets only the files that the frontend stages.
+
+If you write a model config by hand and the model has files like these, list them under `download_files:`. If you do not, the worker gets only the first shard and the load fails with `failed to load GGUF split`.
+
+The file sizes used for the load deadline and for the disk headroom check include all of these files.
+
### Model artifact staging
For managed Hugging Face artifacts, the controller resolves the repository and
diff --git a/docs/content/features/distributed_inferencing.md b/docs/content/features/distributed_inferencing.md
index c14c01072..72cfb6c78 100644
--- a/docs/content/features/distributed_inferencing.md
+++ b/docs/content/features/distributed_inferencing.md
@@ -98,7 +98,7 @@ LLAMACPP_GRPC_SERVERS="address1:port,address2:port" local-ai run
```
The workload on the LocalAI server will then be distributed across the specified nodes.
-Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggerganov/llama.cpp/blob/master/examples/rpc/README.md), which is compatible with LocalAI.
+Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggml-org/llama.cpp/blob/master/tools/rpc/README.md), which is compatible with LocalAI.
## Manual example (worker)
diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md
index 0bee9c890..09063fdd7 100644
--- a/docs/content/features/model-gallery.md
+++ b/docs/content/features/model-gallery.md
@@ -373,7 +373,7 @@ curl $LOCALAI/models/apply -H "Content-Type: application/json" -d '{
where:
- `localai` is the repository. It is optional and can be omitted. If the repository is omitted LocalAI will search the model by name in all the repositories. In the case the same model name is present in both galleries the first match wins.
- `bert-embeddings` is the model name in the gallery
- (read its [config here](https://github.com/mudler/LocalAI/tree/master/gallery/blob/main/bert-embeddings.yaml)).
+ (read its [config here](https://github.com/mudler/LocalAI/blob/master/gallery/index.yaml)).
### Model variants
diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md
index ba9f5266f..d06269cf7 100644
--- a/docs/content/features/text-generation.md
+++ b/docs/content/features/text-generation.md
@@ -587,7 +587,7 @@ The `llama.cpp` backend supports additional configuration options that can be sp
|--------|------|-------------|---------|
| `use_jinja` or `jinja` | boolean | Enable Jinja2 template processing for chat templates. When enabled, the backend uses Jinja2-based chat templates from the model for formatting messages. | `use_jinja:true` |
| `context_shift` | boolean | Enable context shifting, which allows the model to dynamically adjust context window usage. | `context_shift:true` |
-| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `-1` (no limit). `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
+| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `8192` MiB (llama.cpp default). `-1` removes the limit. `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
| `parallel` or `n_parallel` | integer | Enable parallel request processing. When set to a value greater than 1, enables continuous batching for handling multiple requests concurrently. | `parallel:4` |
| `grpc_servers` or `rpc_servers` | string | Comma-separated list of gRPC server addresses for distributed inference. Allows distributing workload across multiple llama.cpp workers. | `grpc_servers:localhost:50051,localhost:50052` |
| `fit_params` or `fit` | boolean | Enable auto-adjustment of model/context parameters to fit available device memory. Default: `true`. | `fit_params:true` |
@@ -643,7 +643,7 @@ Agents, coding assistants, and Anthropic/OpenAI-compatible CLIs typically resend
| Setting | Default | Role |
|---|---|---|
-| `cache_ram:N` | `-1` (no limit) | Allocates the host-side prompt cache. `0` disables it. |
+| `cache_ram:N` | `8192` (llama.cpp default) | Allocates the host-side prompt cache. `0` disables it. |
| `kv_unified:true` | `true` | Single unified KV buffer (**prerequisite** for idle-slot saving). |
| `cache_idle_slots:true` | `true` | Persists the idle slot's KV into the prompt cache on task switch. |
@@ -658,6 +658,8 @@ options:
Set `cache_ram:0` to opt out of the prompt cache entirely (saves host RAM at the cost of re-prefilling repeated prompts).
+`cache_ram:-1` removes the limit. With idle-slot saving on, every distinct prompt then leaves its slot state in host RAM, so a workload with many different prompts (classification, ingestion) grows the backend by roughly the KV size of each prompt until the host runs out of memory.
+
#### Reference
- [llama](https://github.com/ggerganov/llama.cpp)
@@ -988,6 +990,41 @@ options:
The full list of registered parsers lives in `sglang.srt.function_call`
and `sglang.srt.parser.reasoning_parser`.
+#### Reasoning defaults and token budgets
+
+Set SGLang reasoning options in the model's `options:` list:
+
+```yaml
+options:
+ - reasoning_parser:qwen3
+ - thinking_budget:512
+ - reasoning_default:on
+engine_args:
+ enable_strict_thinking: true
+```
+
+`thinking_budget` sets a positive integer token budget for reasoning on each request.
+Invalid, zero, and negative values produce a warning and leave the budget unset.
+SGLang requires `engine_args.enable_strict_thinking: true` to enforce the budget.
+LocalAI warns if you configure a budget without that engine option.
+Keep the budget well below the `max_tokens` of your requests: if `max_tokens` is reached first,
+the budget never triggers and the whole reply can be spent on reasoning, leaving the answer empty.
+
+`reasoning_default:on` or `reasoning_default:off` sets the default for LocalAI's tokenizer chat template.
+Request metadata `enable_thinking` set to `"true"` or `"false"` overrides this default.
+An explicit prompt bypasses tokenizer template rendering.
+When no default or request override is set, the template keeps its own behavior.
+
+LocalAI signals required reasoning when the rendered prompt ends with the configured parser's opening reasoning token.
+An explicit output grammar disables this detection.
+Configure a reasoning parser that matches your model.
+
+The backend reads these options when it loads the model.
+`POST /models/reload` rereads model configuration files but does not update options in an already loaded backend.
+Restarting only the backend does not reread configuration files.
+Restart LocalAI after changing these options to reload both the configuration and the backend.
+
+
### vllm.cpp
[vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of
diff --git a/docs/content/getting-started/containers.md b/docs/content/getting-started/containers.md
index 6e6ca9526..d5616b201 100644
--- a/docs/content/getting-started/containers.md
+++ b/docs/content/getting-started/containers.md
@@ -108,6 +108,8 @@ docker run -ti --name local-ai -p 8080:8080 --runtime nvidia --gpus all localai/
## Using Compose
+The repository's `docker-compose.yaml` installs `phi-2-chat` from the model gallery by default. Change its `command` list to select a different gallery model.
+
For a more manageable setup, especially with persistent volumes, use Docker Compose or Podman Compose:
### Using CDI (Container Device Interface) - Recommended for NVIDIA Container Toolkit 1.14+
diff --git a/docs/content/getting-started/customize-model.md b/docs/content/getting-started/customize-model.md
index 751a2e6fa..172e6051e 100644
--- a/docs/content/getting-started/customize-model.md
+++ b/docs/content/getting-started/customize-model.md
@@ -25,17 +25,17 @@ Here's an example to initiate the **phi-2** model:
docker run -p 8080:8080 localai/localai:{{< version >}} https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
```
-You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/embedded/models).
+You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/gallery).
{{% notice tip %}}
-The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/embedded/models](https://github.com/mudler/LocalAI/tree/master/embedded/models). Contributions are welcome; please feel free to submit a Pull Request.
+The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/gallery](https://github.com/mudler/LocalAI/tree/master/gallery). Contributions are welcome; please feel free to submit a Pull Request.
-The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml).
+The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml).
{{% /notice %}}
## Example: Customizing the Prompt Template
-To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml). Alter the fields as needed:
+To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml). Alter the fields as needed:
```yaml
name: phi-2
diff --git a/docs/content/integrations.md b/docs/content/integrations.md
index 8d8bd4104..6fb84a250 100644
--- a/docs/content/integrations.md
+++ b/docs/content/integrations.md
@@ -98,7 +98,7 @@ availability may lag upstream releases.
- [AnythingLLM](https://github.com/Mintplex-Labs/anything-llm)
- [Logseq GPT3 OpenAI plugin](https://github.com/briansunter/logseq-plugin-gpt3-openai)
- [CodeGPT (JetBrains)](https://plugins.jetbrains.com/plugin/21056-codegpt) - Custom OpenAI-compatible endpoints
-- [Wave Terminal](https://docs.waveterm.dev/features/supportedLLMs/localai) - Native LocalAI support
+- [Wave Terminal](https://docs.waveterm.dev/ai-presets) - Native LocalAI support
- [Obsidian BMO Chatbot](https://github.com/longy2k/obsidian-bmo-chatbot)
- [spark](https://github.com/cedriking/spark)
- [openops (Mattermost)](https://github.com/mattermost/openops)
diff --git a/docs/content/reference/compatibility-table.md b/docs/content/reference/compatibility-table.md
index 7e6bfd73d..40c6ae4f3 100644
--- a/docs/content/reference/compatibility-table.md
+++ b/docs/content/reference/compatibility-table.md
@@ -45,7 +45,7 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
| [moonshine](https://github.com/moonshine-ai/moonshine) | Ultra-fast transcription for low-end devices (ONNX) | CPU, CUDA 12/13, Metal |
| [parakeet.cpp](https://github.com/mudler/parakeet.cpp) | C++/GGML port of NVIDIA NeMo Parakeet (tdt/ctc/rnnt/hybrid), with cache-aware streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
| [CrispASR](https://github.com/CrispStrobe/CrispASR) | Unified speech engine (whisper.cpp fork) supporting Parakeet, Canary, and many ASR architectures, plus TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
-| [voxtral](https://github.com/mudler/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
+| [voxtral](https://github.com/antirez/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
| [Qwen3-ASR](https://github.com/QwenLM/Qwen3-ASR) | Qwen3 automatic speech recognition | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
| [NeMo](https://github.com/NVIDIA/NeMo) | NVIDIA NeMo ASR toolkit | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
| [sherpa-onnx](https://k2-fsa.github.io/sherpa/onnx/) | Sherpa-ONNX ASR (Whisper, Paraformer, SenseVoice) and TTS | CPU, CUDA 12, Metal |
@@ -70,10 +70,10 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
| [OmniVoice](https://github.com/ServeurpersoCom/omnivoice.cpp) | Native C++/GGML TTS with voice cloning, voice design, and streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
| [fish-speech](https://github.com/fishaudio/fish-speech) | High-quality TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
| [Pocket TTS](https://github.com/kyutai-labs/pocket-tts) | Lightweight CPU-efficient TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
-| [OuteTTS](https://github.com/OuteAI/outetts) | TTS with custom speaker voices | CPU, CUDA 12 |
+| [OuteTTS](https://github.com/edwko/OuteTTS) | TTS with custom speaker voices | CPU, CUDA 12 |
| [faster-qwen3-tts](https://github.com/andimarafioti/faster-qwen3-tts) | Real-time Qwen3-TTS with CUDA graph capture | CPU, CUDA 12/13, Jetson L4T |
| [NeuTTS Air](https://github.com/neuphonic/neutts-air) | Instant voice cloning, on-device TTS | CPU, CUDA 12, ROCm |
-| [VoxCPM](https://github.com/ModelBest/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
+| [VoxCPM](https://github.com/OpenBMB/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
| [Kitten TTS](https://github.com/KittenML/KittenTTS) | Kitten TTS model | CPU, Metal |
| [Supertonic](https://github.com/supertone-inc/supertonic) | Lightning-fast on-device multilingual TTS via ONNX | CPU |
| [MLX-Audio](https://github.com/Blaizzy/mlx-audio) | Audio models on Apple Silicon | CPU, CUDA 12/13, Metal, Jetson L4T |
diff --git a/gallery/index.yaml b/gallery/index.yaml
index f6fa9361c..c2298226b 100644
--- a/gallery/index.yaml
+++ b/gallery/index.yaml
@@ -297,7 +297,7 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
- sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
+ sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6
- name: "qwopus3.8-27b-flash-v2"
variants:
- model: qwopus3.8-27b-flash-v2-q8
@@ -6420,7 +6420,6 @@
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
-
- name: sharp-spark-x2.5-4b-q5
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -6457,7 +6456,6 @@
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
-
- name: sharp-spark-x2.5-4b-q6
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -6494,7 +6492,6 @@
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
-
- &spark-x2-5-4b
name: "spark-x2.5-4b-q4"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
@@ -9309,7 +9306,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
- sha256: 35026b978de6ee6ff63870d3d68be90ab0797a9a6cd5c83332e1f9c510c6a695
+ sha256: 97602082e8639b1e6660b36de0193861742e035711b865faf76abf54a5a2be09
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
@@ -9399,7 +9396,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
- sha256: 612561952f0698539a133a479e1cf18ffce85bb6b4311a5d05e50d6160970c75
+ sha256: 3efbc83f38ffa48251ff4c072fbbefce63c995df3cba20fd6bf0d8d7ab965cd9
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
@@ -9451,7 +9448,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
- sha256: 7a17aaff5ec81ba34e33d3a8b032a699abf4231d6cbe9930d1c9a159db3e87dc
+ sha256: 27a2edec6f66e585fbf02eabe397f694378474786711362d9f78cd129a0e8313
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
diff --git a/pkg/utils/path.go b/pkg/utils/path.go
index 1ae11d123..16c29de15 100644
--- a/pkg/utils/path.go
+++ b/pkg/utils/path.go
@@ -13,21 +13,35 @@ func ExistsInPath(path string, s string) bool {
}
func InTrustedRoot(path string, trustedRoot string) error {
- for path != "/" {
- path = filepath.Dir(path)
+ for {
+ parent := filepath.Dir(path)
+ // Dir stops changing at "/" for an absolute path and at "." for a
+ // relative one; waiting for "/" alone spins forever on the latter.
+ if parent == path {
+ return fmt.Errorf("path is outside of trusted root")
+ }
+ path = parent
if path == trustedRoot {
return nil
}
}
- return fmt.Errorf("path is outside of trusted root")
}
-// VerifyPath verifies that path is based in basePath.
+// VerifyPath verifies that path, taken relative to basePath, is based in
+// basePath. It joins path onto basePath first, so an absolute path is read as
+// relative to the base as well: give it the untrusted relative name, never a
+// path that has already been joined. For a full path use VerifyResolvedPath.
func VerifyPath(path, basePath string) error {
c := filepath.Clean(filepath.Join(basePath, path))
return InTrustedRoot(c, filepath.Clean(basePath))
}
+// VerifyResolvedPath verifies that path, a full path rather than one relative
+// to basePath, is based in basePath.
+func VerifyResolvedPath(path, basePath string) error {
+ return InTrustedRoot(filepath.Clean(path), filepath.Clean(basePath))
+}
+
// SanitizeFileName sanitizes the given filename
func SanitizeFileName(fileName string) string {
// filepath.Clean to clean the path
diff --git a/pkg/utils/path_test.go b/pkg/utils/path_test.go
index 79c415cd4..e970a45ab 100644
--- a/pkg/utils/path_test.go
+++ b/pkg/utils/path_test.go
@@ -3,6 +3,7 @@ package utils_test
import (
"os"
"path/filepath"
+ "time"
. "github.com/mudler/LocalAI/pkg/utils"
. "github.com/onsi/ginkgo/v2"
@@ -71,6 +72,25 @@ var _ = Describe("utils/path tests", func() {
})
})
+ Describe("VerifyResolvedPath", func() {
+ It("accepts a full path inside the base", func() {
+ Expect(VerifyResolvedPath("/srv/models/a/model.yaml", "/srv/models")).To(Succeed())
+ })
+
+ It("rejects a full path outside the base", func() {
+ // VerifyPath would join this onto the base and accept it.
+ Expect(VerifyResolvedPath("/etc/passwd", "/srv/models")).ToNot(Succeed())
+ })
+
+ It("rejects a joined path that climbed out of the base", func() {
+ Expect(VerifyResolvedPath(filepath.Join("/srv/models", "../other/x"), "/srv/models")).ToNot(Succeed())
+ })
+
+ It("cleans both paths before comparing", func() {
+ Expect(VerifyResolvedPath("/srv/models/./a/../b.yaml", "/srv/models/")).To(Succeed())
+ })
+ })
+
Describe("InTrustedRoot", func() {
It("accepts a strict descendant of the trusted root", func() {
Expect(InTrustedRoot("/srv/models/file", "/srv/models")).To(Succeed())
@@ -93,6 +113,21 @@ var _ = Describe("utils/path tests", func() {
It("rejects an unrelated absolute path", func() {
Expect(InTrustedRoot("/etc/passwd", "/srv/models")).ToNot(Succeed())
})
+
+ It("rejects a relative path outside a relative root instead of looping", func() {
+ // Walking up a relative path ends at ".", never at "/", so the
+ // walk must stop when it stops making progress.
+ done := make(chan error, 1)
+ go func() { done <- InTrustedRoot("x", "models") }()
+ Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
+
+ go func() { done <- VerifyPath("../x", "models") }()
+ Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
+ })
+
+ It("accepts a relative descendant of a relative root", func() {
+ Expect(InTrustedRoot("models/a/file", "models")).To(Succeed())
+ })
})
Describe("SanitizeFileName", func() {
diff --git a/swagger/docs.go b/swagger/docs.go
index c447043ae..0950fdb6c 100644
--- a/swagger/docs.go
+++ b/swagger/docs.go
@@ -3770,6 +3770,90 @@ const docTemplate = `{
}
}
},
+ "/v1/systemone": {
+ "post": {
+ "description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Answer structured-extraction questions over state text.",
+ "parameters": [
+ {
+ "description": "state + questions",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneResponse"
+ }
+ }
+ }
+ }
+ },
+ "/v1/systemone/permute": {
+ "post": {
+ "description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Re-run a choice question under multiple option orders.",
+ "parameters": [
+ {
+ "description": "request + question + n_perm + seed",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOnePermuteRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOnePermuteResponse"
+ }
+ }
+ }
+ }
+ },
+ "/v1/systemone/separate": {
+ "post": {
+ "description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Answer each question in a separate NER pass.",
+ "parameters": [
+ {
+ "description": "state + questions",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneResponse"
+ }
+ }
+ }
+ }
+ },
"/v1/text-to-speech/{voice-id}": {
"post": {
"tags": [
@@ -4093,8 +4177,16 @@ const docTemplate = `{
"config.Gallery": {
"type": "object",
"properties": {
+ "artifact_verification": {
+ "description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
+ "allOf": [
+ {
+ "$ref": "#/definitions/config.GalleryVerification"
+ }
+ ]
+ },
"mirrors": {
- "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
+ "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
"type": "array",
"items": {
"type": "string"
@@ -4129,6 +4221,10 @@ const docTemplate = `{
"not_before": {
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
"type": "string"
+ },
+ "source_repository": {
+ "description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
+ "type": "string"
}
}
},
@@ -7811,12 +7907,42 @@ const docTemplate = `{
"id": {
"type": "string"
},
+ "process": {
+ "description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
+ "allOf": [
+ {
+ "$ref": "#/definitions/schema.SysInfoProcess"
+ }
+ ]
+ },
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
+ "schema.SysInfoProcess": {
+ "type": "object",
+ "properties": {
+ "cpu_percent": {
+ "description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
+ "type": "number"
+ },
+ "memory_percent": {
+ "type": "number"
+ },
+ "pid": {
+ "type": "integer"
+ },
+ "rss_bytes": {
+ "description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
+ "type": "integer"
+ },
+ "started_at": {
+ "type": "string"
+ }
+ }
+ },
"schema.SystemInformationResponse": {
"type": "object",
"properties": {
@@ -7836,6 +7962,158 @@ const docTemplate = `{
}
}
},
+ "schema.SystemOneAnswer": {
+ "type": "object",
+ "properties": {
+ "choice": {
+ "type": "string"
+ },
+ "confidence": {
+ "type": "number"
+ },
+ "entities": {
+ "type": "array",
+ "items": {
+ "$ref": "#/definitions/schema.SystemOneEntity"
+ }
+ },
+ "legend": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "string"
+ }
+ },
+ "noul": {
+ "type": "number"
+ },
+ "probabilities": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ },
+ "score": {
+ "type": "number"
+ },
+ "type": {
+ "type": "string"
+ }
+ }
+ },
+ "schema.SystemOneEntity": {
+ "type": "object",
+ "properties": {
+ "confidence": {
+ "type": "number"
+ },
+ "end": {
+ "type": "integer"
+ },
+ "start": {
+ "type": "integer"
+ },
+ "text": {
+ "type": "string"
+ }
+ }
+ },
+ "schema.SystemOnePermuteRequest": {
+ "type": "object",
+ "properties": {
+ "n_perm": {
+ "type": "integer"
+ },
+ "question": {
+ "type": "string"
+ },
+ "request": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ },
+ "seed": {
+ "type": "integer"
+ }
+ }
+ },
+ "schema.SystemOnePermuteResponse": {
+ "type": "object",
+ "properties": {
+ "argmax_stable": {
+ "type": "boolean"
+ },
+ "runs": {
+ "type": "array",
+ "items": {
+ "$ref": "#/definitions/schema.SystemOnePermuteRun"
+ }
+ },
+ "spread": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ }
+ }
+ },
+ "schema.SystemOnePermuteRun": {
+ "type": "object",
+ "properties": {
+ "choice": {
+ "type": "string"
+ },
+ "latency_ms": {
+ "type": "number"
+ },
+ "order": {
+ "type": "array",
+ "items": {
+ "type": "string"
+ }
+ },
+ "probabilities": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ }
+ }
+ },
+ "schema.SystemOneRequest": {
+ "type": "object"
+ },
+ "schema.SystemOneResponse": {
+ "type": "object",
+ "properties": {
+ "answers": {
+ "type": "object",
+ "additionalProperties": {
+ "$ref": "#/definitions/schema.SystemOneAnswer"
+ }
+ },
+ "latency_ms": {
+ "type": "number"
+ },
+ "model": {
+ "type": "string"
+ },
+ "usage": {
+ "$ref": "#/definitions/schema.SystemOneUsage"
+ }
+ }
+ },
+ "schema.SystemOneUsage": {
+ "type": "object",
+ "properties": {
+ "input_tokens": {
+ "type": "integer"
+ },
+ "output_tokens": {
+ "type": "integer"
+ }
+ }
+ },
"schema.TTSRequest": {
"description": "TTS request body",
"type": "object",
diff --git a/swagger/swagger.json b/swagger/swagger.json
index b4e49b347..06d6c2c8d 100644
--- a/swagger/swagger.json
+++ b/swagger/swagger.json
@@ -3767,6 +3767,90 @@
}
}
},
+ "/v1/systemone": {
+ "post": {
+ "description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Answer structured-extraction questions over state text.",
+ "parameters": [
+ {
+ "description": "state + questions",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneResponse"
+ }
+ }
+ }
+ }
+ },
+ "/v1/systemone/permute": {
+ "post": {
+ "description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Re-run a choice question under multiple option orders.",
+ "parameters": [
+ {
+ "description": "request + question + n_perm + seed",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOnePermuteRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOnePermuteResponse"
+ }
+ }
+ }
+ }
+ },
+ "/v1/systemone/separate": {
+ "post": {
+ "description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
+ "tags": [
+ "systemone"
+ ],
+ "summary": "Answer each question in a separate NER pass.",
+ "parameters": [
+ {
+ "description": "state + questions",
+ "name": "request",
+ "in": "body",
+ "required": true,
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ }
+ }
+ ],
+ "responses": {
+ "200": {
+ "description": "OK",
+ "schema": {
+ "$ref": "#/definitions/schema.SystemOneResponse"
+ }
+ }
+ }
+ }
+ },
"/v1/text-to-speech/{voice-id}": {
"post": {
"tags": [
@@ -4090,8 +4174,16 @@
"config.Gallery": {
"type": "object",
"properties": {
+ "artifact_verification": {
+ "description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
+ "allOf": [
+ {
+ "$ref": "#/definitions/config.GalleryVerification"
+ }
+ ]
+ },
"mirrors": {
- "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
+ "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
"type": "array",
"items": {
"type": "string"
@@ -4126,6 +4218,10 @@
"not_before": {
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
"type": "string"
+ },
+ "source_repository": {
+ "description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
+ "type": "string"
}
}
},
@@ -7808,12 +7904,42 @@
"id": {
"type": "string"
},
+ "process": {
+ "description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
+ "allOf": [
+ {
+ "$ref": "#/definitions/schema.SysInfoProcess"
+ }
+ ]
+ },
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
+ "schema.SysInfoProcess": {
+ "type": "object",
+ "properties": {
+ "cpu_percent": {
+ "description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
+ "type": "number"
+ },
+ "memory_percent": {
+ "type": "number"
+ },
+ "pid": {
+ "type": "integer"
+ },
+ "rss_bytes": {
+ "description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
+ "type": "integer"
+ },
+ "started_at": {
+ "type": "string"
+ }
+ }
+ },
"schema.SystemInformationResponse": {
"type": "object",
"properties": {
@@ -7833,6 +7959,158 @@
}
}
},
+ "schema.SystemOneAnswer": {
+ "type": "object",
+ "properties": {
+ "choice": {
+ "type": "string"
+ },
+ "confidence": {
+ "type": "number"
+ },
+ "entities": {
+ "type": "array",
+ "items": {
+ "$ref": "#/definitions/schema.SystemOneEntity"
+ }
+ },
+ "legend": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "string"
+ }
+ },
+ "noul": {
+ "type": "number"
+ },
+ "probabilities": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ },
+ "score": {
+ "type": "number"
+ },
+ "type": {
+ "type": "string"
+ }
+ }
+ },
+ "schema.SystemOneEntity": {
+ "type": "object",
+ "properties": {
+ "confidence": {
+ "type": "number"
+ },
+ "end": {
+ "type": "integer"
+ },
+ "start": {
+ "type": "integer"
+ },
+ "text": {
+ "type": "string"
+ }
+ }
+ },
+ "schema.SystemOnePermuteRequest": {
+ "type": "object",
+ "properties": {
+ "n_perm": {
+ "type": "integer"
+ },
+ "question": {
+ "type": "string"
+ },
+ "request": {
+ "$ref": "#/definitions/schema.SystemOneRequest"
+ },
+ "seed": {
+ "type": "integer"
+ }
+ }
+ },
+ "schema.SystemOnePermuteResponse": {
+ "type": "object",
+ "properties": {
+ "argmax_stable": {
+ "type": "boolean"
+ },
+ "runs": {
+ "type": "array",
+ "items": {
+ "$ref": "#/definitions/schema.SystemOnePermuteRun"
+ }
+ },
+ "spread": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ }
+ }
+ },
+ "schema.SystemOnePermuteRun": {
+ "type": "object",
+ "properties": {
+ "choice": {
+ "type": "string"
+ },
+ "latency_ms": {
+ "type": "number"
+ },
+ "order": {
+ "type": "array",
+ "items": {
+ "type": "string"
+ }
+ },
+ "probabilities": {
+ "type": "object",
+ "additionalProperties": {
+ "type": "number",
+ "format": "float64"
+ }
+ }
+ }
+ },
+ "schema.SystemOneRequest": {
+ "type": "object"
+ },
+ "schema.SystemOneResponse": {
+ "type": "object",
+ "properties": {
+ "answers": {
+ "type": "object",
+ "additionalProperties": {
+ "$ref": "#/definitions/schema.SystemOneAnswer"
+ }
+ },
+ "latency_ms": {
+ "type": "number"
+ },
+ "model": {
+ "type": "string"
+ },
+ "usage": {
+ "$ref": "#/definitions/schema.SystemOneUsage"
+ }
+ }
+ },
+ "schema.SystemOneUsage": {
+ "type": "object",
+ "properties": {
+ "input_tokens": {
+ "type": "integer"
+ },
+ "output_tokens": {
+ "type": "integer"
+ }
+ }
+ },
"schema.TTSRequest": {
"description": "TTS request body",
"type": "object",
diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml
index 34fcfc8d7..1337afa9a 100644
--- a/swagger/swagger.yaml
+++ b/swagger/swagger.yaml
@@ -2,13 +2,19 @@ basePath: /
definitions:
config.Gallery:
properties:
+ artifact_verification:
+ allOf:
+ - $ref: '#/definitions/config.GalleryVerification'
+ description: |-
+ ArtifactVerification overrides Verification only for the gallery OCI artifact.
+ Backend images keep their separate Verification policy.
mirrors:
description: |-
Mirrors are tried in order when URL cannot be fetched. They are a
fallback for availability, not a load-balancing pool: the primary is
always preferred, and a mirror is only consulted after the one before
it fails. Any URI the gallery loader understands works here
- (https://, github:, file://).
+ (https://, github:, file://, oci://).
items:
type: string
type: array
@@ -32,6 +38,11 @@ definitions:
not_before:
description: NotBefore is an RFC3339 timestamp. Empty disables the time check.
type: string
+ source_repository:
+ description: |-
+ SourceRepository is an https URL compared exactly against the
+ certificate's source-repository extension. Empty skips the check.
+ type: string
type: object
config.TTSVoice:
properties:
@@ -2654,12 +2665,38 @@ definitions:
type: string
id:
type: string
+ process:
+ allOf:
+ - $ref: '#/definitions/schema.SysInfoProcess'
+ description: |-
+ Process is the backend process serving the model on this host. Absent
+ when the model has no local process (a distributed worker holds it) or
+ the process could not be read.
size_vram:
description: |-
SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
the backend process tree has no complete supported reading.
type: integer
type: object
+ schema.SysInfoProcess:
+ properties:
+ cpu_percent:
+ description: |-
+ CPUPercent is the share of the whole host's CPU used since the previous
+ reading, 0-100. Absent on the first reading of a process.
+ type: number
+ memory_percent:
+ type: number
+ pid:
+ type: integer
+ rss_bytes:
+ description: |-
+ RSSBytes is resident host memory. Weights offloaded to a GPU are not
+ in it.
+ type: integer
+ started_at:
+ type: string
+ type: object
schema.SystemInformationResponse:
properties:
backends:
@@ -2673,6 +2710,106 @@ definitions:
$ref: '#/definitions/schema.SysInfoModel'
type: array
type: object
+ schema.SystemOneAnswer:
+ properties:
+ choice:
+ type: string
+ confidence:
+ type: number
+ entities:
+ items:
+ $ref: '#/definitions/schema.SystemOneEntity'
+ type: array
+ legend:
+ additionalProperties:
+ type: string
+ type: object
+ noul:
+ type: number
+ probabilities:
+ additionalProperties:
+ format: float64
+ type: number
+ type: object
+ score:
+ type: number
+ type:
+ type: string
+ type: object
+ schema.SystemOneEntity:
+ properties:
+ confidence:
+ type: number
+ end:
+ type: integer
+ start:
+ type: integer
+ text:
+ type: string
+ type: object
+ schema.SystemOnePermuteRequest:
+ properties:
+ n_perm:
+ type: integer
+ question:
+ type: string
+ request:
+ $ref: '#/definitions/schema.SystemOneRequest'
+ seed:
+ type: integer
+ type: object
+ schema.SystemOnePermuteResponse:
+ properties:
+ argmax_stable:
+ type: boolean
+ runs:
+ items:
+ $ref: '#/definitions/schema.SystemOnePermuteRun'
+ type: array
+ spread:
+ additionalProperties:
+ format: float64
+ type: number
+ type: object
+ type: object
+ schema.SystemOnePermuteRun:
+ properties:
+ choice:
+ type: string
+ latency_ms:
+ type: number
+ order:
+ items:
+ type: string
+ type: array
+ probabilities:
+ additionalProperties:
+ format: float64
+ type: number
+ type: object
+ type: object
+ schema.SystemOneRequest:
+ type: object
+ schema.SystemOneResponse:
+ properties:
+ answers:
+ additionalProperties:
+ $ref: '#/definitions/schema.SystemOneAnswer'
+ type: object
+ latency_ms:
+ type: number
+ model:
+ type: string
+ usage:
+ $ref: '#/definitions/schema.SystemOneUsage'
+ type: object
+ schema.SystemOneUsage:
+ properties:
+ input_tokens:
+ type: integer
+ output_tokens:
+ type: integer
+ type: object
schema.TTSRequest:
description: TTS request body
properties:
@@ -5672,6 +5809,64 @@ paths:
summary: Generates audio from the input text.
tags:
- audio
+ /v1/systemone:
+ post:
+ description: 'Runs zero-shot NER over the supplied state and answers each question.
+ Question types: noul (binary entity presence), choice (pick one option), score
+ (pick one level).'
+ parameters:
+ - description: state + questions
+ in: body
+ name: request
+ required: true
+ schema:
+ $ref: '#/definitions/schema.SystemOneRequest'
+ responses:
+ "200":
+ description: OK
+ schema:
+ $ref: '#/definitions/schema.SystemOneResponse'
+ summary: Answer structured-extraction questions over state text.
+ tags:
+ - systemone
+ /v1/systemone/permute:
+ post:
+ description: Re-runs one choice question under n_perm option orders. Reports
+ per-order probabilities, argmax stability, and spread.
+ parameters:
+ - description: request + question + n_perm + seed
+ in: body
+ name: request
+ required: true
+ schema:
+ $ref: '#/definitions/schema.SystemOnePermuteRequest'
+ responses:
+ "200":
+ description: OK
+ schema:
+ $ref: '#/definitions/schema.SystemOnePermuteResponse'
+ summary: Re-run a choice question under multiple option orders.
+ tags:
+ - systemone
+ /v1/systemone/separate:
+ post:
+ description: Runs N independent NER passes, one per question, against the same
+ state. Response shape matches /v1/systemone.
+ parameters:
+ - description: state + questions
+ in: body
+ name: request
+ required: true
+ schema:
+ $ref: '#/definitions/schema.SystemOneRequest'
+ responses:
+ "200":
+ description: OK
+ schema:
+ $ref: '#/definitions/schema.SystemOneResponse'
+ summary: Answer each question in a separate NER pass.
+ tags:
+ - systemone
/v1/text-to-speech/{voice-id}:
post:
parameters: