mirror of
https://github.com/mudler/LocalAI.git
synced 2026-06-13 03:09:03 -04:00
* fix(router): score classifier production-readiness Conversation trimming runs through the classifier model's chat template and trims by exact token count, sized to the model's n_batch which is now scaled to context so long probes can't crash the backend. Missing chat_message templates are a hard error at router build time. Router- facing factories (Embedder/Scorer/Reranker/TokenCounter) re-resolve ModelConfig per call so a model installed post-startup doesn't bind a stub Backend="" config and silently fall into the loader's auto- iterate path. New 'vector_store' backend trace recorded inside localVectorStore on every Search/Insert — including the backend-load-failure path that previously vanished into an xlog.Warn — with outcome tagging (hit/miss/empty_store/backend_load_error/find_error/insert_error/ok). Companion cleanup drops misleading similarity:0 and input_tokens_count:0 from non-hit and text-mode traces. Gallery local-store-development aliases to 'local-store' so the master image satisfies pkg/model.LocalStoreBackend lookups from the embedding cache. Misc: llama-cpp TokenizeString reads the correct 'prompt' JSON key (the original bug); ModelTokenize nil-guard; non-fatal mitm proxy startup; PII 'route_local' renamed to 'allow' with docs/UI in sync; model-editor footer no longer eats the edit area on small screens; several config-editor template/dropdown/section fixes. Tests: e2e router specs (casual/code-hint + long-conversation trim), vector_store trace specs, lazy-factory specs, gallery dev-alias resolution, Playwright trace badge + scroll regression. Assisted-by: Claude:claude-opus-4-7 [Claude Code] Signed-off-by: Richard Palethorpe <io@richiejp.com> * feat(backend): auto-size batch to context for embedding and rerank models Embedding and rerank models pool over the whole input in a single physical batch (n_ubatch). With batch left at the 512 default, the backend rejects longer inputs with "input is too large to process", silently capping a large-context embedder (e.g. 8k/32k) at 512 tokens. Size n_batch to the context for these single-pass usecases, mirroring the existing FLAG_SCORE behaviour; an explicit batch: still wins. Extracts EffectiveContextSize/EffectiveBatchSize from grpcModelOpts so the effective decode window has one home for other callers to reuse. Adds an e2e-aio regression test that embeds a >512-token input. The AIO embedding model is switched to nomic-embed-text-v1.5 (2048 context) because the previous granite model was capped at 512 tokens and could not exercise the larger batch. Assisted-by: claude-code:claude-opus-4-8 [Claude Code] Signed-off-by: Richard Palethorpe <io@richiejp.com> * fix(gallery): raise arch-router scoring output cap via parallel:64 Scoring decodes the whole prompt+candidate in a single llama_decode and reads one logit row per candidate token. The vendored llama.cpp server caps causal output rows at n_parallel, so the default of 1 aborts with GGML_ASSERT(n_outputs_max <= cparams.n_outputs_max) on multi-token route labels. Set options: [parallel:64] on both arch-router quant entries to lift the cap; kv_unified (the grpc-server default) keeps the full context per sequence, so this does not split the KV cache. Assisted-by: claude-code:claude-opus-4-8 [Claude Code] Signed-off-by: Richard Palethorpe <io@richiejp.com> --------- Signed-off-by: Richard Palethorpe <io@richiejp.com>
136 lines
2.8 KiB
Go
136 lines
2.8 KiB
Go
package meta
|
|
|
|
import (
|
|
"reflect"
|
|
"sort"
|
|
"sync"
|
|
)
|
|
|
|
var (
|
|
cachedMetadata *ConfigMetadata
|
|
cacheMu sync.RWMutex
|
|
)
|
|
|
|
// BuildConfigMetadata reflects on the given struct type (ModelConfig),
|
|
// merges the enrichment registry, and returns the full ConfigMetadata.
|
|
// The result is cached in memory after the first call.
|
|
func BuildConfigMetadata(modelConfigType reflect.Type) *ConfigMetadata {
|
|
cacheMu.RLock()
|
|
if cachedMetadata != nil {
|
|
cacheMu.RUnlock()
|
|
return cachedMetadata
|
|
}
|
|
cacheMu.RUnlock()
|
|
|
|
cacheMu.Lock()
|
|
defer cacheMu.Unlock()
|
|
|
|
if cachedMetadata != nil {
|
|
return cachedMetadata
|
|
}
|
|
|
|
cachedMetadata = buildConfigMetadataUncached(modelConfigType, DefaultRegistry())
|
|
return cachedMetadata
|
|
}
|
|
|
|
// buildConfigMetadataUncached does the actual work without caching.
|
|
func buildConfigMetadataUncached(modelConfigType reflect.Type, registry map[string]FieldMetaOverride) *ConfigMetadata {
|
|
fields := WalkModelConfig(modelConfigType)
|
|
|
|
for i := range fields {
|
|
override, ok := registry[fields[i].Path]
|
|
if !ok {
|
|
continue
|
|
}
|
|
applyOverride(&fields[i], override)
|
|
}
|
|
|
|
allSections := DefaultSections()
|
|
|
|
sectionOrder := make(map[string]int, len(allSections))
|
|
for _, s := range allSections {
|
|
sectionOrder[s.ID] = s.Order
|
|
}
|
|
|
|
sort.SliceStable(fields, func(i, j int) bool {
|
|
si := sectionOrder[fields[i].Section]
|
|
sj := sectionOrder[fields[j].Section]
|
|
if si != sj {
|
|
return si < sj
|
|
}
|
|
return fields[i].Order < fields[j].Order
|
|
})
|
|
|
|
usedSections := make(map[string]bool)
|
|
for _, f := range fields {
|
|
usedSections[f.Section] = true
|
|
}
|
|
|
|
var sections []Section
|
|
for _, s := range allSections {
|
|
if usedSections[s.ID] {
|
|
sections = append(sections, s)
|
|
}
|
|
}
|
|
|
|
return &ConfigMetadata{
|
|
Sections: sections,
|
|
Fields: fields,
|
|
}
|
|
}
|
|
|
|
// applyOverride merges non-zero override values into the field.
|
|
func applyOverride(f *FieldMeta, o FieldMetaOverride) {
|
|
if o.Section != "" {
|
|
f.Section = o.Section
|
|
}
|
|
if o.Label != "" {
|
|
f.Label = o.Label
|
|
}
|
|
if o.Description != "" {
|
|
f.Description = o.Description
|
|
}
|
|
if o.Component != "" {
|
|
f.Component = o.Component
|
|
}
|
|
if o.Language != "" {
|
|
f.Language = o.Language
|
|
}
|
|
if o.Placeholder != "" {
|
|
f.Placeholder = o.Placeholder
|
|
}
|
|
if o.Default != nil {
|
|
f.Default = o.Default
|
|
}
|
|
if o.Min != nil {
|
|
f.Min = o.Min
|
|
}
|
|
if o.Max != nil {
|
|
f.Max = o.Max
|
|
}
|
|
if o.Step != nil {
|
|
f.Step = o.Step
|
|
}
|
|
if o.Options != nil {
|
|
f.Options = o.Options
|
|
}
|
|
if o.AutocompleteProvider != "" {
|
|
f.AutocompleteProvider = o.AutocompleteProvider
|
|
}
|
|
if o.VRAMImpact {
|
|
f.VRAMImpact = true
|
|
}
|
|
if o.Advanced {
|
|
f.Advanced = true
|
|
}
|
|
if o.Order != 0 {
|
|
f.Order = o.Order
|
|
}
|
|
}
|
|
|
|
// BuildForTest builds metadata without caching, for use in tests.
|
|
func BuildForTest(modelConfigType reflect.Type, registry map[string]FieldMetaOverride) *ConfigMetadata {
|
|
return buildConfigMetadataUncached(modelConfigType, registry)
|
|
}
|
|
|