Files
LocalAI/core/services/compression/inference.go
T
localai-org-maint-botandlocalai-org-maint-bot 0761bd02c7 feat(chat): add end-to-end context compression (#11556)
* feat(config): add context compression policy

Define the opt-in model configuration contract before the chat middleware consumes it. Document each policy field so later request handling does not invent a second schema.\n\nRefs #9534\n\nAssisted-by: Codex:gpt-5

* fix(config): register compression fields

The model editor metadata gate rejects new config fields without descriptions and suitable controls. Register the compression policy so operators can edit its six fields safely.

Assisted-by: Codex:gpt-5 [monitoring-prs]

* feat(chat): compress long contexts

Long conversations currently fail once they reach the model context window. The opt-in policy now summarizes complete older turns before primary inference and preserves the newest tool chains.

Both OpenAI and MCP chat routes share the same transformation. Usage metadata and metrics expose each compression event.

Refs #9534

Assisted-by: Codex:gpt-5

---------

Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
2026-08-18 11:31:03 +00:00

63 lines
2.2 KiB
Go

package compression
import (
"context"
"encoding/json"
"fmt"
"strings"
"github.com/mudler/LocalAI/core/backend"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/pkg/functions"
"github.com/mudler/LocalAI/pkg/model"
)
const summarizerInstruction = `Summarize the following conversation for an AI agent to continue coherently. Preserve names, numbers, decisions, URLs, error messages, tool names, and tool results. Drop pleasantries and repetition.`
type InferenceSummarizer struct {
configs *config.ModelConfigLoader
models *model.ModelLoader
app *config.ApplicationConfig
}
func NewInferenceSummarizer(configs *config.ModelConfigLoader, models *model.ModelLoader, app *config.ApplicationConfig) *InferenceSummarizer {
return &InferenceSummarizer{configs: configs, models: models, app: app}
}
func (s *InferenceSummarizer) Summarize(ctx context.Context, modelName string, messages []schema.Message, maxTokens int) (string, int, error) {
cfg, err := s.configs.LoadModelConfigFileByNameDefaultOptions(modelName, s.app)
if err != nil {
return "", 0, fmt.Errorf("load compressor model %q: %w", modelName, err)
}
if cfg == nil {
return "", 0, fmt.Errorf("load compressor model %q: configuration not found", modelName)
}
runtimeCfg := *cfg
cfg = &runtimeCfg
cfg.Maxtokens = &maxTokens
temperature := 0.0
cfg.Temperature = &temperature
payload, err := json.Marshal(messages)
if err != nil {
return "", 0, fmt.Errorf("encode conversation: %w", err)
}
prompt := fmt.Sprintf("%s\n\nMaximum summary length: %d tokens.\n\nConversation:\n%s", summarizerInstruction, maxTokens, payload)
fn, err := backend.ModelInference(ctx, prompt, nil, nil, nil, nil, s.models, cfg, s.configs, s.app, nil, "", "", nil, nil, nil, nil)
if err != nil {
return "", 0, err
}
response, err := fn()
if err != nil {
return "", 0, err
}
summary := strings.TrimSpace(functions.ContentFromChatDeltas(response.ChatDeltas))
if summary == "" {
summary = strings.TrimSpace(response.Response)
}
if summary == "" {
return "", 0, fmt.Errorf("compressor model %q returned an empty summary", modelName)
}
return summary, response.Usage.Completion, nil
}