chore: merge master into distributed transport PR

Bring the distributed branch onto current master before the CI fix.

Assisted-by: Codex:gpt-6
This commit is contained in:
localai-org-maint-bot committed 2026-10-01 03:12:40 +00:00
commit 4bc0e92a3d
102 files changed
+7143 -205

No files matched your search

+6
View File
@@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{
// Detection
{"POST", "/v1/detection", FeatureDetection},
// Decisions API (SystemOne wire contract)
{"POST", "/v1/systemone", FeatureDecisions},
{"POST", "/v1/systemone/permute", FeatureDecisions},
{"POST", "/v1/systemone/separate", FeatureDecisions},
// Face recognition
{"POST", "/v1/face/verify", FeatureFaceRecognition},
{"POST", "/v1/face/analyze", FeatureFaceRecognition},
@@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta {
{FeatureVoiceRecognition, "Voice Recognition", true},
{FeatureAudioTransform, "Audio Transform", true},
{FeaturePIIFilter, "PII Analyze / Redact", true},
{FeatureDecisions, "Decisions", true},
}
}
+24
View File
@@ -0,0 +1,24 @@
package auth_test
import (
. "github.com/mudler/LocalAI/core/http/auth"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("Decisions feature registration", func() {
It("gates the three decision routes behind one default-on API feature", func() {
Expect(APIFeatures).To(ContainElement(FeatureDecisions))
patterns := []string{}
for _, route := range RouteFeatureRegistry {
if route.Feature == FeatureDecisions {
Expect(route.Method).To(Equal("POST"))
patterns = append(patterns, route.Pattern)
}
}
Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"))
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true}))
})
})
+2 -1
View File
@@ -59,6 +59,7 @@ const (
FeatureFaceRecognition = "face_recognition"
FeatureVoiceRecognition = "voice_recognition"
FeatureAudioTransform = "audio_transform"
FeatureDecisions = "decisions"
// FeaturePIIFilter gates the synchronous PII analyze/redact service
// (POST /api/pii/{analyze,redact}). Default ON like the other API
// features; the admin-only events log is gated separately in-handler.
@@ -78,7 +79,7 @@ var APIFeatures = []string{
FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound,
FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores,
FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform,
FeaturePIIFilter,
FeaturePIIFilter, FeatureDecisions,
}
// AllFeatures lists all known features (used by UI and validation).
@@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{
Tags: []string{"voice-recognition"},
Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.",
},
{
Name: "decisions",
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
Tags: []string{"systemone"},
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.",
},
{
Name: "branding",
Description: "Whitelabel the instance: configure name, tagline, logo, and favicon",
@@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() {
instructions, ok := resp["instructions"].([]any)
Expect(ok).To(BeTrue())
Expect(instructions).To(HaveLen(20))
Expect(instructions).To(HaveLen(21))
// Verify each instruction has required fields and correct URL format
for _, s := range instructions {
@@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() {
"voice-library",
"3d",
"failover",
"decisions",
))
})
})
@@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() {
Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations"))
})
It("should advertise the Decisions API", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil)
rec := httptest.NewRecorder()
app.ServeHTTP(rec, req)
Expect(rec.Code).To(Equal(http.StatusOK))
body, _ := io.ReadAll(rec.Body)
Expect(string(body)).To(ContainSubstring("POST /v1/systemone"))
Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]"))
})
It("should return JSON fragment when format=json", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil)
rec := httptest.NewRecorder()
+217 -7
View File
@@ -2,6 +2,7 @@ package localai
import (
"encoding/json"
"errors"
"fmt"
"math"
"math/rand"
@@ -371,6 +372,191 @@ func systemOneError(c echo.Context, status int, msg string) error {
})
}
// systemOneModelAllowed keeps chat and embedding models out of the decision
// API with an actionable error instead of a backend failure. A config that
// declares no usecases predates the flag and stays allowed, and a
// token_classify model is allowed because the NER path serves it.
func systemOneModelAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 {
return nil
}
return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name)
}
// checkSystemOneModel applies systemOneModelAllowed to a model looked up by
// name. An unknown model passes here so the existing not-found handling
// downstream keeps its status code.
func checkSystemOneModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneModelAllowed(cfg)
}
// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the
// request to the backend's Score RPC (the decision pipeline) for this model.
// A model that declares token_classify without systemone is a zero-shot NER
// model: the backend's decision entry point refuses those architectures, so it
// goes to the NER path instead. A config that declares nothing keeps the
// decision pipeline, which is what setups that predate the decisions usecase
// relied on.
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
if !backendSupportsScore(cfg.Backend) {
return false
}
if cfg.KnownUsecases == nil {
return true
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 {
return true
}
return declared&config.FLAG_TOKEN_CLASSIFY == 0
}
// systemOneNERAllowed guards /permute and /separate, which always run the NER
// path. A decision model cannot serve them: the backend's NER entry point
// refuses its architecture, and the caller would see a backend error.
func systemOneNERAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
}
return nil
}
// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by
// name; an unknown model passes so the not-found handling keeps its status.
func checkSystemOneNERModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneNERAllowed(cfg)
}
// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a
// single call from pinning a decision model on an unbounded prompt, and match
// the limits Ollama documents for the same wire contract, so a client written
// for one server behaves the same on the other. The engine enforces any
// per-model option cap (letter-answer models refuse more than 26 options).
const (
systemOneMaxBody = 64 << 10
systemOneMaxQuestions = 64
)
// systemOneBind binds the JSON body with a size cap. Bind reads the whole body
// first, so the cap has to be on the reader.
func systemOneBind(c echo.Context, v any) error {
c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody)
return c.Bind(v)
}
// systemOneBindStatus maps a bind failure to its status: 413 when the body
// exceeded the cap, 400 for anything else.
func systemOneBindStatus(err error) int {
var tooLarge *http.MaxBytesError
if errors.As(err, &tooLarge) {
return http.StatusRequestEntityTooLarge
}
return http.StatusBadRequest
}
func systemOneBindMessage(err error) string {
if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge {
return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10)
}
return "invalid request body"
}
// validateSystemOneRequest checks the structure every path needs, before the
// request is forwarded to a decision model or run through the NER path. The
// forwarded path never sees parseSystemOneRequest, so without this a malformed
// question would surface as a backend error instead of a 400.
func validateSystemOneRequest(req *schema.SystemOneRequest) error {
if len(req.State) == 0 || string(req.State) == "null" {
return fmt.Errorf("state is required")
}
var state any
if err := json.Unmarshal(req.State, &state); err != nil {
return fmt.Errorf("state is not valid JSON: %w", err)
}
if s, ok := state.(string); ok && strings.TrimSpace(s) == "" {
return fmt.Errorf("state is required")
}
if len(req.Questions) == 0 {
return fmt.Errorf("questions is required and must contain at least one question")
}
if len(req.Questions) > systemOneMaxQuestions {
return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions)
}
qids := make([]string, 0, len(req.Questions))
for id := range req.Questions {
qids = append(qids, id)
}
sort.Strings(qids)
for _, id := range qids {
if strings.TrimSpace(id) == "" {
return fmt.Errorf("question ids must not be blank")
}
q := req.Questions[id]
switch q.Type {
case "choice":
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (choice) requires a criteria object", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (choice) requires at least 2 options", id)
}
for k := range criteria {
if strings.TrimSpace(k) == "" {
return fmt.Errorf("question %q (choice) has a blank option key", id)
}
}
case "score":
var criteria []json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (score) requires a criteria array", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (score) requires at least 2 levels", id)
}
case "noul":
if len(q.Criteria) == 0 || string(q.Criteria) == "null" {
continue
}
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id)
}
for k := range criteria {
if k != "false" && k != "true" {
return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id)
}
}
default:
return fmt.Errorf("question %q has unknown type: %s", id, q.Type)
}
}
return nil
}
// backendSupportsScore reports whether the named backend implements the
// Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms
// scoring via the unified vllm_decide C ABI); other backends fall through to
@@ -402,18 +588,24 @@ func backendSupportsScore(backendName string) bool {
func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
// vllm-cpp models (kev/laya) implement the decision pipeline natively
// via the vllm_decide C ABI. Forward the raw request JSON through the
// Score RPC and return the backend's response as-is.
cl := app.ModelConfigLoader()
if cl != nil {
if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) {
if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) {
reqJSON, err := json.Marshal(req)
if err != nil {
return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error())
@@ -468,12 +660,21 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOnePermuteRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Request.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req.Request); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if req.Question == "" {
return systemOneError(c, http.StatusBadRequest, "question is required")
}
@@ -604,12 +805,21 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
parsed, err := parseSystemOneRequest(&req)
if err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
@@ -0,0 +1,74 @@
package localai
import (
"github.com/mudler/LocalAI/core/config"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("systemOneModelAllowed", func() {
mk := func(usecases ...string) config.ModelConfig {
return config.ModelConfig{
Name: "m",
Backend: "vllm-cpp",
KnownUsecases: config.GetUsecasesFromYAML(usecases),
}
}
It("accepts a declared decisions model", func() {
Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed())
})
It("accepts a token_classify model, which the NER path serves", func() {
Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed())
})
It("keeps configs that declare no usecases working", func() {
Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed())
})
It("refuses a chat-only model with an actionable message", func() {
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]")))
})
})
var _ = Describe("systemone routing by model kind", func() {
mk := func(backend string, usecases ...string) config.ModelConfig {
c := config.ModelConfig{Name: "m", Backend: backend}
if len(usecases) > 0 {
c.KnownUsecases = config.GetUsecasesFromYAML(usecases)
}
return c
}
Describe("systemOneUsesDecisionPipeline", func() {
It("sends a declared decision model to the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue())
})
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
})
It("keeps configs that declare nothing on the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
})
It("prefers the decision pipeline when both usecases are declared", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue())
})
It("never uses it for a backend without the Score RPC", func() {
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse())
})
})
Describe("systemOneNERAllowed", func() {
It("refuses a decision model on the NER-only routes with an actionable message", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone")))
})
It("accepts a token_classify model", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
})
It("accepts configs that declare nothing", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed())
})
})
})
@@ -0,0 +1,96 @@
package localai
import (
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/schema"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("validateSystemOneRequest", func() {
req := func(state string, questions string) *schema.SystemOneRequest {
r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)}
Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed())
return r
}
It("accepts the three question types", func() {
r := req(`"ticket text"`, `{
"team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}},
"refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}},
"urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]}
}`)
Expect(validateSystemOneRequest(r)).To(Succeed())
})
It("accepts a noul question with no criteria", func() {
Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed())
})
DescribeTable("refuses a malformed request with a message that names the problem",
func(state, questions, want string) {
Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want)))
},
Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("no questions", `"x"`, `{}`, "at least one question"),
Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"),
Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"),
Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"),
Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"),
Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"),
Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`),
)
It("refuses more than 64 questions", func() {
var b strings.Builder
b.WriteString("{")
for i := 0; i < 65; i++ {
if i > 0 {
b.WriteString(",")
}
b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`)
}
b.WriteString("}")
Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64")))
})
})
var _ = Describe("systemOneBind", func() {
bind := func(body string) (int, error) {
e := echo.New()
r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body))
r.Header.Set("Content-Type", "application/json")
c := e.NewContext(r, httptest.NewRecorder())
var out schema.SystemOneRequest
if err := systemOneBind(c, &out); err != nil {
return systemOneBindStatus(err), err
}
return http.StatusOK, nil
}
It("binds a normal body", func() {
status, err := bind(`{"model":"m","state":"x","questions":{}}`)
Expect(err).ToNot(HaveOccurred())
Expect(status).To(Equal(http.StatusOK))
})
It("answers 413 for a body over 64 KiB", func() {
status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusRequestEntityTooLarge))
})
It("answers 400 for malformed JSON", func() {
status, err := bind(`{not json`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusBadRequest))
})
})
@@ -99,6 +99,7 @@ type fakeModel struct {
transcribeDeltas []string
transcribeFinal *schema.TranscriptionResult
transcribeErr error
lastDiarize bool // diarize flag of the last Transcribe/TranscribeStream call
// TranscribeLive scripting: liveErr makes the open fail (degrade path);
// liveEvents are delivered to onEvent synchronously at open;
@@ -200,7 +201,8 @@ func (m *fakeModel) VAD(_ context.Context, req *schema.VADRequest) (*schema.VADR
return &schema.VADResponse{Segments: m.vadSegments}, nil
}
func (m *fakeModel) Transcribe(context.Context, string, string, bool, bool, string) (*schema.TranscriptionResult, error) {
func (m *fakeModel) Transcribe(_ context.Context, _, _ string, _, diarize bool, _ string) (*schema.TranscriptionResult, error) {
m.lastDiarize = diarize
return m.transcribeFinal, m.transcribeErr
}
@@ -247,7 +249,8 @@ func (m *fakeModel) TTSStream(_ context.Context, _, _, _ string, onAudio func(pc
return nil
}
func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, _ bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) {
func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, diarize bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) {
m.lastDiarize = diarize
for _, d := range m.transcribeDeltas {
onDelta(d)
}
@@ -209,6 +209,35 @@ func (l *liveTurnState) drainEvents(audioSec float64) {
if ev.Final != nil && strings.TrimSpace(ev.Final.Text) != "" {
l.finalText = ev.Final.Text
}
// Speaker and sound events from a companion diarization/scene
// stream: forward each as its own event under the turn's item
// id, same as caption deltas. Text is empty — the event exists
// to carry the speaker/segment boundary, not transcript text.
if l.transport != nil && l.itemID != "" {
for _, seg := range ev.Speakers {
sendEvent(l.transport, types.ConversationItemInputAudioTranscriptionSegmentEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: l.itemID,
ContentIndex: 0,
Speaker: seg.Speaker,
Start: seg.Start,
End: seg.End,
})
}
for _, sound := range ev.Sounds {
start, end := sound.Start, sound.End
sendEvent(l.transport, types.ConversationItemSoundDetectionEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: l.itemID,
ContentIndex: 0,
Detections: []types.SoundDetectionTag{
{Label: sound.Label, Score: sound.Peak, Index: sound.Index},
},
Start: &start,
End: &end,
})
}
}
default:
return
}
@@ -291,6 +291,67 @@ var _ = Describe("liveTurnState", func() {
Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionFailed)).To(Equal(0))
})
})
Describe("scene events (speakers and sounds)", func() {
It("emits a segment event per speaker with empty text under the turn's item id", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
turnID := lts.itemID
m.liveSession.onEvent(backend.LiveTranscriptionEvent{
Speakers: []backend.LiveSpeakerSegment{{Speaker: "1", Start: 1.2, End: 3.4}},
})
lts.drainEvents(3.4)
var got []types.ConversationItemInputAudioTranscriptionSegmentEvent
for _, e := range ftr.events() {
if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok {
got = append(got, seg)
}
}
Expect(got).To(HaveLen(1))
Expect(got[0].ItemID).To(Equal(turnID))
Expect(got[0].Speaker).To(Equal("1"))
Expect(got[0].Start).To(BeNumerically("~", 1.2, 1e-9))
Expect(got[0].End).To(BeNumerically("~", 3.4, 1e-9))
Expect(got[0].Text).To(BeEmpty())
})
It("emits a sound_detection event per sound with one tag and start/end", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
turnID := lts.itemID
m.liveSession.onEvent(backend.LiveTranscriptionEvent{
Sounds: []backend.LiveSoundEvent{{Label: "Dog bark", Index: 5, Peak: 0.8, Start: 0.5, End: 0.9}},
})
lts.drainEvents(1.0)
var got []types.ConversationItemSoundDetectionEvent
for _, e := range ftr.events() {
if sd, ok := e.(types.ConversationItemSoundDetectionEvent); ok {
got = append(got, sd)
}
}
Expect(got).To(HaveLen(1))
Expect(got[0].ItemID).To(Equal(turnID))
Expect(got[0].Detections).To(HaveLen(1))
Expect(got[0].Detections[0].Label).To(Equal("Dog bark"))
Expect(got[0].Detections[0].Score).To(BeNumerically("~", 0.8, 1e-6))
Expect(got[0].Detections[0].Index).To(Equal(5))
Expect(got[0].Start).NotTo(BeNil())
Expect(*got[0].Start).To(BeNumerically("~", 0.5, 1e-9))
Expect(got[0].End).NotTo(BeNil())
Expect(*got[0].End).To(BeNumerically("~", 0.9, 1e-9))
})
It("sends neither event when a live event carries no speakers or sounds", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
m.liveSession.onEvent(backend.LiveTranscriptionEvent{Delta: "hi"})
lts.drainEvents(1.0)
Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionSegment)).To(Equal(0))
Expect(ftr.countEvents(types.ServerEventTypeConversationItemSoundDetection)).To(Equal(0))
})
})
})
// commitUtteranceWithTranscript routes the three transcript sources: the
@@ -3,6 +3,7 @@ package openai
import (
"context"
"encoding/binary"
"encoding/json"
"errors"
"os"
@@ -14,6 +15,69 @@ import (
"github.com/mudler/LocalAI/core/schema"
)
// ConversationItemSoundDetectionEvent gained optional Start/End (seconds)
// for the live scene-event path; the unary/windowed paths never set them,
// so existing consumers must see no start/end keys at all.
var _ = Describe("ConversationItemSoundDetectionEvent JSON", func() {
It("omits start and end when nil", func() {
ev := types.ConversationItemSoundDetectionEvent{
ItemID: "item1",
Detections: []types.SoundDetectionTag{{Label: "Speech", Score: 0.5, Index: 7}},
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
_, hasStart := got["start"]
_, hasEnd := got["end"]
Expect(hasStart).To(BeFalse())
Expect(hasEnd).To(BeFalse())
})
It("includes start and end when set", func() {
start, end := 0.5, 0.9
ev := types.ConversationItemSoundDetectionEvent{
ItemID: "item1",
Start: &start,
End: &end,
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
Expect(got["start"]).To(BeNumerically("~", 0.5, 1e-9))
Expect(got["end"]).To(BeNumerically("~", 0.9, 1e-9))
})
})
// ConversationItemInputAudioTranscriptionSegmentEvent.Start/End are plain
// float64 (no omitempty): a speaker segment starting at 0.0s must still
// carry "start" in the JSON, unlike the sound-detection event's optional
// pointer fields above.
var _ = Describe("ConversationItemInputAudioTranscriptionSegmentEvent JSON", func() {
It("marshals start:0 and end:1.5 even when start is the zero value", func() {
ev := types.ConversationItemInputAudioTranscriptionSegmentEvent{
ItemID: "item1",
Speaker: "1",
Start: 0,
End: 1.5,
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
_, hasStart := got["start"]
_, hasEnd := got["end"]
Expect(hasStart).To(BeTrue())
Expect(hasEnd).To(BeTrue())
Expect(got["start"]).To(BeNumerically("~", 0.0, 1e-9))
Expect(got["end"]).To(BeNumerically("~", 1.5, 1e-9))
})
})
// emitSoundDetection classifies a committed utterance and emits a single
// conversation.item.sound_detection event carrying the scored AudioSet tags.
var _ = Describe("emitSoundDetection", func() {
@@ -5,6 +5,7 @@ import (
"fmt"
"github.com/mudler/LocalAI/core/http/endpoints/openai/types"
"github.com/mudler/LocalAI/core/schema"
)
// emitPrecomputedTranscription emits the transcription events for a turn
@@ -42,9 +43,10 @@ func emitPrecomputedTranscription(t Transport, itemID string, deltas []string, t
// a single completed event. delta and completed events share itemID.
func emitTranscription(ctx context.Context, t Transport, session *Session, itemID, audioPath string) (string, error) {
cfg := session.InputAudioTranscription
diarize := session.ModelConfig != nil && session.ModelConfig.Pipeline.Diarization
if session.ModelConfig != nil && session.ModelConfig.Pipeline.StreamTranscription() {
final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, false, cfg.Prompt, func(delta string) {
final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt, func(delta string) {
_ = t.SendEvent(types.ConversationItemInputAudioTranscriptionDeltaEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
@@ -58,6 +60,11 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
transcript := ""
if final != nil {
transcript = final.Text
if diarize {
if err := emitSpeakerSegments(t, itemID, final); err != nil {
return "", err
}
}
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
@@ -71,13 +78,18 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
}
// Unary fallback: transcribe the whole utterance, emit one completed event.
tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, false, cfg.Prompt)
tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt)
if err != nil {
return "", err
}
if tr == nil {
return "", fmt.Errorf("transcribe result is nil")
}
if diarize {
if err := emitSpeakerSegments(t, itemID, tr); err != nil {
return "", err
}
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
@@ -88,3 +100,29 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
}
return tr.Text, nil
}
// emitSpeakerSegments forwards each speaker-labelled segment of a committed
// turn's transcript as a conversation.item.input_audio_transcription.segment
// event (pipeline.diarization), before the turn's completed event. Times are
// relative to the turn's audio and speaker labels are only consistent within
// the turn, as on the live path. Segments without a speaker are skipped.
func emitSpeakerSegments(t Transport, itemID string, tr *schema.TranscriptionResult) error {
for _, seg := range tr.Segments {
if seg.Speaker == "" {
continue
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionSegmentEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
ContentIndex: 0,
ID: fmt.Sprintf("seg_%d", seg.Id),
Speaker: seg.Speaker,
Start: seg.Start.Seconds(),
End: seg.End.Seconds(),
Text: seg.Text,
}); err != nil {
return err
}
}
return nil
}
@@ -2,6 +2,7 @@ package openai
import (
"context"
"time"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
@@ -51,4 +52,86 @@ var _ = Describe("emitTranscription", func() {
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionDelta)).To(Equal(0))
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1))
})
Context("pipeline.diarization", func() {
labelled := &schema.TranscriptionResult{
Text: "hi there. hello",
Segments: []schema.TranscriptionSegment{
{Id: 0, Text: "hi there.", Start: 0, End: 600 * time.Millisecond, Speaker: "0"},
{Id: 1, Text: "hello", Start: time.Second, End: 1400 * time.Millisecond, Speaker: "1"},
{Id: 2, Text: "unlabelled"},
},
}
segmentEvents := func(t *fakeTransport) []types.ConversationItemInputAudioTranscriptionSegmentEvent {
var out []types.ConversationItemInputAudioTranscriptionSegmentEvent
for _, e := range t.sent {
if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok {
out = append(out, seg)
}
}
return out
}
It("requests speakers and emits one segment event per labelled segment", func() {
m := &fakeModel{transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{Diarization: true}},
ModelInterface: m,
}
t := &fakeTransport{}
transcript, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(transcript).To(Equal("hi there. hello"))
Expect(m.lastDiarize).To(BeTrue())
segs := segmentEvents(t)
Expect(segs).To(HaveLen(2))
Expect(segs[0].ItemID).To(Equal("item1"))
Expect(segs[0].Speaker).To(Equal("0"))
Expect(segs[0].Text).To(Equal("hi there."))
Expect(segs[1].Speaker).To(Equal("1"))
Expect(segs[1].Start).To(BeNumerically("~", 1.0, 1e-9))
Expect(segs[1].End).To(BeNumerically("~", 1.4, 1e-9))
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1))
})
It("also emits segment events on the streaming transcription path", func() {
on := true
m := &fakeModel{transcribeDeltas: []string{"hi"}, transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{
Diarization: true,
Streaming: config.PipelineStreaming{Transcription: &on},
}},
ModelInterface: m,
}
t := &fakeTransport{}
_, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(m.lastDiarize).To(BeTrue())
Expect(segmentEvents(t)).To(HaveLen(2))
})
It("neither asks for speakers nor emits segments when off", func() {
m := &fakeModel{transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{},
ModelInterface: m,
}
t := &fakeTransport{}
_, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(m.lastDiarize).To(BeFalse())
Expect(segmentEvents(t)).To(BeEmpty())
})
})
})
+14 -8
View File
@@ -210,18 +210,20 @@ func TranscriptEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
}
for _, word := range tr.Words {
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
for _, seg := range tr.Segments {
segWords := []schema.TranscriptionWordSeconds{}
for _, word := range seg.Words {
segWords = append(segWords, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
@@ -338,12 +340,16 @@ func streamTranscription(c echo.Context, req backend.TranscriptionRequest, ml *m
if len(finalResult.Segments) > 0 {
segs := make([]map[string]any, 0, len(finalResult.Segments))
for _, seg := range finalResult.Segments {
segs = append(segs, map[string]any{
entry := map[string]any{
"id": seg.Id,
"start": seg.Start.Seconds(),
"end": seg.End.Seconds(),
"text": seg.Text,
})
}
if seg.Speaker != "" {
entry["speaker"] = seg.Speaker
}
segs = append(segs, entry)
}
doneEvent["segments"] = segs
}
@@ -512,6 +512,15 @@ type ConversationItemSoundDetectionEvent struct {
// The scored sound-event tags, in score-descending order.
Detections []SoundDetectionTag `json:"detections"`
// The start time of the detection window in seconds, when known. Set by
// the live scene-event path (a companion sound stream alongside live
// transcription); omitted by the unary/windowed sound-detection paths,
// which have no per-event timing.
Start *float64 `json:"start,omitempty"`
// The end time of the detection window in seconds, when known.
End *float64 `json:"end,omitempty"`
}
func (m ConversationItemSoundDetectionEvent) ServerEventType() ServerEventType {
@@ -586,11 +595,13 @@ type ConversationItemInputAudioTranscriptionSegmentEvent struct {
// The speaker label for the segment, if available.
Speaker string `json:"speaker,omitempty"`
// The start time of the segment in seconds.
Start float64 `json:"start,omitempty"`
// The start time of the segment in seconds. Always present (not
// omitempty: a segment starting at 0.0s must still carry "start").
Start float64 `json:"start"`
// The end time of the segment in seconds.
End float64 `json:"end,omitempty"`
// The end time of the segment in seconds. Always present (not
// omitempty: see Start).
End float64 `json:"end"`
// The text content of the segment.
Text string `json:"text,omitempty"`
@@ -0,0 +1,97 @@
import { test, expect } from './coverage-fixtures.js'
// Single-page PDF with a real text layer, built byte by byte so the xref
// offsets are valid and the spec needs no binary fixture.
function buildPdf(text) {
const stream = `BT /F1 18 Tf 20 100 Td (${text}) Tj ET`
const objs = [
'<< /Type /Catalog /Pages 2 0 R >>',
'<< /Type /Pages /Kids [3 0 R] /Count 1 >>',
'<< /Type /Page /Parent 2 0 R /MediaBox [0 0 300 200] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>',
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
'<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>',
]
let out = '%PDF-1.4\n'
const offsets = []
objs.forEach((body, i) => {
offsets.push(out.length)
out += `${i + 1} 0 obj\n${body}\nendobj\n`
})
const xref = out.length
out += `xref\n0 ${objs.length + 1}\n0000000000 65535 f \n`
for (const o of offsets) out += `${String(o).padStart(10, '0')} 00000 n \n`
out += `trailer\n<< /Size ${objs.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`
return Buffer.from(out, 'latin1')
}
async function openChat(page) {
await page.route('**/api/models/capabilities', (route) => {
route.fulfill({
contentType: 'application/json',
body: JSON.stringify({ data: [{ id: 'test-model', capabilities: ['FLAG_CHAT'] }] }),
})
})
await page.goto('/app/chat')
await expect(page.getByRole('button', { name: 'test-model' })).toBeVisible({ timeout: 10_000 })
}
test.describe('Chat - PDF attachments', () => {
test('sends the extracted text layer, not the raw PDF bytes', async ({ page }) => {
let requestBody = ''
await page.route('**/v1/chat/completions', (route) => {
requestBody = route.request().postData() || ''
route.fulfill({ status: 500, contentType: 'application/json', body: JSON.stringify({ error: { message: 'stop' } }) })
})
await openChat(page)
await page.locator('input[type=file]').setInputFiles({
name: 'report.pdf',
mimeType: 'application/pdf',
buffer: buildPdf('Quarterly revenue grew 42 percent'),
})
await expect(page.locator('.chat-file-name', { hasText: 'report.pdf' })).toBeVisible()
await page.locator('.chat-input').fill('Summarize')
await page.locator('.chat-send-btn').click()
await expect.poll(() => requestBody).toContain('Quarterly revenue grew 42 percent')
expect(requestBody).toContain('File: report.pdf')
expect(requestBody).not.toContain('%PDF')
})
test('rejects a PDF that cannot be parsed instead of attaching garbage', async ({ page }) => {
await openChat(page)
await page.locator('input[type=file]').setInputFiles({
name: 'broken.pdf',
mimeType: 'application/pdf',
buffer: Buffer.from('%PDF-1.4 this is not a real document'),
})
await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 })
await expect(page.locator('.chat-file-name', { hasText: 'broken.pdf' })).toHaveCount(0)
})
})
test.describe('Home - PDF attachments', () => {
test('attaches a PDF that has a text layer', async ({ page }) => {
await page.goto('/app')
await page.locator('input[type=file][accept*="pdf"]').setInputFiles({
name: 'notes.pdf',
mimeType: 'application/pdf',
buffer: buildPdf('Meeting notes for Tuesday'),
})
await expect(page.locator('.home-file-tag', { hasText: 'notes.pdf' })).toBeVisible({ timeout: 10_000 })
})
test('rejects a PDF that cannot be parsed', async ({ page }) => {
await page.goto('/app')
await page.locator('input[type=file][accept*="pdf"]').setInputFiles({
name: 'broken.pdf',
mimeType: 'application/pdf',
buffer: Buffer.from('%PDF-1.4 this is not a real document'),
})
await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 })
await expect(page.locator('.home-file-tag')).toHaveCount(0)
})
})
@@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => {
await expect(installedPane(page)).toContainText('Worker one')
})
test('shows the decisions use case on a decision model', async ({ page }) => {
await page.route('**/api/models/capabilities', route => route.fulfill({
contentType: 'application/json',
body: JSON.stringify({
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }],
}),
}))
await page.goto('/app/models?view=installed&model=decider')
await expect(installedPane(page)).toContainText('decider')
await expect(installedPane(page)).toContainText('Decisions')
})
test('stops a running model with confirmation', async ({ page }) => {
await page.goto('/app/models?view=installed&model=alpha')
+271
View File
@@ -29,6 +29,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"i18next-http-backend": "^3.0.6",
"marked": "^15.0.7",
"pdfjs-dist": "^5.6.205",
"react": "^19.1.0",
"react-dom": "^19.1.0",
"react-i18next": "^17.0.6",
@@ -1021,6 +1022,256 @@
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz",
"integrity": "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug=="
},
"node_modules/@napi-rs/canvas": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas/-/canvas-0.1.100.tgz",
"integrity": "sha512-xglYA6q3XO5P3BNJYxVZ1IV7DLVjp1Py6nwag88YntrS+3vKHyYcMqXVS4ZztJmwz2uGvz1FWhI/4LgbR5uQDA==",
"license": "MIT",
"optional": true,
"workspaces": [
"e2e/*"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
},
"optionalDependencies": {
"@napi-rs/canvas-android-arm64": "0.1.100",
"@napi-rs/canvas-darwin-arm64": "0.1.100",
"@napi-rs/canvas-darwin-x64": "0.1.100",
"@napi-rs/canvas-linux-arm-gnueabihf": "0.1.100",
"@napi-rs/canvas-linux-arm64-gnu": "0.1.100",
"@napi-rs/canvas-linux-arm64-musl": "0.1.100",
"@napi-rs/canvas-linux-riscv64-gnu": "0.1.100",
"@napi-rs/canvas-linux-x64-gnu": "0.1.100",
"@napi-rs/canvas-linux-x64-musl": "0.1.100",
"@napi-rs/canvas-win32-arm64-msvc": "0.1.100",
"@napi-rs/canvas-win32-x64-msvc": "0.1.100"
}
},
"node_modules/@napi-rs/canvas-android-arm64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-android-arm64/-/canvas-android-arm64-0.1.100.tgz",
"integrity": "sha512-hjhCKhntPv9+t4ckHymdx0phYNcVW+GKQR6Lzw2zE+pOVjOplSmtx9nNNknTjbEDLcuLZqA1y8ufKg1XfgftzQ==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"android"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-darwin-arm64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-arm64/-/canvas-darwin-arm64-0.1.100.tgz",
"integrity": "sha512-2PcswRaC7Ly645DGt88///zuFDhJxJYdKAs1uU3mfk1atYkXufgcgLfBpk6Tm12nCQBaNt1wpybuPZ4qOhTo8A==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-darwin-x64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-x64/-/canvas-darwin-x64-0.1.100.tgz",
"integrity": "sha512-ePNZtj7pNIva/siZMg+HmbeozkIjqUIYdoymH8HaA3qK7LfzFN4WMBM8G6HQ9ZC+H3+Dnn5pqtiXpgLykaPOhw==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm-gnueabihf": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm-gnueabihf/-/canvas-linux-arm-gnueabihf-0.1.100.tgz",
"integrity": "sha512-d5cDB48oWFGU8/XPhUOFAlySgb/VAu7D+s8fi55K1Pcfg8aPplHWqMgibhVLU8ky7Pyg/fuiVLz4Nf3JrSTuUA==",
"cpu": [
"arm"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-gnu/-/canvas-linux-arm64-gnu-0.1.100.tgz",
"integrity": "sha512-rDxgxRu69RvDlX/bh9o22DxLsGr8EqsNgotL9+RwQE1S0b0cqeatqsw6aW45mukm0B42DIAaAacKaYQ8cqS1nw==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm64-musl": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-musl/-/canvas-linux-arm64-musl-0.1.100.tgz",
"integrity": "sha512-K3mDW66N+xT2/V439u1alFANiBUjdEx2gLiNYnCmUsva5jZMxWTjafBYwTzYK+EMFMHrUoabuU+T1BIP5CgbYQ==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-riscv64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-riscv64-gnu/-/canvas-linux-riscv64-gnu-0.1.100.tgz",
"integrity": "sha512-mooqUBTIsccZpnoQC4NgrC1v6C1vof39etLNMnBwCY+p0gajWJvAHLGQ6g/gGyS5YrpDW+GefSN4+Cvcr08UWw==",
"cpu": [
"riscv64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-x64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-gnu/-/canvas-linux-x64-gnu-0.1.100.tgz",
"integrity": "sha512-1eCvkDCazm7FFhsT7DfGOdSaHgZVK3bt/dSBl5EWHOWmnz+I7j8tPseJqqD81NF+MH21jKUK4wQSDjN0mdhnTg==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-x64-musl": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-musl/-/canvas-linux-x64-musl-0.1.100.tgz",
"integrity": "sha512-20arT6lnI19S68qNlii73TSEDbECNgzMz2EpldC1V3mZFuRkeujXkcebRk0LRJe9SEUAooYiLokfMViY8IX7yA==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-win32-arm64-msvc": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-arm64-msvc/-/canvas-win32-arm64-msvc-0.1.100.tgz",
"integrity": "sha512-DZFFT1wIAg37LJw37yhMRFfjATd3vTQzjZ1Yki8u2vhO6Hi5VE6BVaGQ1aaDu7xb4iMErz+9EOwjpS7xcxFeBw==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-win32-x64-msvc": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-x64-msvc/-/canvas-win32-x64-msvc-0.1.100.tgz",
"integrity": "sha512-MyT1j3mHC2+Lu4pBi9mKyMJhtP6U7k7EldY7sj/uS5gJA65gTXt8MefJQXLJo5d/vZbuWmfxzkEUNc/urV3pHA==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/wasm-runtime": {
"version": "1.1.5",
"resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.1.5.tgz",
@@ -5219,6 +5470,13 @@
"node": ">=8"
}
},
"node_modules/node-readable-to-web-readable-stream": {
"version": "0.4.2",
"resolved": "https://registry.npmjs.org/node-readable-to-web-readable-stream/-/node-readable-to-web-readable-stream-0.4.2.tgz",
"integrity": "sha512-/cMZNI34v//jUTrI+UIo4ieHAB5EZRY/+7OmXZgBxaWBMcW2tGdceIw06RFxWxrKZ5Jp3sI2i5TsRo+CBhtVLQ==",
"license": "MIT",
"optional": true
},
"node_modules/node-releases": {
"version": "2.0.54",
"resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.54.tgz",
@@ -5752,6 +6010,19 @@
"url": "https://opencollective.com/express"
}
},
"node_modules/pdfjs-dist": {
"version": "5.6.205",
"resolved": "https://registry.npmjs.org/pdfjs-dist/-/pdfjs-dist-5.6.205.tgz",
"integrity": "sha512-tlUj+2IDa7G1SbvBNN74UHRLJybZDWYom+k6p5KIZl7huBvsA4APi6mKL+zCxd3tLjN5hOOEE9Tv7VdzO88pfg==",
"license": "Apache-2.0",
"engines": {
"node": ">=20.19.0 || >=22.13.0 || >=24"
},
"optionalDependencies": {
"@napi-rs/canvas": "^0.1.96",
"node-readable-to-web-readable-stream": "^0.4.2"
}
},
"node_modules/picocolors": {
"version": "1.1.1",
"resolved": "https://registry.npmjs.org/picocolors/-/picocolors-1.1.1.tgz",
+1
View File
@@ -45,6 +45,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"i18next-http-backend": "^3.0.6",
"marked": "^15.0.7",
"pdfjs-dist": "^5.6.205",
"react": "^19.1.0",
"react-dom": "^19.1.0",
"react-i18next": "^17.0.6",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -117,7 +117,8 @@
"copied": "Copied to clipboard",
"copyFailed": "Could not copy to clipboard",
"chatCopied": "Chat copied to clipboard",
"forked": "Created a new chat"
"forked": "Created a new chat",
"pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged."
},
"menu": {
"trigger": "Chats",
@@ -35,7 +35,8 @@
"enterToSend": "Enter to send",
"selectModelFirst": "Select a model first",
"sendMessage": "Send message",
"selectModelToast": "Please select a model first"
"selectModelToast": "Please select a model first",
"pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged."
},
"quickLinks": {
"manageByChat": "Manage by chat",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
+8 -2
View File
@@ -9,6 +9,7 @@ import { extractCodeArtifacts, renderMarkdownWithArtifacts } from '../utils/arti
import CanvasPanel from '../components/CanvasPanel'
import Toggle from '../components/Toggle'
import { fileToBase64, modelsApi, mcpApi } from '../utils/api'
import { readAttachmentText } from '../utils/pdf'
import { CAP_CHAT } from '../utils/capabilities'
import { useMCPClient } from '../hooks/useMCPClient'
import MCPAppFrame from '../components/MCPAppFrame'
@@ -842,13 +843,18 @@ export default function Chat() {
const base64 = await fileToBase64(file)
const entry = { name: file.name, type: file.type, base64 }
if (!file.type.startsWith('image/') && !file.type.startsWith('audio/') && !file.type.startsWith('video/')) {
entry.textContent = await file.text().catch(() => '')
try {
entry.textContent = await readAttachmentText(file)
} catch {
addToast(t('toasts.pdfReadFailed', { name: file.name }), 'error')
continue
}
}
newFiles.push(entry)
}
setFiles(prev => [...prev, ...newFiles])
e.target.value = ''
}, [])
}, [addToast, t])
const handlePaste = useCallback(async (e) => {
const items = e.clipboardData?.items
+8 -2
View File
@@ -12,6 +12,7 @@ import HomeConnect from '../components/HomeConnect'
import { useResources } from '../hooks/useResources'
import { usePolling } from '../hooks/usePolling'
import { fileToBase64, backendControlApi, systemApi, modelsApi, mcpApi, nodesApi } from '../utils/api'
import { readAttachmentText } from '../utils/pdf'
import { API_CONFIG } from '../utils/config'
import { greetingKey } from '../utils/greeting'
import StatusPill from '../components/StatusPill'
@@ -158,12 +159,17 @@ export default function Home() {
const base64 = await fileToBase64(file)
const entry = { name: file.name, type: file.type, base64 }
if (!file.type.startsWith('image/') && !file.type.startsWith('audio/')) {
entry.textContent = await file.text().catch(() => '')
try {
entry.textContent = await readAttachmentText(file)
} catch {
addToast(t('input.pdfReadFailed', { name: file.name }), 'error')
continue
}
}
newFiles.push(entry)
}
setter(prev => [...prev, ...newFiles])
}, [])
}, [addToast, t])
const removeFile = useCallback((file) => {
const removeFn = (prev) => prev.filter(f => f !== file)
@@ -22,7 +22,7 @@ import {
CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS,
CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION,
CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK,
CAP_VAD, CAP_SCORE,
CAP_VAD, CAP_SCORE, CAP_DECISIONS,
} from '../utils/capabilities'
const USE_CASES = [
@@ -39,6 +39,7 @@ const USE_CASES = [
{ cap: CAP_RERANK, labelKey: 'rerank' },
{ cap: CAP_VAD, labelKey: 'vad' },
{ cap: CAP_SCORE, labelKey: 'score' },
{ cap: CAP_DECISIONS, labelKey: 'decisions' },
]
export function modelUseCases(model) {
+1
View File
@@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION'
export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM'
export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO'
export const CAP_SCORE = 'FLAG_SCORE'
export const CAP_DECISIONS = 'FLAG_DECISIONS'
export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY'
+48
View File
@@ -0,0 +1,48 @@
export function isPdf(file) {
return file?.type === 'application/pdf' || /\.pdf$/i.test(file?.name || '')
}
// pdf.js and its worker are loaded on first use so the main bundle does not
// pay for them when nobody attaches a PDF.
async function loadPdfjs() {
const [pdfjs, worker] = await Promise.all([
import('pdfjs-dist'),
import('pdfjs-dist/build/pdf.worker.min.mjs?url'),
])
pdfjs.GlobalWorkerOptions.workerSrc = worker.default
return pdfjs
}
// Returns the text layer of every page, one block per page. Throws when the
// file cannot be parsed or has no text layer (scanned PDFs): sending an empty
// attachment to the model would look like success and silently lose the file.
export async function extractPdfText(file) {
const pdfjs = await loadPdfjs()
const data = new Uint8Array(await file.arrayBuffer())
const doc = await pdfjs.getDocument({ data }).promise
try {
const pages = []
for (let i = 1; i <= doc.numPages; i++) {
const page = await doc.getPage(i)
const content = await page.getTextContent()
let text = ''
for (const item of content.items) {
text += item.str
text += item.hasEOL ? '\n' : ''
}
pages.push(text.trim())
}
const text = pages.filter(Boolean).join('\n\n')
if (!text) throw new Error('PDF has no extractable text')
return text
} finally {
await doc.destroy()
}
}
// Text of an attached non-media file. PDFs go through pdf.js; everything else
// is read as UTF-8.
export async function readAttachmentText(file) {
if (isPdf(file)) return extractPdfText(file)
return file.text().catch(() => '')
}