mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-02 19:14:38 -04:00
feat(transcription): carry speaker labels on words and streamed segments
A diarizing backend could label transcript segments, but two paths dropped the label: TranscriptWord had no speaker field, so live transcription words and word-level timestamps could not carry one, and the stream=true transcript.text.done event left the speaker out of its segments. TranscriptWord gains an optional speaker (proto field 4, additive). It flows through the live event and result mapping, the JSON word output of the endpoint and the CLI, and transcript.text.done now includes a segment's speaker when there is one. Empty labels are omitted, so responses without diarization are unchanged. Assisted-by: Claude:claude-opus-5-5 [Claude Code]
This commit is contained in:
1 parent
761505237d
commit
2f0049f979
7 files changed
+53
-26
No files matched your search
@@ -647,6 +647,7 @@ message TranscriptWord {
|
||||
int64 start = 1;
|
||||
int64 end = 2;
|
||||
string text = 3;
|
||||
string speaker = 4; // backend speaker label when diarizing; empty otherwise
|
||||
}
|
||||
|
||||
message TranscriptSegment {
|
||||
|
||||
@@ -213,9 +213,10 @@ func transcriptResultFromProto(r *proto.TranscriptResult) *schema.TranscriptionR
|
||||
var words []schema.TranscriptionWord
|
||||
for _, w := range s.Words {
|
||||
var word = schema.TranscriptionWord{
|
||||
Start: time.Duration(w.Start),
|
||||
End: time.Duration(w.End),
|
||||
Text: w.Text,
|
||||
Start: time.Duration(w.Start),
|
||||
End: time.Duration(w.End),
|
||||
Text: w.Text,
|
||||
Speaker: w.Speaker,
|
||||
}
|
||||
words = append(words, word)
|
||||
tr.Words = append(tr.Words, word)
|
||||
|
||||
@@ -298,9 +298,10 @@ func liveEventFromProto(r *proto.TranscriptLiveResponse) LiveTranscriptionEvent
|
||||
}
|
||||
for _, w := range r.GetWords() {
|
||||
ev.Words = append(ev.Words, schema.TranscriptionWord{
|
||||
Start: time.Duration(w.Start),
|
||||
End: time.Duration(w.End),
|
||||
Text: w.Text,
|
||||
Start: time.Duration(w.Start),
|
||||
End: time.Duration(w.End),
|
||||
Text: w.Text,
|
||||
Speaker: w.Speaker,
|
||||
})
|
||||
}
|
||||
if r.GetFinalResult() != nil {
|
||||
|
||||
@@ -54,6 +54,20 @@ var _ = Describe("liveEventFromProto", func() {
|
||||
Expect(ev.Final).To(BeNil())
|
||||
})
|
||||
|
||||
It("carries word speakers and final segment speakers from a diarizing backend", func() {
|
||||
ev := liveEventFromProto(&proto.TranscriptLiveResponse{
|
||||
Words: []*proto.TranscriptWord{{Text: "hi", Speaker: "1"}},
|
||||
})
|
||||
Expect(ev.Words[0].Speaker).To(Equal("1"))
|
||||
ev = liveEventFromProto(&proto.TranscriptLiveResponse{
|
||||
FinalResult: &proto.TranscriptResult{
|
||||
Text: "hi there",
|
||||
Segments: []*proto.TranscriptSegment{{Text: "hi", Speaker: "0"}, {Text: "there", Speaker: "1"}},
|
||||
},
|
||||
})
|
||||
Expect(ev.Final.Segments[1].Speaker).To(Equal("1"))
|
||||
})
|
||||
|
||||
It("maps the eob backchannel flag separately from eou", func() {
|
||||
ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "uh-huh", Eob: true})
|
||||
Expect(ev.Eob).To(BeTrue())
|
||||
|
||||
@@ -93,18 +93,20 @@ func (t *TranscriptCMD) Run(ctx *cliContext.Context) error {
|
||||
}
|
||||
for _, word := range(tr.Words) {
|
||||
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Speaker: word.Speaker,
|
||||
})
|
||||
}
|
||||
for _, seg := range(tr.Segments) {
|
||||
segWords := []schema.TranscriptionWordSeconds{}
|
||||
for _, word := range(seg.Words) {
|
||||
segWords = append(segWords, schema.TranscriptionWordSeconds{
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Speaker: word.Speaker,
|
||||
})
|
||||
}
|
||||
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
|
||||
|
||||
@@ -191,18 +191,20 @@ func TranscriptEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
|
||||
}
|
||||
for _, word := range tr.Words {
|
||||
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Speaker: word.Speaker,
|
||||
})
|
||||
}
|
||||
for _, seg := range tr.Segments {
|
||||
segWords := []schema.TranscriptionWordSeconds{}
|
||||
for _, word := range seg.Words {
|
||||
segWords = append(segWords, schema.TranscriptionWordSeconds{
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Start: word.Start.Seconds(),
|
||||
End: word.End.Seconds(),
|
||||
Text: word.Text,
|
||||
Speaker: word.Speaker,
|
||||
})
|
||||
}
|
||||
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
|
||||
@@ -309,12 +311,16 @@ func streamTranscription(c echo.Context, req backend.TranscriptionRequest, ml *m
|
||||
if len(finalResult.Segments) > 0 {
|
||||
segs := make([]map[string]any, 0, len(finalResult.Segments))
|
||||
for _, seg := range finalResult.Segments {
|
||||
segs = append(segs, map[string]any{
|
||||
entry := map[string]any{
|
||||
"id": seg.Id,
|
||||
"start": seg.Start.Seconds(),
|
||||
"end": seg.End.Seconds(),
|
||||
"text": seg.Text,
|
||||
})
|
||||
}
|
||||
if seg.Speaker != "" {
|
||||
entry["speaker"] = seg.Speaker
|
||||
}
|
||||
segs = append(segs, entry)
|
||||
}
|
||||
doneEvent["segments"] = segs
|
||||
}
|
||||
|
||||
@@ -13,9 +13,10 @@ type TranscriptionSegment struct {
|
||||
}
|
||||
|
||||
type TranscriptionWord struct {
|
||||
Start time.Duration `json:"start"`
|
||||
End time.Duration `json:"end"`
|
||||
Text string `json:"text"`
|
||||
Start time.Duration `json:"start"`
|
||||
End time.Duration `json:"end"`
|
||||
Text string `json:"text"`
|
||||
Speaker string `json:"speaker,omitempty"`
|
||||
}
|
||||
|
||||
type TranscriptionResult struct {
|
||||
@@ -42,9 +43,10 @@ type TranscriptionSegmentSeconds struct {
|
||||
}
|
||||
|
||||
type TranscriptionWordSeconds struct {
|
||||
Start float64 `json:"start"`
|
||||
End float64 `json:"end"`
|
||||
Text string `json:"text"`
|
||||
Start float64 `json:"start"`
|
||||
End float64 `json:"end"`
|
||||
Text string `json:"text"`
|
||||
Speaker string `json:"speaker,omitempty"`
|
||||
}
|
||||
|
||||
type TranscriptionResultSeconds struct {
|
||||
|
||||
Reference in new issue
Block a user