feat(transcription): carry speaker labels on words and streamed segments

A diarizing backend could label transcript segments, but two paths
dropped the label: TranscriptWord had no speaker field, so live
transcription words and word-level timestamps could not carry one, and
the stream=true transcript.text.done event left the speaker out of
its segments.

TranscriptWord gains an optional speaker (proto field 4, additive).
It flows through the live event and result mapping, the JSON word
output of the endpoint and the CLI, and transcript.text.done now
includes a segment's speaker when there is one. Empty labels are
omitted, so responses without diarization are unchanged.

Assisted-by: Claude:claude-opus-5-5 [Claude Code]
This commit is contained in:
Ettore Di Giacinto committed 2026-09-28 09:40:09 +00:00
1 parent 761505237d
commit 2f0049f979
7 files changed
+53 -26

No files matched your search

+1
View File
@@ -647,6 +647,7 @@ message TranscriptWord {
int64 start = 1;
int64 end = 2;
string text = 3;
string speaker = 4; // backend speaker label when diarizing; empty otherwise
}
message TranscriptSegment {
+4 -3
View File
@@ -213,9 +213,10 @@ func transcriptResultFromProto(r *proto.TranscriptResult) *schema.TranscriptionR
var words []schema.TranscriptionWord
for _, w := range s.Words {
var word = schema.TranscriptionWord{
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Speaker: w.Speaker,
}
words = append(words, word)
tr.Words = append(tr.Words, word)
+4 -3
View File
@@ -298,9 +298,10 @@ func liveEventFromProto(r *proto.TranscriptLiveResponse) LiveTranscriptionEvent
}
for _, w := range r.GetWords() {
ev.Words = append(ev.Words, schema.TranscriptionWord{
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Speaker: w.Speaker,
})
}
if r.GetFinalResult() != nil {
@@ -54,6 +54,20 @@ var _ = Describe("liveEventFromProto", func() {
Expect(ev.Final).To(BeNil())
})
It("carries word speakers and final segment speakers from a diarizing backend", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{
Words: []*proto.TranscriptWord{{Text: "hi", Speaker: "1"}},
})
Expect(ev.Words[0].Speaker).To(Equal("1"))
ev = liveEventFromProto(&proto.TranscriptLiveResponse{
FinalResult: &proto.TranscriptResult{
Text: "hi there",
Segments: []*proto.TranscriptSegment{{Text: "hi", Speaker: "0"}, {Text: "there", Speaker: "1"}},
},
})
Expect(ev.Final.Segments[1].Speaker).To(Equal("1"))
})
It("maps the eob backchannel flag separately from eou", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "uh-huh", Eob: true})
Expect(ev.Eob).To(BeTrue())
+8 -6
View File
@@ -93,18 +93,20 @@ func (t *TranscriptCMD) Run(ctx *cliContext.Context) error {
}
for _, word := range(tr.Words) {
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
for _, seg := range(tr.Segments) {
segWords := []schema.TranscriptionWordSeconds{}
for _, word := range(seg.Words) {
segWords = append(segWords, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
+14 -8
View File
@@ -191,18 +191,20 @@ func TranscriptEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
}
for _, word := range tr.Words {
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
for _, seg := range tr.Segments {
segWords := []schema.TranscriptionWordSeconds{}
for _, word := range seg.Words {
segWords = append(segWords, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
@@ -309,12 +311,16 @@ func streamTranscription(c echo.Context, req backend.TranscriptionRequest, ml *m
if len(finalResult.Segments) > 0 {
segs := make([]map[string]any, 0, len(finalResult.Segments))
for _, seg := range finalResult.Segments {
segs = append(segs, map[string]any{
entry := map[string]any{
"id": seg.Id,
"start": seg.Start.Seconds(),
"end": seg.End.Seconds(),
"text": seg.Text,
})
}
if seg.Speaker != "" {
entry["speaker"] = seg.Speaker
}
segs = append(segs, entry)
}
doneEvent["segments"] = segs
}
+8 -6
View File
@@ -13,9 +13,10 @@ type TranscriptionSegment struct {
}
type TranscriptionWord struct {
Start time.Duration `json:"start"`
End time.Duration `json:"end"`
Text string `json:"text"`
Start time.Duration `json:"start"`
End time.Duration `json:"end"`
Text string `json:"text"`
Speaker string `json:"speaker,omitempty"`
}
type TranscriptionResult struct {
@@ -42,9 +43,10 @@ type TranscriptionSegmentSeconds struct {
}
type TranscriptionWordSeconds struct {
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Speaker string `json:"speaker,omitempty"`
}
type TranscriptionResultSeconds struct {