diff --git a/backend/cpp/audio-cpp/transcript_assembly.cpp b/backend/cpp/audio-cpp/transcript_assembly.cpp index 0bf0987e6..887dad112 100644 --- a/backend/cpp/audio-cpp/transcript_assembly.cpp +++ b/backend/cpp/audio-cpp/transcript_assembly.cpp @@ -51,21 +51,39 @@ std::string speaker_for(const Span &segment, return best; } +// The chosen segmentation. labels is empty unless the spans were sourced from +// the speaker turns themselves, in which case it is parallel to spans and holds +// the label each span arrived with. +struct SegmentSource { + std::vector spans; + std::vector labels; +}; + // Chooses the segment spans, per the documented precedence. -std::vector choose_segment_spans(const std::string &text_output, - const std::vector &speech_segments, - const std::vector &speaker_turns, - const std::vector &words) { +SegmentSource choose_segment_spans(const std::string &text_output, + const std::vector &speech_segments, + const std::vector &speaker_turns, + const std::vector &words) { if (!speech_segments.empty()) { - return speech_segments; + return {speech_segments, {}}; } if (!speaker_turns.empty()) { - std::vector spans; - spans.reserve(speaker_turns.size()); + // The labels are carried out rather than re-derived by overlap later. A + // turn wholly contained in another speaker's turn overlaps its own span + // completely, which is the largest overlap possible, so it can only tie + // with the containing turn and would then lose that tie on order. + // sortformer_diar binarizes each speaker's track independently and + // sorts the result by start sample, so the container always comes + // first, and the interjecting speaker would be silently relabelled to + // the speaker it interrupted. + SegmentSource source; + source.spans.reserve(speaker_turns.size()); + source.labels.reserve(speaker_turns.size()); for (const auto &turn : speaker_turns) { - spans.push_back(turn.span); + source.spans.push_back(turn.span); + source.labels.push_back(turn.speaker); } - return spans; + return source; } if (!words.empty()) { Span covering = words.front().span; @@ -74,12 +92,12 @@ std::vector choose_segment_spans(const std::string &text_output, std::min(covering.start_sample, word.span.start_sample); covering.end_sample = std::max(covering.end_sample, word.span.end_sample); } - return {covering}; + return {{covering}, {}}; } if (!text_output.empty()) { // A zero span rather than a fabricated duration: the model reported no // timing, and inventing one would be a lie the caller cannot detect. - return {Span{0, 0}}; + return {{Span{0, 0}}, {}}; } return {}; } @@ -116,8 +134,9 @@ AssembledTranscript assemble_transcript(const std::string &text_output, // THE RULE. Never derived from spans. assembled.text = text_output; - const std::vector spans = + const SegmentSource source = choose_segment_spans(text_output, speech_segments, speaker_turns, words); + const std::vector &spans = source.spans; if (spans.empty()) { return assembled; } @@ -128,7 +147,11 @@ AssembledTranscript assemble_transcript(const std::string &text_output, segment.id = static_cast(i); segment.start_ns = samples_to_nanoseconds(spans[i].start_sample, sample_rate); segment.end_ns = samples_to_nanoseconds(spans[i].end_sample, sample_rate); - segment.speaker = speaker_for(spans[i], speaker_turns); + // A segment that came from a speaker turn already knows its speaker. + // Only the other three sources have to look one up by overlap. + segment.speaker = source.labels.empty() + ? speaker_for(spans[i], speaker_turns) + : source.labels[i]; } for (const auto &word : words) { diff --git a/backend/cpp/audio-cpp/transcript_assembly.h b/backend/cpp/audio-cpp/transcript_assembly.h index f5866bdc2..4b3a3d088 100644 --- a/backend/cpp/audio-cpp/transcript_assembly.h +++ b/backend/cpp/audio-cpp/transcript_assembly.h @@ -64,7 +64,13 @@ struct AssembledTranscript { // outside every segment attaches to the nearest one by midpoint distance so it // is never silently dropped. A segment's text is its words joined by a single // space, except that a lone segment with no words carries the full text. -// A segment's speaker is the speaker turn with the greatest overlap. +// +// A segment's speaker is the speaker turn with the greatest overlap, except +// when the segments came from the speaker turns themselves (source 2), where +// each segment keeps its own turn's label. Re-deriving it there loses a turn +// nested inside another speaker's turn: the nested turn overlaps its own span +// completely, so it can only tie with the containing turn, which is listed +// first and wins the tie. AssembledTranscript assemble_transcript(const std::string &text_output, const std::vector &speech_segments, const std::vector &speaker_turns, diff --git a/backend/cpp/audio-cpp/transcript_assembly_test.cpp b/backend/cpp/audio-cpp/transcript_assembly_test.cpp index 889418b4f..16fe9e2e9 100644 --- a/backend/cpp/audio-cpp/transcript_assembly_test.cpp +++ b/backend/cpp/audio-cpp/transcript_assembly_test.cpp @@ -2,9 +2,12 @@ // compiles this as a single translation unit, so both implementations are // included directly rather than linked. // -// Every fixture below mirrors a producer shape actually observed from -// audio.cpp families. Do not replace them with invented shapes: an invented -// shape is what let the earlier attempt ship an empty transcript. +// Every fixture below either mirrors a producer shape actually observed from +// audio.cpp families, and names the families it was checked against, or says in +// its own comment that it is defensive. Do not replace an observed shape with an +// invented one and do not quietly promote a defensive fixture to an observed +// one: an invented shape is what let the earlier attempt ship an empty +// transcript. #include "audio_units.cpp" #include "transcript_assembly.cpp" @@ -117,6 +120,26 @@ static void test_covering_span_spans_every_word() { "the covering span reaches the latest word end, not the last word's"); } +// Defensive, not observed: no pinned family emits a word with no text. +// nemotron_asr's build_token_timestamps (models/nemotron_asr/decoder.cpp:97) +// skips a token that decodes to an empty chunk before it ever becomes a +// WordTimestamp. join_words guards against one anyway, and an unexercised guard +// is a guard the next reader deletes as dead weight. +static void test_empty_word_contributes_no_separator() { + const std::vector words = { + {{0, 4000}, "alpha"}, + {{4000, 8000}, ""}, + {{8000, 12000}, "beta"}, + }; + const auto out = assemble_transcript("alpha beta", {}, {}, words, kRate); + + check(out.segments.size() == 1, "one segment"); + check(segment_at(out, 0, "sole segment").text == "alpha beta", + "an empty word adds no separator to the segment text"); + check(segment_at(out, 0, "sole segment").words.size() == 3, + "the empty word still reports its span"); +} + // Shape B: speech segments, no words. Emitted by ASR families that report // utterance boundaries without word-level timing. static void test_segments_without_words() { @@ -153,10 +176,14 @@ static void test_segments_with_speaker_turns_no_words() { "second speaker assigned"); } -// Shape C': the same producer, but the utterance boundaries and the speaker -// boundaries disagree. VibeVoice chunk merging shifts both lists independently, -// so they need not line up. speech_segments is the segment source and speaker -// turns only label it, which the matching-span fixture above cannot show. +// Defensive, not observed: speech segments and speaker turns that disagree. +// vibevoice_asr builds each SpeakerTurn with turn.span = speech_segment.span in +// one loop (models/vibevoice_asr/session.cpp:965) and shifts and clips both +// lists identically when merging chunks, so in practice the two lists are 1:1 +// with identical spans. That is exactly why the shape C fixture above cannot +// show which list is the segment source: swapping the precedence there produces +// byte-identical output. This fixture pins the precedence, and it is the shape +// any future family that segments and diarizes separately would produce. static void test_speech_segments_outrank_speaker_turns() { const std::vector segments = {{0, 32000}}; const std::vector turns = { @@ -171,8 +198,9 @@ static void test_speech_segments_outrank_speaker_turns() { "the utterance keeps its own span"); } -// Speaker turns outrank word timestamps as a segment source: a diarized result -// is segmented by who spoke, and words only fill the turns in. +// Defensive, not observed: no pinned family emits speaker turns and word +// timestamps together. It pins rule 2 against rule 3, which nothing else does: +// a diarized result is segmented by who spoke, and words only fill the turns in. static void test_speaker_turns_outrank_words() { const std::vector turns = { {{0, 16000}, "SPEAKER_00"}, @@ -219,6 +247,28 @@ static void test_speaker_turns_only() { "second turn starts at 1.5s"); } +// Shape E', the same producer with one speaker talking over another. +// decode_sortformer_speaker_turns (models/sortformer_diar/postprocess.cpp) +// binarizes each speaker's probability track independently, which is the whole +// point of sortformer, then sorts the turns by start sample. So a turn can be +// wholly contained in another speaker's turn, and the containing turn always +// comes first. A segment sourced from a speaker turn must keep that turn's own +// label: re-deriving it by overlap can only ever tie with the containing turn, +// which then wins on order and silently erases the interjecting speaker. +static void test_nested_speaker_turn_keeps_its_own_label() { + const std::vector turns = { + {{0, 100000}, "speaker_0"}, + {{10000, 20000}, "speaker_1"}, + }; + const auto out = assemble_transcript("", {}, turns, {}, kRate); + + check(out.segments.size() == 2, "both turns become segments"); + check(segment_at(out, 0, "containing turn").speaker == "speaker_0", + "the containing turn keeps its label"); + check(segment_at(out, 1, "nested turn").speaker == "speaker_1", + "a turn nested inside another is not relabelled to the container"); +} + // Shape F: nothing at all. A model that ran but produced no output must not // crash or fabricate a segment. static void test_empty() { @@ -326,6 +376,31 @@ static void test_stray_word_goes_to_the_nearest_segment() { "the stray word attaches to the nearest segment"); } +// "Nearest" is measured from the segment's midpoint, and it is neither "the +// first segment" nor "the last". A leading stray word is the case a +// trailing-only fixture cannot reach: forced-aligner words scored against VAD +// segments produce one, and with only trailing coverage it would land at the +// end of the transcript with the suite green. Here the leading word's nearest +// midpoint is the first segment while its nearest start is the second, and the +// trailing word's nearest midpoint is the third while its nearest end is the +// second, so no endpoint rule reproduces this assignment either. +static void test_stray_word_distance_is_measured_from_the_midpoint() { + const std::vector segments = {{0, 2000}, {8000, 200000}, {300000, 302000}}; + const std::vector words = { + {{4000, 6000}, "lead"}, + {{249000, 251000}, "trail"}, + }; + const auto out = assemble_transcript("lead trail", segments, {}, words, kRate); + + check(out.segments.size() == 3, "three segments"); + check(segment_at(out, 0, "first segment").text == "lead", + "the leading stray word goes to the nearest segment by midpoint"); + check(segment_at(out, 2, "third segment").text == "trail", + "the trailing stray word goes to the nearest segment by midpoint"); + check(segment_at(out, 1, "middle segment").words.empty(), + "the long middle segment claims neither stray word"); +} + // Speaker assignment uses greatest overlap, not first match, so a turn that // barely touches a segment does not win over one that covers it. static void test_speaker_assigned_by_greatest_overlap() { @@ -366,18 +441,21 @@ int main() { test_words_only(); test_words_only_with_punctuated_text_output(); test_covering_span_spans_every_word(); + test_empty_word_contributes_no_separator(); test_segments_without_words(); test_segments_with_speaker_turns_no_words(); test_speech_segments_outrank_speaker_turns(); test_speaker_turns_outrank_words(); test_text_only(); test_speaker_turns_only(); + test_nested_speaker_turn_keeps_its_own_label(); test_empty(); test_vad_segments_without_text(); test_words_distributed_into_segments(); test_words_assigned_by_midpoint_not_endpoint(); test_word_outside_all_segments(); test_stray_word_goes_to_the_nearest_segment(); + test_stray_word_distance_is_measured_from_the_midpoint(); test_speaker_assigned_by_greatest_overlap(); test_segment_without_any_overlapping_turn_has_no_speaker(); test_zero_sample_rate_is_safe();