Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -95,6 +95,9 @@ result-*
/tools/stt-eval/whispercpp-dtw-poc/fixtures/*.wav
/tools/stt-eval/whispercpp-dtw-poc/results/

# Word-timing harness (tools/stt-eval/word-timing): TTS corpus and results.
/tools/stt-eval/word-timing/data/

opencode.json
opencode.json

Expand Down
46 changes: 31 additions & 15 deletions electron/native/whisper-stt/src/main.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -223,9 +223,12 @@ std::string detect_active_backend() {
}

struct Word {
double start = 0.0;
double end = 0.0;
double prob = 0.0;
double start = 0.0;
double end = 0.0;
// t_dtw of the word's first token: the end of that token, so always inside
// the word. Only used to tell which stretch of speech a word belongs to.
double anchor = 0.0;
double prob = 0.0;
std::string text;
};

Expand Down Expand Up @@ -517,18 +520,28 @@ int main(int argc, char** argv) {
};
std::vector<Segment> segments;

// whisper.cpp writes a token's t_dtw when the DTW path enters the decoder
// row that PREDICTS the next token (whisper_exp_compute_token_level_
// timestamps_dtw, v1.9.1): it marks the END of the token, not its start.
// So a word runs from the t_dtw of the text token before it to the t_dtw
// of its own last token. Taking its first token's t_dtw as the start put
// every word one token late (+175 ms median, tools/stt-eval/word-timing).
// The previous token carries across segments; the request's very first
// word has none and starts on its first token's time (the Node side
// anchors it on the speech onset anyway).
double prev_tok_end = -1.0;
const int n_segments = input.empty() ? 0 : whisper_full_n_segments(ctx);
for (int si = 0; si < n_segments; ++si) {
Segment seg;
seg.start = to_original_sec(whisper_full_get_segment_t0(ctx, si), kept);
seg.end = to_original_sec(whisper_full_get_segment_t1(ctx, si), kept);
if (const char* t = whisper_full_get_segment_text(ctx, si)) seg.text = t;

struct W { double t_dtw_first; double p_sum; int p_n; std::string text; };
struct W { double start; double end; double anchor; double p_sum; int p_n; std::string text; };
std::vector<W> word_buf;
std::string cur_text;
bool in_word = false;
double w_first_t_dtw = 0;
double w_start = 0, w_end = 0, w_anchor = 0;
double w_p_sum = 0; int w_p_n = 0;

const int n_tokens = whisper_full_n_tokens(ctx, si);
Expand All @@ -555,30 +568,32 @@ int main(int argc, char** argv) {
const double td_dtw = to_original_sec(td.t_dtw >= 0 ? td.t_dtw : 0, kept);

if (starts_word && in_word) {
word_buf.push_back({ w_first_t_dtw, w_p_sum, w_p_n, cur_text });
word_buf.push_back({ w_start, w_end, w_anchor, w_p_sum, w_p_n, cur_text });
w_p_sum = 0; w_p_n = 0; cur_text.clear();
}
if (starts_word) {
in_word = true;
w_first_t_dtw = td_dtw;
w_start = prev_tok_end >= 0 ? prev_tok_end : td_dtw;
w_anchor = td_dtw;
cur_text = (!raw.empty() && raw[0] == ' ') ? raw.substr(1) : raw;
} else {
cur_text += raw;
}
w_end = td_dtw;
prev_tok_end = td_dtw;
w_p_sum += td.p;
w_p_n += 1;
}
if (in_word) {
word_buf.push_back({ w_first_t_dtw, w_p_sum, w_p_n, cur_text });
word_buf.push_back({ w_start, w_end, w_anchor, w_p_sum, w_p_n, cur_text });
}
for (size_t wi = 0; wi < word_buf.size(); ++wi) {
for (const W& b : word_buf) {
Word w;
w.start = word_buf[wi].t_dtw_first;
w.end = (wi + 1 < word_buf.size())
? word_buf[wi + 1].t_dtw_first
: seg.end;
w.prob = word_buf[wi].p_sum / std::max(1, word_buf[wi].p_n);
w.text = word_buf[wi].text;
w.start = b.start;
w.end = b.end;
w.anchor = b.anchor;
w.prob = b.p_sum / std::max(1, b.p_n);
w.text = b.text;
seg.words.push_back(w);
}
segments.push_back(std::move(seg));
Expand Down Expand Up @@ -642,6 +657,7 @@ int main(int argc, char** argv) {
{"word", w.text},
{"start", w.start},
{"end", w.end},
{"anchor", w.anchor},
{"probability", w.prob},
});
}
Expand Down
207 changes: 53 additions & 154 deletions electron/stt/snapWordBoundaries.test.ts
Original file line number Diff line number Diff line change
@@ -1,199 +1,98 @@
import { describe, expect, it } from "vitest";
import { snapWordBoundariesToAudio } from "./snapWordBoundaries";
import { anchorWordsOnSpeech, type HelperWord } from "./snapWordBoundaries";
import type { SttWordSegment } from "./transcriptionContract";

const SAMPLE_RATE = 16_000;

/** Mono 16 kHz buffer that is loud everywhere except the given silent spans. */
function audioWithSilences(durationSec: number, silences: Array<[number, number]>): Float32Array {
const samples = new Float32Array(Math.round(durationSec * SAMPLE_RATE));
for (let i = 0; i < samples.length; i++) {
const t = i / SAMPLE_RATE;
const silent = silences.some(([from, to]) => t >= from && t < to);
// Alternating ±0.5 gives a flat, non-zero RMS without needing a real tone.
samples[i] = silent ? 0 : i % 2 === 0 ? 0.5 : -0.5;
}
return samples;
}

const word = (w: Partial<SttWordSegment> = {}): SttWordSegment => ({
word: "w",
startSec: 0,
endSec: 0.1,
...w,
/** A helper word; its anchor defaults to its start, as for a request's first word. */
const word = (
text: string,
startSec: number,
endSec: number,
anchorSec = startSec,
): HelperWord => ({
word: text,
startSec,
endSec,
anchorSec,
});

describe("snapWordBoundariesToAudio", () => {
it("pulls a late boundary back into the silence that precedes it", () => {
// Speech stops at 1.0 and resumes at 1.2; whisper reports the next word
// starting at 1.3 — 100 ms after the audio actually resumed.
const samples = audioWithSilences(3, [[1.0, 1.2]]);
const [snapped] = snapWordBoundariesToAudio([word({ startSec: 1.3, endSec: 1.8 })], samples);
expect(snapped.startSec).toBeGreaterThanOrEqual(1.0);
expect(snapped.startSec).toBeLessThan(1.2);
});

it("leaves a boundary alone when nothing quieter precedes it", () => {
// A word ending a phrase: whisper is already right, the frames before the
// boundary are all speech, so the quietest frame in the window is the
// boundary itself and it must not drift.
const samples = audioWithSilences(3, [[1.5, 2.0]]);
const [snapped] = snapWordBoundariesToAudio([word({ startSec: 1.0, endSec: 1.5 })], samples);
expect(snapped.endSec).toBeCloseTo(1.5, 2);
});

it("never moves a boundary more than the lookback window", () => {
const samples = audioWithSilences(3, [[0.0, 1.0]]);
const [snapped] = snapWordBoundariesToAudio([word({ startSec: 2.0, endSec: 2.5 })], samples);
expect(snapped.startSec).toBeGreaterThanOrEqual(2.0 - 0.15);
});

it("keeps a boundary shared by two words shared", () => {
const samples = audioWithSilences(3, [[1.0, 1.2]]);
const [first, second] = snapWordBoundariesToAudio(
[word({ startSec: 0.5, endSec: 1.3 }), word({ startSec: 1.3, endSec: 1.8 })],
samples,
);
expect(first.endSec).toBeCloseTo(second.startSec, 6);
});

it("leaves boundaries that fall outside the decoded audio alone", () => {
// Clamping these into range would collapse every boundary onto the end of
// the buffer instead of leaving the unmeasurable ones untouched.
const samples = audioWithSilences(0.1, []);
const words = [word({ startSec: 5.51, endSec: 6.85 })];
expect(snapWordBoundariesToAudio(words, samples)).toEqual(words);
});

it("keeps degenerate words non-empty and passes words through without audio", () => {
const samples = audioWithSilences(3, [[1.0, 1.2]]);
const [degenerate] = snapWordBoundariesToAudio([word({ startSec: 1.3, endSec: 1.3 })], samples);
expect(degenerate.endSec).toBeGreaterThan(degenerate.startSec);

const untouched = [word({ startSec: 1.3, endSec: 1.8 })];
expect(snapWordBoundariesToAudio(untouched, new Float32Array(0))).toEqual(untouched);
const ms = (sec: number) => Math.round(sec * 1000) / 1000;
const times = (words: SttWordSegment[]) => words.map((w) => [w.word, ms(w.startSec), ms(w.endSec)]);

describe("anchorWordsOnSpeech", () => {
it("returns the helper's times without speech intervals, minus the anchor", () => {
const out = anchorWordsOnSpeech([word("a", 1, 1.4, 1.2), word("b", 1.4, 1.4, 1.4)]);
expect(out).toEqual([
{ word: "a", startSec: 1, endSec: 1.4 },
{ word: "b", startSec: 1.4, endSec: 1.42 },
]);
});
});

describe("snapWordBoundariesToAudio with speech intervals", () => {
// Loud and flat everywhere, so the RMS snap moves nothing and only the
// anchoring on `speech` is under test.
const flat = audioWithSilences(6, []);
const ms = (sec: number) => Math.round(sec * 1000) / 1000;
const times = (words: SttWordSegment[]) =>
words.map((w) => [w.word, ms(w.startSec), ms(w.endSec)]);

it("puts each phrase's first word on its onset and its closing punctuation on its end", () => {
// The shape of a real French take: whisper put "Salut" 0.58 s and "Bah"
// 0.25 s after the speech started, ran "Salut" on through the pause, and
// dropped the "!" closing the first phrase just after the second began.
it("puts a phrase's first word on its onset whether DTW put it late or early", () => {
// "Salut" opens the request, so it has no previous token and starts late,
// on its own first token. "Bah" starts where the previous token ended, at
// the end of the first phrase: in the pause, before its own speech.
const words = [
word({ word: "Salut", startSec: 2.15, endSec: 3.37 }),
word({ word: "!", startSec: 3.37, endSec: 3.39 }),
word({ word: "Bah", startSec: 3.61, endSec: 4.01 }),
word({ word: "voilà", startSec: 4.01, endSec: 4.35 }),
word("Salut", 2.15, 2.5),
word("!", 2.5, 2.6, 2.6),
word("Bah", 2.6, 3.8, 3.61),
word("voilà", 3.8, 4.35, 4.01),
];
const speech = [
{ startSec: 1.57, endSec: 2.56 },
{ startSec: 3.36, endSec: 4.35 },
];
expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([
expect(times(anchorWordsOnSpeech(words, speech))).toEqual([
["Salut", 1.57, 2.56],
["!", 2.56, 2.56],
["Bah", 3.36, 4.01],
["voilà", 4.01, 4.35],
["Bah", 3.36, 3.8],
["voilà", 3.8, 4.35],
]);
});

it("ends each phrase's last word where its speech stops, stretched or cut back", () => {
// The end of the same take: "tic," ran on into the pause, whisper closed
// the last segment at 2.79 while "tac" ran to 3.04, and dropped the "!" on
// the word itself.
const words = [
word({ word: "tic,", startSec: 1.95, endSec: 2.59 }),
word({ word: "tac", startSec: 2.59, endSec: 2.79 }),
word({ word: "!", startSec: 2.62, endSec: 2.79 }),
word("tic,", 1.95, 2.59),
word("tac", 2.59, 2.79, 2.7),
word("!", 2.79, 2.9, 2.9),
];
const speech = [
{ startSec: 1.95, endSec: 2.37 },
{ startSec: 2.59, endSec: 3.04 },
];
expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([
expect(times(anchorWordsOnSpeech(words, speech))).toEqual([
["tic,", 1.95, 2.37],
["tac", 2.59, 3.04],
["!", 3.04, 3.04],
]);
});

it("leaves a phrase alone when its words are on time, or too far off to be its edges", () => {
const onTime = [word({ startSec: 0.95, endSec: 3 })];
expect(times(snapWordBoundariesToAudio(onTime, flat, [{ startSec: 1, endSec: 3 }]))).toEqual(
times(onTime),
);
it("leaves an edge alone when the word is too far past it to be that edge", () => {
// 1.2 s after the onset and 1.1 s before the offset: more likely a neighbour
// of words whisper dropped than the phrase's own edges.
const tooFar = [word({ startSec: 2.2, endSec: 2.5 })];
expect(times(snapWordBoundariesToAudio(tooFar, flat, [{ startSec: 1, endSec: 3.6 }]))).toEqual(
const tooFar = [word("w", 2.2, 2.5)];
expect(times(anchorWordsOnSpeech(tooFar, [{ startSec: 1, endSec: 3.6 }]))).toEqual(
times(tooFar),
);
});

it("never mistakes a phrase's second word for its first", () => {
// "y" opens the second phrase but was reported just before its onset;
// "z" must not be dragged back over it.
const words = [
word({ word: "x", startSec: 0.5, endSec: 1.95 }),
word({ word: "y", startSec: 1.95, endSec: 2.4 }),
word({ word: "z", startSec: 2.4, endSec: 3 }),
];
const speech = [
{ startSec: 0.5, endSec: 1 },
{ startSec: 2, endSec: 3 },
];
expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([
["x", 0.5, 1],
["y", 1.95, 2.4],
["z", 2.4, 3],
]);
});

it("does not take a word in the previous phrase's tail for the next one's first", () => {
// The helper keeps 0.1 s past each offset. "b" was reported in the first
// phrase's tail, and "c" still has to reach the second onset.
it("gives a word to the stretch its anchor falls in, the kept tail included", () => {
// "b" is anchored in the first stretch's 0.1 s tail: it closes that phrase.
// "c" starts in the first stretch (the previous token's end) but is
// anchored in the second, so it opens the second.
const words = [
word({ word: "a", startSec: 1, endSec: 2.05 }),
word({ word: "b", startSec: 2.05, endSec: 3.4 }),
word({ word: "c", startSec: 3.4, endSec: 4 }),
word("a", 1, 1.9, 1.5),
word("b", 1.9, 2.05, 2.05),
word("c", 2.05, 3.5, 3.2),
word("d", 3.5, 4, 3.8),
];
const speech = [
{ startSec: 1, endSec: 2 },
{ startSec: 3, endSec: 4 },
];
expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([
["a", 1, 2.05],
["b", 2.05, 2.07],
["c", 3, 4],
]);
});

it("keeps a word with the stretch whisper heard it in, once the snap pulls it before the onset", () => {
// 80 ms between two stretches: the snap moves "b" from 2.10 to the silence
// at 2.07, inside what would be the first stretch's tail. It still opens
// the second stretch and keeps its audio.
const samples = audioWithSilences(6, [[2, 2.08]]);
const words = [
word({ word: "a", startSec: 1, endSec: 2.1 }),
word({ word: "b", startSec: 2.1, endSec: 2.4 }),
word({ word: "c", startSec: 2.4, endSec: 3 }),
];
const speech = [
{ startSec: 1, endSec: 2 },
{ startSec: 2.08, endSec: 3 },
];
expect(times(snapWordBoundariesToAudio(words, samples, speech))).toEqual([
["a", 1, 2],
["b", 2.07, 2.4],
["c", 2.4, 3],
expect(times(anchorWordsOnSpeech(words, speech))).toEqual([
["a", 1, 1.9],
["b", 1.9, 2],
["c", 3, 3.5],
["d", 3.5, 4],
]);
});
});
Loading
Loading