From 019e1989efdbc282fcee5790d2e3a6342718c528 Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Thu, 1 Oct 2026 00:59:20 +0200 Subject: [PATCH] fix(stt): start each word at the end of the token before it whisper.cpp's t_dtw marks where a token ends, so taking a word's first-token t_dtw as its start put every word one token late (+175 ms median). The helper now starts a word at the previous token's t_dtw, ends it at its own last token's, and sends the first-token time as `anchor` to tell which stretch of speech owns it. The RMS lookback snap goes; phrase edges are anchored on the VAD in both directions. Adds the word-timing harness used to measure it (issue #948). --- .gitignore | 3 + electron/native/whisper-stt/src/main.cpp | 46 ++- electron/stt/snapWordBoundaries.test.ts | 207 +++--------- electron/stt/snapWordBoundaries.ts | 170 +++------- electron/stt/whisperServer.test.ts | 13 +- electron/stt/whisperServer.ts | 23 +- scripts/test-whisper-stt.mjs | 25 +- .../transcription-and-captions.md | 93 ++++-- tools/stt-eval/word-timing/README.md | 69 ++++ tools/stt-eval/word-timing/corpus-texts.mjs | 40 +++ tools/stt-eval/word-timing/evaluate.mjs | 301 ++++++++++++++++++ tools/stt-eval/word-timing/lib.mjs | 243 ++++++++++++++ tools/stt-eval/word-timing/make-corpus.mjs | 137 ++++++++ tools/stt-eval/word-timing/real-check.mjs | 65 ++++ tools/stt-eval/word-timing/run-helper.mjs | 53 +++ tools/stt-eval/word-timing/summarize.mjs | 39 +++ tools/stt-eval/word-timing/tts.ps1 | 81 +++++ tools/stt-eval/word-timing/validate-ref.mjs | 63 ++++ 18 files changed, 1311 insertions(+), 360 deletions(-) create mode 100644 tools/stt-eval/word-timing/README.md create mode 100644 tools/stt-eval/word-timing/corpus-texts.mjs create mode 100644 tools/stt-eval/word-timing/evaluate.mjs create mode 100644 tools/stt-eval/word-timing/lib.mjs create mode 100644 tools/stt-eval/word-timing/make-corpus.mjs create mode 100644 tools/stt-eval/word-timing/real-check.mjs create mode 100644 tools/stt-eval/word-timing/run-helper.mjs create mode 100644 tools/stt-eval/word-timing/summarize.mjs create mode 100644 tools/stt-eval/word-timing/tts.ps1 create mode 100644 tools/stt-eval/word-timing/validate-ref.mjs diff --git a/.gitignore b/.gitignore index 24ad425eb..8b2682b6e 100644 --- a/.gitignore +++ b/.gitignore @@ -95,6 +95,9 @@ result-* /tools/stt-eval/whispercpp-dtw-poc/fixtures/*.wav /tools/stt-eval/whispercpp-dtw-poc/results/ +# Word-timing harness (tools/stt-eval/word-timing): TTS corpus and results. +/tools/stt-eval/word-timing/data/ + opencode.json opencode.json diff --git a/electron/native/whisper-stt/src/main.cpp b/electron/native/whisper-stt/src/main.cpp index 61303f3c3..037230a57 100644 --- a/electron/native/whisper-stt/src/main.cpp +++ b/electron/native/whisper-stt/src/main.cpp @@ -223,9 +223,12 @@ std::string detect_active_backend() { } struct Word { - double start = 0.0; - double end = 0.0; - double prob = 0.0; + double start = 0.0; + double end = 0.0; + // t_dtw of the word's first token: the end of that token, so always inside + // the word. Only used to tell which stretch of speech a word belongs to. + double anchor = 0.0; + double prob = 0.0; std::string text; }; @@ -517,6 +520,16 @@ int main(int argc, char** argv) { }; std::vector segments; + // whisper.cpp writes a token's t_dtw when the DTW path enters the decoder + // row that PREDICTS the next token (whisper_exp_compute_token_level_ + // timestamps_dtw, v1.9.1): it marks the END of the token, not its start. + // So a word runs from the t_dtw of the text token before it to the t_dtw + // of its own last token. Taking its first token's t_dtw as the start put + // every word one token late (+175 ms median, tools/stt-eval/word-timing). + // The previous token carries across segments; the request's very first + // word has none and starts on its first token's time (the Node side + // anchors it on the speech onset anyway). + double prev_tok_end = -1.0; const int n_segments = input.empty() ? 0 : whisper_full_n_segments(ctx); for (int si = 0; si < n_segments; ++si) { Segment seg; @@ -524,11 +537,11 @@ int main(int argc, char** argv) { seg.end = to_original_sec(whisper_full_get_segment_t1(ctx, si), kept); if (const char* t = whisper_full_get_segment_text(ctx, si)) seg.text = t; - struct W { double t_dtw_first; double p_sum; int p_n; std::string text; }; + struct W { double start; double end; double anchor; double p_sum; int p_n; std::string text; }; std::vector word_buf; std::string cur_text; bool in_word = false; - double w_first_t_dtw = 0; + double w_start = 0, w_end = 0, w_anchor = 0; double w_p_sum = 0; int w_p_n = 0; const int n_tokens = whisper_full_n_tokens(ctx, si); @@ -555,30 +568,32 @@ int main(int argc, char** argv) { const double td_dtw = to_original_sec(td.t_dtw >= 0 ? td.t_dtw : 0, kept); if (starts_word && in_word) { - word_buf.push_back({ w_first_t_dtw, w_p_sum, w_p_n, cur_text }); + word_buf.push_back({ w_start, w_end, w_anchor, w_p_sum, w_p_n, cur_text }); w_p_sum = 0; w_p_n = 0; cur_text.clear(); } if (starts_word) { in_word = true; - w_first_t_dtw = td_dtw; + w_start = prev_tok_end >= 0 ? prev_tok_end : td_dtw; + w_anchor = td_dtw; cur_text = (!raw.empty() && raw[0] == ' ') ? raw.substr(1) : raw; } else { cur_text += raw; } + w_end = td_dtw; + prev_tok_end = td_dtw; w_p_sum += td.p; w_p_n += 1; } if (in_word) { - word_buf.push_back({ w_first_t_dtw, w_p_sum, w_p_n, cur_text }); + word_buf.push_back({ w_start, w_end, w_anchor, w_p_sum, w_p_n, cur_text }); } - for (size_t wi = 0; wi < word_buf.size(); ++wi) { + for (const W& b : word_buf) { Word w; - w.start = word_buf[wi].t_dtw_first; - w.end = (wi + 1 < word_buf.size()) - ? word_buf[wi + 1].t_dtw_first - : seg.end; - w.prob = word_buf[wi].p_sum / std::max(1, word_buf[wi].p_n); - w.text = word_buf[wi].text; + w.start = b.start; + w.end = b.end; + w.anchor = b.anchor; + w.prob = b.p_sum / std::max(1, b.p_n); + w.text = b.text; seg.words.push_back(w); } segments.push_back(std::move(seg)); @@ -642,6 +657,7 @@ int main(int argc, char** argv) { {"word", w.text}, {"start", w.start}, {"end", w.end}, + {"anchor", w.anchor}, {"probability", w.prob}, }); } diff --git a/electron/stt/snapWordBoundaries.test.ts b/electron/stt/snapWordBoundaries.test.ts index 4af4d908e..dcd260f92 100644 --- a/electron/stt/snapWordBoundaries.test.ts +++ b/electron/stt/snapWordBoundaries.test.ts @@ -1,199 +1,98 @@ import { describe, expect, it } from "vitest"; -import { snapWordBoundariesToAudio } from "./snapWordBoundaries"; +import { anchorWordsOnSpeech, type HelperWord } from "./snapWordBoundaries"; import type { SttWordSegment } from "./transcriptionContract"; -const SAMPLE_RATE = 16_000; - -/** Mono 16 kHz buffer that is loud everywhere except the given silent spans. */ -function audioWithSilences(durationSec: number, silences: Array<[number, number]>): Float32Array { - const samples = new Float32Array(Math.round(durationSec * SAMPLE_RATE)); - for (let i = 0; i < samples.length; i++) { - const t = i / SAMPLE_RATE; - const silent = silences.some(([from, to]) => t >= from && t < to); - // Alternating ±0.5 gives a flat, non-zero RMS without needing a real tone. - samples[i] = silent ? 0 : i % 2 === 0 ? 0.5 : -0.5; - } - return samples; -} - -const word = (w: Partial = {}): SttWordSegment => ({ - word: "w", - startSec: 0, - endSec: 0.1, - ...w, +/** A helper word; its anchor defaults to its start, as for a request's first word. */ +const word = ( + text: string, + startSec: number, + endSec: number, + anchorSec = startSec, +): HelperWord => ({ + word: text, + startSec, + endSec, + anchorSec, }); - -describe("snapWordBoundariesToAudio", () => { - it("pulls a late boundary back into the silence that precedes it", () => { - // Speech stops at 1.0 and resumes at 1.2; whisper reports the next word - // starting at 1.3 — 100 ms after the audio actually resumed. - const samples = audioWithSilences(3, [[1.0, 1.2]]); - const [snapped] = snapWordBoundariesToAudio([word({ startSec: 1.3, endSec: 1.8 })], samples); - expect(snapped.startSec).toBeGreaterThanOrEqual(1.0); - expect(snapped.startSec).toBeLessThan(1.2); - }); - - it("leaves a boundary alone when nothing quieter precedes it", () => { - // A word ending a phrase: whisper is already right, the frames before the - // boundary are all speech, so the quietest frame in the window is the - // boundary itself and it must not drift. - const samples = audioWithSilences(3, [[1.5, 2.0]]); - const [snapped] = snapWordBoundariesToAudio([word({ startSec: 1.0, endSec: 1.5 })], samples); - expect(snapped.endSec).toBeCloseTo(1.5, 2); - }); - - it("never moves a boundary more than the lookback window", () => { - const samples = audioWithSilences(3, [[0.0, 1.0]]); - const [snapped] = snapWordBoundariesToAudio([word({ startSec: 2.0, endSec: 2.5 })], samples); - expect(snapped.startSec).toBeGreaterThanOrEqual(2.0 - 0.15); - }); - - it("keeps a boundary shared by two words shared", () => { - const samples = audioWithSilences(3, [[1.0, 1.2]]); - const [first, second] = snapWordBoundariesToAudio( - [word({ startSec: 0.5, endSec: 1.3 }), word({ startSec: 1.3, endSec: 1.8 })], - samples, - ); - expect(first.endSec).toBeCloseTo(second.startSec, 6); - }); - - it("leaves boundaries that fall outside the decoded audio alone", () => { - // Clamping these into range would collapse every boundary onto the end of - // the buffer instead of leaving the unmeasurable ones untouched. - const samples = audioWithSilences(0.1, []); - const words = [word({ startSec: 5.51, endSec: 6.85 })]; - expect(snapWordBoundariesToAudio(words, samples)).toEqual(words); - }); - - it("keeps degenerate words non-empty and passes words through without audio", () => { - const samples = audioWithSilences(3, [[1.0, 1.2]]); - const [degenerate] = snapWordBoundariesToAudio([word({ startSec: 1.3, endSec: 1.3 })], samples); - expect(degenerate.endSec).toBeGreaterThan(degenerate.startSec); - - const untouched = [word({ startSec: 1.3, endSec: 1.8 })]; - expect(snapWordBoundariesToAudio(untouched, new Float32Array(0))).toEqual(untouched); +const ms = (sec: number) => Math.round(sec * 1000) / 1000; +const times = (words: SttWordSegment[]) => words.map((w) => [w.word, ms(w.startSec), ms(w.endSec)]); + +describe("anchorWordsOnSpeech", () => { + it("returns the helper's times without speech intervals, minus the anchor", () => { + const out = anchorWordsOnSpeech([word("a", 1, 1.4, 1.2), word("b", 1.4, 1.4, 1.4)]); + expect(out).toEqual([ + { word: "a", startSec: 1, endSec: 1.4 }, + { word: "b", startSec: 1.4, endSec: 1.42 }, + ]); }); -}); -describe("snapWordBoundariesToAudio with speech intervals", () => { - // Loud and flat everywhere, so the RMS snap moves nothing and only the - // anchoring on `speech` is under test. - const flat = audioWithSilences(6, []); - const ms = (sec: number) => Math.round(sec * 1000) / 1000; - const times = (words: SttWordSegment[]) => - words.map((w) => [w.word, ms(w.startSec), ms(w.endSec)]); - - it("puts each phrase's first word on its onset and its closing punctuation on its end", () => { - // The shape of a real French take: whisper put "Salut" 0.58 s and "Bah" - // 0.25 s after the speech started, ran "Salut" on through the pause, and - // dropped the "!" closing the first phrase just after the second began. + it("puts a phrase's first word on its onset whether DTW put it late or early", () => { + // "Salut" opens the request, so it has no previous token and starts late, + // on its own first token. "Bah" starts where the previous token ended, at + // the end of the first phrase: in the pause, before its own speech. const words = [ - word({ word: "Salut", startSec: 2.15, endSec: 3.37 }), - word({ word: "!", startSec: 3.37, endSec: 3.39 }), - word({ word: "Bah", startSec: 3.61, endSec: 4.01 }), - word({ word: "voilà", startSec: 4.01, endSec: 4.35 }), + word("Salut", 2.15, 2.5), + word("!", 2.5, 2.6, 2.6), + word("Bah", 2.6, 3.8, 3.61), + word("voilà", 3.8, 4.35, 4.01), ]; const speech = [ { startSec: 1.57, endSec: 2.56 }, { startSec: 3.36, endSec: 4.35 }, ]; - expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([ + expect(times(anchorWordsOnSpeech(words, speech))).toEqual([ ["Salut", 1.57, 2.56], ["!", 2.56, 2.56], - ["Bah", 3.36, 4.01], - ["voilà", 4.01, 4.35], + ["Bah", 3.36, 3.8], + ["voilà", 3.8, 4.35], ]); }); it("ends each phrase's last word where its speech stops, stretched or cut back", () => { - // The end of the same take: "tic," ran on into the pause, whisper closed - // the last segment at 2.79 while "tac" ran to 3.04, and dropped the "!" on - // the word itself. const words = [ - word({ word: "tic,", startSec: 1.95, endSec: 2.59 }), - word({ word: "tac", startSec: 2.59, endSec: 2.79 }), - word({ word: "!", startSec: 2.62, endSec: 2.79 }), + word("tic,", 1.95, 2.59), + word("tac", 2.59, 2.79, 2.7), + word("!", 2.79, 2.9, 2.9), ]; const speech = [ { startSec: 1.95, endSec: 2.37 }, { startSec: 2.59, endSec: 3.04 }, ]; - expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([ + expect(times(anchorWordsOnSpeech(words, speech))).toEqual([ ["tic,", 1.95, 2.37], ["tac", 2.59, 3.04], ["!", 3.04, 3.04], ]); }); - it("leaves a phrase alone when its words are on time, or too far off to be its edges", () => { - const onTime = [word({ startSec: 0.95, endSec: 3 })]; - expect(times(snapWordBoundariesToAudio(onTime, flat, [{ startSec: 1, endSec: 3 }]))).toEqual( - times(onTime), - ); + it("leaves an edge alone when the word is too far past it to be that edge", () => { // 1.2 s after the onset and 1.1 s before the offset: more likely a neighbour // of words whisper dropped than the phrase's own edges. - const tooFar = [word({ startSec: 2.2, endSec: 2.5 })]; - expect(times(snapWordBoundariesToAudio(tooFar, flat, [{ startSec: 1, endSec: 3.6 }]))).toEqual( + const tooFar = [word("w", 2.2, 2.5)]; + expect(times(anchorWordsOnSpeech(tooFar, [{ startSec: 1, endSec: 3.6 }]))).toEqual( times(tooFar), ); }); - it("never mistakes a phrase's second word for its first", () => { - // "y" opens the second phrase but was reported just before its onset; - // "z" must not be dragged back over it. - const words = [ - word({ word: "x", startSec: 0.5, endSec: 1.95 }), - word({ word: "y", startSec: 1.95, endSec: 2.4 }), - word({ word: "z", startSec: 2.4, endSec: 3 }), - ]; - const speech = [ - { startSec: 0.5, endSec: 1 }, - { startSec: 2, endSec: 3 }, - ]; - expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([ - ["x", 0.5, 1], - ["y", 1.95, 2.4], - ["z", 2.4, 3], - ]); - }); - - it("does not take a word in the previous phrase's tail for the next one's first", () => { - // The helper keeps 0.1 s past each offset. "b" was reported in the first - // phrase's tail, and "c" still has to reach the second onset. + it("gives a word to the stretch its anchor falls in, the kept tail included", () => { + // "b" is anchored in the first stretch's 0.1 s tail: it closes that phrase. + // "c" starts in the first stretch (the previous token's end) but is + // anchored in the second, so it opens the second. const words = [ - word({ word: "a", startSec: 1, endSec: 2.05 }), - word({ word: "b", startSec: 2.05, endSec: 3.4 }), - word({ word: "c", startSec: 3.4, endSec: 4 }), + word("a", 1, 1.9, 1.5), + word("b", 1.9, 2.05, 2.05), + word("c", 2.05, 3.5, 3.2), + word("d", 3.5, 4, 3.8), ]; const speech = [ { startSec: 1, endSec: 2 }, { startSec: 3, endSec: 4 }, ]; - expect(times(snapWordBoundariesToAudio(words, flat, speech))).toEqual([ - ["a", 1, 2.05], - ["b", 2.05, 2.07], - ["c", 3, 4], - ]); - }); - - it("keeps a word with the stretch whisper heard it in, once the snap pulls it before the onset", () => { - // 80 ms between two stretches: the snap moves "b" from 2.10 to the silence - // at 2.07, inside what would be the first stretch's tail. It still opens - // the second stretch and keeps its audio. - const samples = audioWithSilences(6, [[2, 2.08]]); - const words = [ - word({ word: "a", startSec: 1, endSec: 2.1 }), - word({ word: "b", startSec: 2.1, endSec: 2.4 }), - word({ word: "c", startSec: 2.4, endSec: 3 }), - ]; - const speech = [ - { startSec: 1, endSec: 2 }, - { startSec: 2.08, endSec: 3 }, - ]; - expect(times(snapWordBoundariesToAudio(words, samples, speech))).toEqual([ - ["a", 1, 2], - ["b", 2.07, 2.4], - ["c", 2.4, 3], + expect(times(anchorWordsOnSpeech(words, speech))).toEqual([ + ["a", 1, 1.9], + ["b", 1.9, 2], + ["c", 3, 3.5], + ["d", 3.5, 4], ]); }); }); diff --git a/electron/stt/snapWordBoundaries.ts b/electron/stt/snapWordBoundaries.ts index 70e8eada6..247eac6ad 100644 --- a/electron/stt/snapWordBoundaries.ts +++ b/electron/stt/snapWordBoundaries.ts @@ -1,119 +1,88 @@ -// Pulls whisper.cpp's DTW word boundaries back onto the audio they describe. +// Puts the edges of every phrase on the speech the helper's VAD found. // -// whisper.cpp reports one time per token and derives word spans from it, so a -// word's `start` is the point where the decoder *emitted* the token, not where -// the speaker started saying it. Measured on real recordings, that lands -// consistently 80–150 ms late, and because consecutive words share a boundary -// (`word[i].end === word[i+1].start`) the whole transcript is dragged right by -// roughly a syllable. +// The helper times a word from the end of the token before it to the end of its +// own last token (whisper.cpp's t_dtw marks where a token ENDS, see +// whisper-stt/src/main.cpp). Inside continuous speech that lands within ~30 ms +// of the audio (measured on a TTS corpus with exact word times, +// tools/stt-eval/word-timing), so inner boundaries are left alone: an RMS snap +// on top of them would drag correct boundaries early. // -// That is invisible in captions but not in the transcript editor: deleting -// words there turns the selection into a trim of exactly -// `[firstWord.startSec, lastWord.endSec]`, so a late boundary leaves the attack -// of the first removed word audible and bites into the following kept word. +// The edges of a phrase are the exception. The token before a phrase's first +// word is the previous phrase's last one, so that word "starts" at the end of +// the previous phrase, in the pause, or DTW puts it late when the helper +// starts a new decode window there. Its last word ends on whatever token DTW +// last aligned, short of the speech or past it. When the helper sent its speech +// intervals, both are put on the edges of the stretch of speech they belong to. // -// The correction is to look at the audio instead of guessing an offset: each -// boundary moves back to the quietest 10 ms frame within the preceding -// `LOOKBACK_SEC`. It is self-limiting — on a decaying tail (a word that ends a -// phrase, where whisper is already right) the quietest frame IS the reported -// one, so the boundary doesn't move at all. -// -// The edges of a phrase are the exception. Its first word, DTW reports 0.1–0.6 s -// late (measured against Silero VAD on a real French recording), far past what -// the lookback can reach. Its last word ends where whisper's segment ends, which -// can stop short of the speech, or where the next word starts, which runs on -// through the whole pause. When the helper sent its speech intervals, both are -// put on the edges of the stretch of speech they belong to. +// That matters because deleting words in the transcript editor trims exactly +// `[firstWord.startSec, lastWord.endSec]`: a late boundary leaves the attack of +// the first removed word audible, an early one bites into the kept word. import type { SttVadSegment, SttWordSegment } from "./transcriptionContract"; -/** Matches `writeSamplesAsWav` — the samples handed to whisper are mono 16 kHz. */ -const SAMPLE_RATE = 16_000; - -/** RMS envelope resolution; finer than any boundary error worth correcting. */ -const FRAME_SEC = 0.01; - -/** - * How far back a boundary may travel — the calibration knob for this - * correction. Sized to the measured DTW lag (~80–150 ms). Widen it and - * boundaries inside continuous speech start snapping onto the *previous* - * syllable's trough; narrow it and the lag survives. - */ -const LOOKBACK_SEC = 0.15; +/** A word as the helper reports it: `anchorSec` is always inside the word. */ +export interface HelperWord extends SttWordSegment { + /** End of the word's first token. Only tells which stretch of speech owns the word. */ + anchorSec: number; +} -/** Keep degenerate words (whisper sometimes reports end <= start) non-empty. */ +/** Keep degenerate words non-empty. */ const MIN_WORD_SEC = 0.02; /** * How far a phrase's first word may start after its speech, or its last word * end before it, and still be stretched onto that edge — the calibration knob of - * the anchoring step, sized above the measured 0.1–0.6 s. Past it, the word is - * more likely a neighbour of one whisper dropped, and stretching it over that - * audio would be a guess. Cutting a word back to where the speech stops needs - * no such bound: the VAD says nothing is said past it. + * the anchoring step. Past it, the word is more likely a neighbour of one + * whisper dropped, and stretching it over that audio would be a guess. Pulling + * a word forward to the onset, or cutting it back to where the speech stops, + * needs no such bound: the VAD says nothing is said outside it. */ const MAX_ANCHOR_SEC = 1; /** * Audio the helper keeps past each speech offset, as whisper.cpp does, so a soft - * ending survives. A word DTW reports in there belongs to the stretch it trails. + * ending survives. A word whose anchor falls in there belongs to the stretch it + * trails. */ const TAIL_SEC = 0.1; /** French puts a space before `!`, `?`, `:` and `;`, so whisper emits them as words. */ const isPunctuation = (word: string) => /^[\p{P}\p{S}]+$/u.test(word); -/** Per-frame RMS of the mono signal — the cheapest usable "is this speech" proxy. */ -function rmsEnvelope(samples: Float32Array): Float32Array { - const frameLength = Math.round(SAMPLE_RATE * FRAME_SEC); - const frameCount = Math.floor(samples.length / frameLength); - const envelope = new Float32Array(frameCount); - for (let f = 0; f < frameCount; f++) { - const start = f * frameLength; - let sum = 0; - for (let i = start; i < start + frameLength; i++) { - const v = samples[i] ?? 0; - sum += v * v; - } - envelope[f] = Math.sqrt(sum / frameLength); - } - return envelope; -} - /** * Put the first and last word of every speech stretch on the stretch's edges. * The onset already carries the VAD's 30 ms pad, so a cut there lands just * before the attack. Words own the speech and no pause: the last word ends on * the offset, and the punctuation closing the phrase collapses to a point - * there, wherever DTW dropped it. - * - * Which stretch a word belongs to is read off `reportedStarts`, whisper's own - * times: the RMS snap pulls a phrase's first word back into the silence before - * it, which can lie in the previous stretch's tail. + * there, wherever DTW dropped it. Without `speech` (the helper ran without its + * VAD model), the words come back as the helper timed them. */ -function anchorOnSpeech( - words: SttWordSegment[], - reportedStarts: number[], - speech: SttVadSegment[], +export function anchorWordsOnSpeech( + words: HelperWord[], + speech?: SttVadSegment[], ): SttWordSegment[] { - const out = words.map((w) => ({ ...w })); + const out: SttWordSegment[] = words.map(({ anchorSec: _, ...w }) => ({ + ...w, + endSec: Math.max(w.endSec, w.startSec + MIN_WORD_SEC), + })); + if (!speech) return out; let k = 0; for (let i = 0; i < speech.length; i++) { const { startSec: onset, endSec: offset } = speech[i]; - // A stretch owns the words reported before the end of the audio the helper + // A stretch owns the words anchored before the end of the audio the helper // kept for it: its speech, plus a tail that stops short of the next one. const tail = Math.min(offset + TAIL_SEC, speech[i + 1]?.startSec ?? Number.POSITIVE_INFINITY); let first = -1; let last = -1; - for (; k < out.length && reportedStarts[k] < tail; k++) { + for (; k < out.length && words[k].anchorSec < tail; k++) { if (isPunctuation(out[k].word)) continue; if (first < 0) first = k; last = k; } if (first < 0) continue; - const late = out[first].startSec - onset; - if (late > 0 && late <= MAX_ANCHOR_SEC) { + if (out[first].startSec - onset <= MAX_ANCHOR_SEC) { out[first].startSec = onset; + out[first].endSec = Math.max(out[first].endSec, onset + MIN_WORD_SEC); for (let j = first - 1; j >= 0 && out[j].endSec > onset; j--) { out[j].endSec = onset; out[j].startSec = Math.min(out[j].startSec, onset); @@ -128,58 +97,3 @@ function anchorOnSpeech( } return out; } - -/** - * Move every word boundary back to the quietest frame in the `LOOKBACK_SEC` - * preceding it. Boundaries shared by two words snap identically (same input - * time), so the transcript stays gap-free where whisper made it gap-free. - * With `speech` (the helper's VAD intervals), phrase edges are then anchored - * on the speech. Returns the words unchanged when there is no audio to - * measure against. - */ -export function snapWordBoundariesToAudio( - words: SttWordSegment[], - samples: Float32Array, - speech?: SttVadSegment[], -): SttWordSegment[] { - const envelope = rmsEnvelope(samples); - if (envelope.length === 0) return words; - - const lookbackFrames = Math.round(LOOKBACK_SEC / FRAME_SEC); - const snap = (timeSec: number): number => { - const hi = Math.round(timeSec / FRAME_SEC); - // A boundary outside the audio we can measure (whisper occasionally reports - // times past the end of the samples) has nothing to snap to — clamping it - // into range would drag it to the end of the buffer instead. - if (hi <= 0 || hi >= envelope.length) return timeSec; - const lo = Math.max(0, hi - lookbackFrames); - let bestFrame = hi; - let bestRms = envelope[hi]; - // Strict `<` while walking backwards keeps the frame CLOSEST to the - // reported boundary when a whole stretch is equally quiet, so digital - // silence can't drag a boundary the full window. - for (let f = hi - 1; f >= lo; f--) { - if (envelope[f] < bestRms) { - bestRms = envelope[f]; - bestFrame = f; - } - } - return bestFrame * FRAME_SEC; - }; - - const snapped = words.map((w) => { - const startSec = snap(w.startSec); - return { - ...w, - startSec, - endSec: Math.max(snap(w.endSec), startSec + MIN_WORD_SEC), - }; - }); - return speech - ? anchorOnSpeech( - snapped, - words.map((w) => w.startSec), - speech, - ) - : snapped; -} diff --git a/electron/stt/whisperServer.test.ts b/electron/stt/whisperServer.test.ts index 7f52d5a6e..62dd7442d 100644 --- a/electron/stt/whisperServer.test.ts +++ b/electron/stt/whisperServer.test.ts @@ -349,17 +349,22 @@ describe("WhisperServerManager", () => { } }); - it("anchors a phrase's first word on the speech onset the helper reports", async () => { + it("anchors a phrase's first word on the onset of the speech its anchor falls in", async () => { + // "Salut" starts where the previous token ended, in the first stretch's + // tail, but its anchor puts it in the second stretch. const fakeJson = { segments: [ { text: " Salut", start: 1.57, end: 2.56, - words: [{ word: " Salut", start: 2.15, end: 2.56 }], + words: [{ word: " Salut", start: 1.05, end: 2.56, anchor: 1.8 }], }, ], - speech: [{ start: 1.57, end: 2.56 }], + speech: [ + { start: 0.2, end: 1 }, + { start: 1.57, end: 2.56 }, + ], backend: "whispercpp-cpu", }; vi.stubGlobal( @@ -371,7 +376,7 @@ describe("WhisperServerManager", () => { (mgr as unknown as { process: unknown; port: number }).process = {}; (mgr as unknown as { process: unknown; port: number }).port = 9999; const result = await mgr.transcribe({ samples: new Float32Array(16_000 * 3) }); - expect(result.wordSegments[0].startSec).toBeCloseTo(1.57, 6); + expect(result.wordSegments).toEqual([{ word: "Salut", startSec: 1.57, endSec: 2.56 }]); } finally { vi.unstubAllGlobals(); } diff --git a/electron/stt/whisperServer.ts b/electron/stt/whisperServer.ts index 9a27c6dee..a03c05449 100644 --- a/electron/stt/whisperServer.ts +++ b/electron/stt/whisperServer.ts @@ -7,7 +7,7 @@ import path from "node:path"; import type { Readable } from "node:stream"; import { resolveBinaryPath } from "./gpuDetector"; -import { snapWordBoundariesToAudio } from "./snapWordBoundaries"; +import { anchorWordsOnSpeech } from "./snapWordBoundaries"; import type { SttBackend, SttPhraseSegment, @@ -81,6 +81,8 @@ interface WhisperJsonWord { word?: string; start?: number; end?: number; + /** End of the word's first token: always inside the word (see snapWordBoundaries.ts). */ + anchor?: number; probability?: number; } @@ -576,23 +578,28 @@ export class WhisperServerManager { startSec: this.toSec(s.start, 0), endSec: this.toSec(s.end, 0), })); - // whisper.cpp's DTW boundaries run ~80–150 ms behind the audio, and a - // phrase's first word up to 0.6 s, which the transcript editor turns into - // imprecise trims (see snapWordBoundaries.ts). Re-anchor them on the same - // samples whisper was given, and on the helper's speech intervals. - const wordSegments: SttWordSegment[] = snapWordBoundariesToAudio( + // The helper's word times are right inside a phrase but not on its edges, + // which the transcript editor turns into imprecise trims: put those on the + // helper's speech intervals (see snapWordBoundaries.ts). + const wordSegments: SttWordSegment[] = anchorWordsOnSpeech( raw .flatMap((seg) => (seg.words ?? []).map((w) => { const word = (w.word ?? "").trim(); const startSec = this.toSec(w.start, 0); const endSec = this.toSec(w.end, startSec + 0.05); + const anchorSec = this.toSec(w.anchor, startSec); const confidence = typeof w.probability === "number" ? w.probability : undefined; - return { word, startSec, endSec: Math.max(startSec + 0.02, endSec), confidence }; + return { + word, + startSec, + endSec: Math.max(startSec + 0.02, endSec), + anchorSec, + confidence, + }; }), ) .filter((w) => w.word.length > 0), - opts.samples, speech, ); const detectedLanguage = json.detected_language ?? json.language ?? "auto"; diff --git a/scripts/test-whisper-stt.mjs b/scripts/test-whisper-stt.mjs index 7cef820bb..2a0e11223 100644 --- a/scripts/test-whisper-stt.mjs +++ b/scripts/test-whisper-stt.mjs @@ -345,23 +345,16 @@ async function main() { ); } - // NOT a failure. whisper.cpp gives the last word of a segment the segment's - // own t1 as its end, while the word's DTW start runs 80–150 ms late — so a - // token emitted near the boundary (typically standalone punctuation) can land - // after it and invert. The contract's [startSec, endSec) guarantee is enforced - // one layer up, deliberately: whisperServer.ts clamps to - // `Math.max(startSec + 0.02, endSec)` and snapWordBoundaries.ts keeps - // "degenerate words (whisper sometimes reports end <= start) non-empty". This - // harness speaks to the raw helper, below that clamp, so it reports the count - // as information instead of asserting on it. + // A word runs from the end of the token before it to the end of its own last + // token, and the helper rejects a non-monotonic DTW path, so it cannot end + // before it starts. It can be empty (end == start): whisperServer.ts widens + // those to 20 ms before they reach the document. const inverted = words.filter((w) => w.end < w.start); - if (inverted.length > 0) { - console.log( - ` info ${inverted.length}/${words.length} raw word(s) have end < start ` + - `(${inverted.map((w) => JSON.stringify(w.word)).join(", ")}) — expected at segment ` + - "boundaries; whisperServer.ts clamps these before they reach the document.", - ); - } + check( + inverted.length === 0, + "no word ends before it starts", + inverted.map((w) => JSON.stringify(w.word)).join(", "), + ); if (refText) { const rate = wer(refText, text); diff --git a/technical-documentation/architecture/transcription-and-captions.md b/technical-documentation/architecture/transcription-and-captions.md index 0e42ed1b4..65c058b2b 100644 --- a/technical-documentation/architecture/transcription-and-captions.md +++ b/technical-documentation/architecture/transcription-and-captions.md @@ -215,41 +215,64 @@ it verbatim in the response. 3. **Word grouping** — BPE tokens join into a single word whenever the detokenized text begins with a space, or at the first token of the segment. -4. **Word range** — `word.start = t_dtw of the word's first token`, and - `word.end = t_dtw of the next word's first token` (or the segment's `t1` - for the segment's last word). The result is a monotonic, gap-free timeline - of word ranges that downstream code can use without rebasing. -5. **Re-anchor on the audio** — `t_dtw` marks where the decoder *emitted* a - token, not where the speaker started saying it, so every boundary lands - 80–150 ms late (measured across real recordings; the mean correction on - the reference clip is 83 ms). Because consecutive words share a boundary, - the whole transcript is dragged right by roughly a syllable. +4. **Word range** — a token's `t_dtw` marks where it **ends**, not where it + starts. In `whisper_exp_compute_token_level_timestamps_dtw` (whisper.cpp + 1.9.1) the alignment rows start at `<|notimestamps|>`, whose output is text + token 0, and `t_dtw[k]` is written when the DTW path enters row *k+1*: the + row that predicts token *k+1*. So: + - `word.start` = `t_dtw` of the text token **before** the word's first + token. The previous token carries across segments; the request's very + first word has none and falls back to its own first token. + - `word.end` = `t_dtw` of the word's **last** token. + - `word.anchor` = `t_dtw` of the word's first token. It always lies inside + the word, and only decides which stretch of speech owns it (step 5). + + The result is a monotonic, gap-free timeline of word ranges on the upload's + clock. Taking the first token's `t_dtw` as the start, as the helper did + before, put every word one token late: +175 ms median on the TTS corpus + below, and a median 200 ms on a real French take. The repo's DTW POC saw + the same thing as whisper.cpp's `t_dtw` matching faster-whisper's word + **end** (`tools/stt-eval/whispercpp-dtw-poc/REPORT.md`). +5. **Anchor phrase edges on the speech** — [`electron/stt/snapWordBoundaries.ts`](../../electron/stt/snapWordBoundaries.ts) - pulls each boundary back to the quietest 10 ms frame in the preceding - 150 ms of the same samples whisper was given. It is self-limiting: on a - decaying tail — a word ending a phrase, where whisper is already right — - the quietest frame *is* the reported one and nothing moves. Boundaries - past the end of the decoded audio are left untouched. - - The edges of a phrase are past that reach. Its first word, DTW reports - 0.1–0.6 s after the speech starts (measured against the VAD on a real - French take). Its last word ends on whisper's segment end, which can stop - short of the speech, or on the next word's start, which runs on through the - pause. With `speech`, the first real word of each stretch starts on the - stretch's onset and its last real word ends on its offset; the punctuation - closing the phrase collapses to a point there. Words own the speech, never - the pause: deleting a phrase's last word keeps the pause after it. A word - reported in the 0.1 s tail the helper keeps past an offset belongs to that - stretch. `MAX_ANCHOR_SEC` bounds the stretching only: a word further off - than 1 s is more likely the neighbour of one whisper dropped, and is left - alone. + leaves boundaries inside a phrase alone: they are within ~30 ms of the audio + already, and an energy snap on top of them drags correct boundaries early. + The edges of a phrase are different. The token before a phrase's first word + is the previous phrase's last one, so that word starts in the pause, as + early as the previous phrase's end; a word that opens a decode window starts + late on its own first token. Its last word ends on the last token DTW + aligned, short of the speech or past it into the pause. With `speech`: + - the first real word of each stretch starts on the stretch's onset, pulled + back when it is late and clamped forward when it is early; + - its last real word ends on the offset, and the punctuation closing the + phrase collapses to a point there; + - a stretch owns the words whose `anchor` falls in it or in the 0.1 s tail + the helper keeps past its offset. + + Words own the speech, never the pause: deleting a phrase's last word keeps + the pause after it. `MAX_ANCHOR_SEC` bounds the stretching only: a word more + than 1 s past an edge is more likely the neighbour of one whisper dropped, + and is left alone. > Why this matters beyond caption timing: the transcript editor turns a word > selection into a trim of exactly `[firstWord.startSec, lastWord.endSec]`, > so a late boundary leaves the attack of the first removed word audible and -> bites into the following kept word. `LOOKBACK_SEC` in that module is the -> calibration knob — widen it and boundaries inside continuous speech start -> snapping onto the previous syllable's trough. +> an early one bites into the kept word next to it. + +**Measuring it** — [`tools/stt-eval/word-timing/`](../../tools/stt-eval/word-timing/README.md) +scores the helper and this post-pass against a Windows-TTS corpus with exact +word times (27 min of French and English narration, clean and noisy): +boundary error by position, and the audible residue and clipping of deleting +one word or one phrase. Issue #948 has the method, the baseline and the plan. + +| Pipeline (clean + noisy) | Inner start: median / P90 / within 50 ms | Phrase-initial start within 50 ms | One-word delete: clean cuts | Phrase delete: clean | +|---|---|---|---|---| +| First-token start + 150 ms RMS snap + VAD edges (before #948) | 105 / 275 ms / 32% | 83% | 5% | 84% | +| Previous-token start + two-way VAD edges | 31 / 125 ms / 64% | 89% | 19% | 88% | + +A +15 ms calibration offset on every boundary gained 4 points of inner +boundaries within 50 ms but dropped noisy phrase deletes to 82%, so it is not +applied. The recognition call returns **both** the phrase segments and the per-word segments in one pass @@ -302,7 +325,7 @@ and the digest together. | Module | Role | |---|---| | `electron/stt/whisperServer.ts` | Server lifecycle; `POST /inference` client; verbose_json parser. | -| `electron/stt/snapWordBoundaries.ts` | Re-anchors DTW word boundaries on the audio's RMS envelope, and phrase edges on the VAD's speech (see step 5 above). | +| `electron/stt/snapWordBoundaries.ts` | Anchors phrase edges on the VAD's speech (see step 5 above). | | `electron/stt/wav.ts` | WAV write + temp-file cleanup helpers. | | `electron/stt/gpuDetector.ts` | Per-platform binary resolver (no GPU probing). | | `electron/stt/modelManager.ts` | Single GGML file download, SHA-256 verify, atomic write. | @@ -769,7 +792,7 @@ it deletes data in `electron/native/whisper-stt/src/main.cpp` are exercised only at runtime. - **Word timing inside a phrase.** Phrase edges sit on the VAD (step 5), but - a word in the middle of continuous speech still starts on its DTW time - pulled back by at most 150 ms of RMS lookback. A dedicated forced aligner - is the upgrade path (issue #626); measure against the VAD-anchored edges - before adopting one. + a word in the middle of continuous speech is only as good as whisper-small's + DTW: about 30 ms median and 125 ms P90, so one single-word delete in five is + clean. Character-level DTW, then a CTC forced aligner, are the upgrade path + (issue #948 Phases 2 and 3); `tools/stt-eval/word-timing` measures them. diff --git a/tools/stt-eval/word-timing/README.md b/tools/stt-eval/word-timing/README.md new file mode 100644 index 000000000..b12ec4649 --- /dev/null +++ b/tools/stt-eval/word-timing/README.md @@ -0,0 +1,69 @@ +# Word-timing harness + +Scores transcript word times against exact ground truth: how far each word +boundary is from the audio, and what deleting one word or one phrase in the +transcript editor would leave audible or clip. Method, baseline and plan: +issue #948. How the app times words: +[transcription-and-captions.md § Word-level alignment](../../../technical-documentation/architecture/transcription-and-captions.md). + +Node 22+, no dependencies. The corpus is synthesized with Windows speech +synthesis, so generating it needs Windows; scoring runs anywhere. + +## Data + +Everything generated goes to `data/` next to these scripts (gitignored), or +wherever `OSC_WORD_TIMING_DATA` points: + +- `clips/`: `.wav` (clean), `.noisy.wav`, `.ref.json` (reference words) +- `corpus-manifest.json` +- `results/raw//`: the helper's raw responses; `results/.{txt,json}`: scores + +## Generate the corpus (Windows, once, about 5 min) + +```sh +node make-corpus.mjs # 48 TTS clips + a noisy copy of each + reference times +node validate-ref.mjs # checks the reference against the audio's energy +``` + +- Voices: OneCore Hortense, Paul and Julie (French), SAPI Zira (English). + Scripts are in `corpus-texts.mjs`, synthesis in `tts.ps1`. +- ffmpeg: `electron/native/bin/win32-x64/ffmpeg.exe`, or set `FFMPEG`. + +## Run and score + +```sh +node run-helper.mjs [--cpu] +node evaluate.mjs [--snap ] [--out ] +node summarize.mjs "Before=" "After=" +``` + +- `run-helper.mjs` starts its own helper on a port in 20500-20599 with the + app's models (`%APPDATA%/openscreen/stt-models/whisper-ggml`: `ggml-small-q8_0.bin` + and `ggml-silero-v6.2.0.bin`), sends every clip like the app does, and stops + it. Vulkan takes about 2 min for the 55 min of audio, CPU about 17. +- `evaluate.mjs` parses the responses as `whisperServer.ts` does and runs the + post-pass: the repo's `electron/stt/snapWordBoundaries.ts` by default, or the + file given with `--snap`. It reports three stages: `raw`, `post` (no speech + intervals) and `post+vad` (what the app ships). +- A baseline is the same two commands on an older helper and post-pass: + build the helper at that revision, and pass + `git show :electron/stt/snapWordBoundaries.ts` saved to a file as `--snap`. + +## Real speech + +```sh +node real-check.mjs take.wav # 16 kHz mono s16 +``` + +No ground truth there: it prints how far each word's first-token time lies after +its start, and where each phrase's first word lands against the VAD onset, raw +and after the post-pass. + +## Reading the numbers + +- Positions come from the reference: *phrase-initial* has a pause of at least + 100 ms before it, *phrase-final* after it, *inner* is everything else. +- A delete is *clean* when at most 20 ms of the deleted word stays audible and + at most 20 ms of its neighbours is cut. +- Synthetic speech flatters every method. Use the harness to rank approaches, + and check a winner on real speech. diff --git a/tools/stt-eval/word-timing/corpus-texts.mjs b/tools/stt-eval/word-timing/corpus-texts.mjs new file mode 100644 index 000000000..294517fe3 --- /dev/null +++ b/tools/stt-eval/word-timing/corpus-texts.mjs @@ -0,0 +1,40 @@ +// Screen-recording narration scripts. `{N}` is an explicit pause of N ms +// (SSML ). Hesitations, restarts, numbers, UI and product names, and +// sentences run together with no pause are deliberate. +export const TEXTS = { + fr: [ + `Bonjour à tous, {400} aujourd'hui je vais vous montrer comment enregistrer votre écran avec OpenScreen. {700} Alors, {250} la première chose à faire, c'est d'ouvrir la barre d'enregistrement en bas de l'écran. {500} Vous voyez ici le bouton Écran, {200} je clique dessus et je choisis mon moniteur principal, celui en 1920 par 1080. {900} Ensuite, euh, {300} j'active le micro, {150} et la caméra si vous voulez apparaître dans la vidéo. {600} Et voilà, il ne reste plus qu'à appuyer sur le bouton rouge pour lancer l'enregistrement. {1200} Un compte à rebours de trois secondes s'affiche, puis ça démarre.`, + `Dans cette vidéo on va parler du montage. {500} Quand l'enregistrement se termine, l'éditeur s'ouvre tout seul avec la vidéo, la caméra en incrustation et les zooms déjà placés là où j'ai cliqué. {800} Si un zoom ne vous plaît pas, {200} vous le sélectionnez sur la timeline et vous appuyez sur Supprimer. {400} Pour en ajouter un, c'est la touche Z. {1000} Bon, {300} on peut aussi régler la durée du zoom, par exemple 1,5 seconde au lieu de 2 secondes, et changer le niveau d'agrandissement à 150 pour cent.`, + `Passons maintenant à la transcription. {600} OpenScreen transcrit l'audio directement sur votre ordinateur, sans rien envoyer sur internet. {500} Le modèle fait environ 264 mégaoctets et il se télécharge une seule fois. {900} Une fois le texte affiché, hum, {400} vous pouvez supprimer des mots comme dans un traitement de texte. {300} Je sélectionne cette phrase, là, {200} je la supprime, et la vidéo est coupée exactement au même endroit. {1500} C'est très pratique pour enlever les hésitations et les silences trop longs.`, + `Alors attention, {300} il y a un petit piège avec les sous-titres. {600} Par défaut ils utilisent la police Inter en taille 48, et si votre vidéo est en format vertical, en 1080 par 1920, le texte risque de déborder. {800} Dans ce cas, euh, {250} ouvrez l'onglet Sous-titres dans le panneau de droite et réduisez la taille à 36. {500} Vous pouvez aussi activer le mode karaoké qui surligne chaque mot au moment où il est prononcé {700} et c'est justement là que la précision des horodatages devient importante.`, + `Je vais exporter la vidéo. {400} Je clique sur Exporter en haut à droite, {300} je choisis le format MP4, la résolution 1080p et 60 images par seconde. {900} Le rendu se fait avec le compositeur natif, qui utilise la carte graphique quand elle est disponible, sinon le processeur. {600} Pour une vidéo de cinq minutes, ça prend environ quarante secondes sur ma machine. {1100} Si vous préférez un GIF pour une documentation, {200} choisissez GIF dans la liste, avec une largeur de 800 pixels, c'est un bon compromis.`, + `Petite astuce pour finir. {500} Si vous enregistrez souvent la même fenêtre, par exemple Visual Studio Code ou votre navigateur, {300} OpenScreen garde votre dernier choix en mémoire. {700} Donc la prochaine fois, un seul clic suffit. {400} Et si vous vous trompez pendant l'enregistrement, pas de panique, {250} le bouton Recommencer efface la prise en cours et relance le compte à rebours. {1300} Voilà, c'est tout pour aujourd'hui, merci de m'avoir suivi et à bientôt pour la version 2.1.`, + `Euh, {300} donc là on est sur le tableau de bord de l'application, {400} vous voyez trois colonnes: {200} les projets récents, les modèles et les paramètres. {800} Je vais créer un nouveau projet que je vais appeler Démo client, {300} voilà, {500} et je vais importer une vidéo existante, un fichier MKV de douze minutes environ. {1000} Pendant l'import, la barre de progression indique 35 pour cent, puis 70, {300} puis c'est terminé. {600} On peut maintenant ajouter la musique de fond, avec un volume à moins 18 décibels pour qu'elle ne couvre pas la voix.`, + `Question fréquente: {400} est-ce que ça marche sur Linux? {600} Oui, {200} OpenScreen fonctionne sous Windows, macOS et Linux, {300} et sur Linux c'est le portail du bureau qui vous demande quelle fenêtre partager au moment de lancer l'enregistrement. {900} Autre question, {300} est-ce que c'est payant? {500} Non, c'est gratuit et open source, et ça le restera. {1200} Dernière chose, si vous avez un bug, hum, {300} utilisez le menu Enregistrer le diagnostic dans l'icône de la barre des tâches et joignez le fichier à votre ticket GitHub.`, + ], + en: [ + `Hi everyone, {400} today I'm going to show you how to record your screen with OpenScreen. {700} So, {250} the first thing to do is open the recording bar at the bottom of the screen. {500} You can see the Screen button right here, {200} I click it and pick my main monitor, the 1920 by 1080 one. {900} Then, uh, {300} I turn on the microphone, {150} and the camera if you want to appear in the video. {600} And that's it, all that's left is to hit the red button to start recording. {1200} A three second countdown shows up, and then it starts.`, + `In this video we're going to talk about editing. {500} When the recording stops, the editor opens by itself with the video, the webcam overlay and the zooms already placed where I clicked. {800} If you don't like a zoom, {200} select it on the timeline and press Delete. {400} To add one, press the Z key. {1000} Okay, {300} you can also change how long the zoom lasts, say 1.5 seconds instead of 2 seconds, and set the magnification to 150 percent.`, + `Now let's move on to transcription. {600} OpenScreen transcribes the audio right on your computer, without sending anything to the internet. {500} The model is about 264 megabytes and it only downloads once. {900} Once the text shows up, um, {400} you can delete words just like in a word processor. {300} I select this sentence here, {200} I delete it, and the video is cut at exactly the same spot. {1500} It's really handy for removing hesitations and long silences.`, + `Now watch out, {300} there's a small catch with captions. {600} By default they use the Inter font at size 48, and if your video is vertical, 1080 by 1920, the text can overflow. {800} In that case, uh, {250} open the Captions tab in the right panel and bring the size down to 36. {500} You can also turn on karaoke mode which highlights each word as it's spoken {700} and that's exactly where timestamp precision starts to matter.`, + `I'm going to export the video. {400} I click Export at the top right, {300} I choose MP4, 1080p resolution and 60 frames per second. {900} Rendering uses the native compositor, which runs on the graphics card when there is one, and on the processor otherwise. {600} For a five minute video, that takes about forty seconds on my machine. {1100} If you'd rather have a GIF for your docs, {200} pick GIF in the list, with a width of 800 pixels, that's a good trade-off.`, + `One last tip. {500} If you often record the same window, like Visual Studio Code or your browser, {300} OpenScreen remembers your last choice. {700} So next time, a single click is enough. {400} And if you make a mistake while recording, don't panic, {250} the Restart button throws away the current take and restarts the countdown. {1300} That's all for today, thanks for watching and see you soon for version 2.1.`, + `Um, {300} so here we're on the app dashboard, {400} you can see three columns: {200} recent projects, templates and settings. {800} I'm going to create a new project and call it Client demo, {300} there we go, {500} and I'll import an existing video, an MKV file of about twelve minutes. {1000} While it imports, the progress bar says 35 percent, then 70, {300} then it's done. {600} Now we can add background music, with the volume at minus 18 decibels so it doesn't drown out the voice.`, + `Frequently asked question: {400} does it work on Linux? {600} Yes, {200} OpenScreen runs on Windows, macOS and Linux, {300} and on Linux it's the desktop portal that asks you which window to share when you start recording. {900} Another question, {300} does it cost anything? {500} No, it's free and open source, and it will stay that way. {1200} One last thing, if you hit a bug, hmm, {300} use the Save Diagnostics menu in the tray icon and attach the file to your GitHub issue.`, + ], +}; + +// Three renders per script (every voice/rate below). OneCore has three French +// voices; English only has SAPI Zira installed on this machine. +export const RENDERS = { + fr: [ + { engine: "onecore", voice: "Microsoft Hortense", rate: 1.0 }, + { engine: "onecore", voice: "Microsoft Paul", rate: 1.25 }, + { engine: "onecore", voice: "Microsoft Julie", rate: 0.9 }, + ], + en: [ + { engine: "sapi", voice: "Microsoft Zira Desktop", rate: 1.0 }, + { engine: "sapi", voice: "Microsoft Zira Desktop", rate: 1.3 }, + { engine: "sapi", voice: "Microsoft Zira Desktop", rate: 0.85 }, + ], +}; diff --git a/tools/stt-eval/word-timing/evaluate.mjs b/tools/stt-eval/word-timing/evaluate.mjs new file mode 100644 index 000000000..b923c5d6a --- /dev/null +++ b/tools/stt-eval/word-timing/evaluate.mjs @@ -0,0 +1,301 @@ +// Scores helper output against the TTS reference, for each pipeline stage. +// Usage: node evaluate.mjs [--snap ] [--out ] [--quiet] +// reads /results/raw//*.json, writes /results/.json and +// .txt, and prints the table. +// Stages: +// raw helper words, parsed as whisperServer.ts transcribeImpl does +// post the post-pass without speech intervals +// post+vad the post-pass with the helper's `speech`, as the app runs it +// `--snap` swaps the post-pass (default: the repo's snapWordBoundaries.ts), so a +// candidate is scored exactly like the shipped code. It takes the current +// `anchorWordsOnSpeech(words, speech)` or the pre-#948 +// `snapWordBoundariesToAudio(words, samples, speech)`, so the old pipeline can +// be scored too (`git show :electron/stt/snapWordBoundaries.ts`). +import { readdirSync, readFileSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { pathToFileURL } from "node:url"; +import { align, CLIPS, envelopeDb, norm, REPO, RESULTS, readWav, stats } from "./lib.mjs"; + +const [tag, ...rest] = process.argv.slice(2); +const flag = (name, dflt) => { + const i = rest.indexOf(name); + return i >= 0 ? rest[i + 1] : dflt; +}; +const snapPath = path.resolve( + flag("--snap", path.join(REPO, "electron/stt/snapWordBoundaries.ts")), +); +const outName = flag("--out", tag); +const quiet = rest.includes("--quiet"); +const mod = await import(pathToFileURL(snapPath).href); +const postPass = mod.anchorWordsOnSpeech + ? (words, _samples, speech) => mod.anchorWordsOnSpeech(words, speech) + : mod.snapWordBoundariesToAudio; + +const PAUSE = 0.1; // reference gap that makes a word phrase-initial / phrase-final +const AUDIBLE_DB = -50; // clean-audio frame energy that counts as audible speech +const CLEAN_MS = 20; // a cut is clean when both audible residue and clipping stay under this +const FRAME = 0.005; +const MS = 1000; + +/** Mirrors whisperServer.ts transcribeImpl. */ +function parse(json) { + const toSec = (v, d) => { + const n = typeof v === "string" ? Number(v) : v; + return Number.isFinite(n) ? n : d; + }; + const words = (json.segments ?? []) + .flatMap((seg) => + (seg.words ?? []).map((w) => { + const word = (w.word ?? "").trim(); + const startSec = toSec(w.start, 0); + const endSec = toSec(w.end, startSec + 0.05); + return { + word, + startSec, + endSec: Math.max(startSec + 0.02, endSec), + anchorSec: toSec(w.anchor, startSec), + }; + }), + ) + .filter((w) => w.word.length > 0); + const speech = json.speech?.map((s) => ({ + startSec: toSec(s.start, 0), + endSec: toSec(s.end, 0), + })); + return { words, speech }; +} + +const overlap = (a0, a1, b0, b1) => Math.max(0, Math.min(a1, b1) - Math.max(a0, b0)); +/** Audible seconds of [a0, a1]: 5 ms frames of the CLEAN audio above AUDIBLE_DB. */ +const audible = (env, a0, a1) => { + if (a1 <= a0) return 0; + let n = 0; + for ( + let f = Math.max(0, Math.floor(a0 / FRAME)); + f < Math.min(env.length, Math.ceil(a1 / FRAME)); + f++ + ) { + if (env[f] > AUDIBLE_DB) n += overlap(a0, a1, f * FRAME, (f + 1) * FRAME); + } + return n; +}; +/** Seconds of [r0, r1] outside the cut [c0, c1], and the audible part of it. */ +const outside = (env, r0, r1, c0, c1) => { + const segs = [ + [r0, Math.min(r1, c0)], + [Math.max(r0, c1), r1], + ].filter(([a, b]) => b > a); + return { + sec: segs.reduce((s, [a, b]) => s + b - a, 0), + aud: segs.reduce((s, [a, b]) => s + audible(env, a, b), 0), + }; +}; +const inside = (env, r0, r1, c0, c1) => { + const a = Math.max(r0, c0), + b = Math.min(r1, c1); + return b > a ? { sec: b - a, aud: audible(env, a, b) } : { sec: 0, aud: 0 }; +}; + +const STAGES = ["raw", "post", "post+vad"]; +const acc = {}; +const push = (k, v) => { + acc[k] ??= []; + acc[k].push(v); +}; +const counts = {}; +const inc = (k, v = 1) => (counts[k] = (counts[k] ?? 0) + v); +const perClip = []; + +const rawDir = path.join(RESULTS, "raw", tag); +for (const file of readdirSync(rawDir) + .filter((f) => f.endsWith(".json")) + .sort()) { + const [id, cond] = file.replace(/\.json$/, "").split("."); + const json = JSON.parse(readFileSync(path.join(rawDir, file), "utf8")); + const ref = JSON.parse(readFileSync(path.join(CLIPS, `${id}.ref.json`), "utf8")); + const samples = readWav(path.join(CLIPS, cond === "clean" ? `${id}.wav` : `${id}.noisy.wav`)); + const env = envelopeDb(readWav(path.join(CLIPS, `${id}.wav`)), FRAME); + const { words: rawWords, speech } = parse(json); + const t0 = performance.now(); + const post = postPass(rawWords, samples); + const full = postPass(rawWords, samples, speech); + const tsMs = performance.now() - t0; + const variants = { raw: rawWords, post, "post+vad": full }; + + const R = ref.words.map((w, i) => ({ ...w, i, n: norm(w.text) })).filter((w) => w.n); + const H = rawWords.map((w, j) => ({ j, n: norm(w.word) })).filter((w) => w.n); + const { pairs, sub, ins, del } = align( + R.map((w) => w.n), + H.map((w) => w.n), + ); + const groups = ["all", cond, `${cond}.${ref.lang}`]; + for (const g of groups) { + inc(`${g}.refWords`, R.length); + inc(`${g}.matched`, pairs.length); + inc(`${g}.sub`, sub); + inc(`${g}.ins`, ins); + inc(`${g}.del`, del); + } + perClip.push({ + id, + cond, + lang: ref.lang, + audioSec: ref.durationSec, + helperSec: json.timing?.elapsed_s, + wallMs: json.wallMs, + tsMs, + refWords: R.length, + matched: pairs.length, + sub, + ins, + del, + }); + + // Phrase position from the reference: a pause (>= PAUSE) before / after the word. + const pos = R.map((w, k) => ({ + initial: k === 0 || w.start - R[k - 1].end >= PAUSE, + final: k === R.length - 1 || R[k + 1].start - w.end >= PAUSE, + })); + const matchOf = new Map(pairs.map(([ri, hj]) => [ri, H[hj].j])); + for (const stage of STAGES) { + const V = variants[stage]; + for (const [ri, hj] of pairs) { + const r = R[ri]; + const h = V[H[hj].j]; + const ds = h.startSec - r.start; + const de = h.endSec - r.end; + const sCls = pos[ri].initial ? "initial" : "inner"; + const eCls = pos[ri].final ? "final" : "inner"; + for (const g of groups) { + push(`${g}|${stage}|start|all`, Math.abs(ds)); + push(`${g}|${stage}|start|${sCls}`, Math.abs(ds)); + push(`${g}|${stage}|end|all`, Math.abs(de)); + push(`${g}|${stage}|end|${eCls}`, Math.abs(de)); + push(`${g}|${stage}|bias_start|${sCls}`, ds); + push(`${g}|${stage}|bias_end|${eCls}`, de); + } + // Trim: this word deleted alone -> cut [h.start, h.end]. + const res = outside(env, r.start, r.end, h.startSec, h.endSec); + let clipSec = 0, + clipAud = 0, + clipShare = 0; + for (const k of [ri - 1, ri + 1]) { + const nb = R[k]; + if (!nb) continue; + const c = inside(env, nb.start, nb.end, h.startSec, h.endSec); + clipSec += c.sec; + clipAud += c.aud; + clipShare = Math.max(clipShare, c.sec / Math.max(1e-3, nb.end - nb.start)); + } + for (const g of groups) { + push(`${g}|${stage}|trim|residueMs`, res.sec * MS); + push(`${g}|${stage}|trim|residueShare`, res.sec / Math.max(1e-3, r.end - r.start)); + push(`${g}|${stage}|trim|residueAudMs`, res.aud * MS); + push(`${g}|${stage}|trim|clipMs`, clipSec * MS); + push(`${g}|${stage}|trim|clipShare`, clipShare); + push(`${g}|${stage}|trim|clipAudMs`, clipAud * MS); + push( + `${g}|${stage}|trim|clean`, + res.aud * MS <= CLEAN_MS && clipAud * MS <= CLEAN_MS ? 1 : 0, + ); + } + } + // Phrase deletion: every reference phrase (words between two pauses) whose + // first and last words are matched -> cut [first.start, last.end]. + let k = 0; + while (k < R.length) { + let e = k; + while (!pos[e].final) e++; + if (matchOf.has(k) && matchOf.has(e)) { + const c0 = V[matchOf.get(k)].startSec, + c1 = V[matchOf.get(e)].endSec; + const res = outside(env, R[k].start, R[e].end, c0, c1); + let clipAud = 0; + for (const nb of [R[k - 1], R[e + 1]]) + if (nb) clipAud += inside(env, nb.start, nb.end, c0, c1).aud; + for (const g of groups) { + push(`${g}|${stage}|phrase|residueAudMs`, res.aud * MS); + push(`${g}|${stage}|phrase|clipAudMs`, clipAud * MS); + push( + `${g}|${stage}|phrase|clean`, + res.aud * MS <= CLEAN_MS && clipAud * MS <= CLEAN_MS ? 1 : 0, + ); + } + } + k = e + 1; + } + } +} + +const summary = {}; +for (const [key, xs] of Object.entries(acc)) { + if (key.includes("|trim|") || key.includes("|phrase|") || key.includes("|bias_")) { + const a = [...xs].sort((p, q) => p - q); + summary[key] = { + n: a.length, + mean: a.reduce((s, x) => s + x, 0) / a.length, + median: a[Math.floor(a.length / 2)], + p90: a[Math.floor(a.length * 0.9)], + }; + } else summary[key] = stats(xs); +} +writeFileSync( + path.join(RESULTS, `${outName}.json`), + JSON.stringify({ tag, snapPath, counts, summary, perClip }, null, 1), +); + +// ---- table ---- +const f = (x) => (x * MS).toFixed(0).padStart(4); +const g0 = (x) => x.toFixed(0).padStart(4); +const pc = (x) => `${(x * 100).toFixed(0)}%`.padStart(4); +const line = (g, st, what) => { + const s = summary[`${g}|${st}|${what}`]; + return s?.n + ? `${f(s.mean)} ${f(s.median)} ${f(s.p90)} ${pc(s.w20)} ${pc(s.w50)} ${pc(s.w100)} n=${s.n}` + : "-"; +}; +const lines = [`# ${outName} (post-pass ${path.basename(snapPath)})`]; +for (const g of ["all", "clean", "noisy", "clean.fr", "clean.en", "noisy.fr", "noisy.en"]) { + if (!counts[`${g}.refWords`]) continue; + const wer = + (counts[`${g}.sub`] + counts[`${g}.ins`] + counts[`${g}.del`]) / counts[`${g}.refWords`]; + lines.push( + `\n=== ${g}: ${counts[`${g}.refWords`]} ref words, matched ${pc(counts[`${g}.matched`] / counts[`${g}.refWords`])}, WER ${pc(wer)} (sub ${counts[`${g}.sub`]} ins ${counts[`${g}.ins`]} del ${counts[`${g}.del`]})`, + ); + lines.push(`${"".padEnd(28)}mean med p90 <20 <50 <100 (ms, |error|)`); + for (const st of STAGES) { + const t = (k) => summary[`${g}|${st}|${k}`]; + for (const what of [ + "start|initial", + "start|inner", + "end|inner", + "end|final", + "start|all", + "end|all", + ]) + lines.push(`${st.padEnd(10)}${what.padEnd(18)}${line(g, st, what)}`); + lines.push( + `${st.padEnd(10)}bias (median signed ms): start initial ${f(t("bias_start|initial").median)} inner ${f(t("bias_start|inner").median)} | end inner ${f(t("bias_end|inner").median)} final ${f(t("bias_end|final").median)}`, + ); + lines.push( + `${st.padEnd(10)}1-word delete: residue ${g0(t("trim|residueMs").mean)} ms (audible ${g0(t("trim|residueAudMs").mean)}, p90 ${g0(t("trim|residueAudMs").p90)}), share ${pc(t("trim|residueShare").mean)} | clipping ${g0(t("trim|clipMs").mean)} ms (audible ${g0(t("trim|clipAudMs").mean)}, p90 ${g0(t("trim|clipAudMs").p90)}), share ${pc(t("trim|clipShare").mean)} | clean cuts ${pc(t("trim|clean").mean)}`, + ); + lines.push( + `${st.padEnd(10)}phrase delete: audible residue ${g0(t("phrase|residueAudMs").mean)} ms (p90 ${g0(t("phrase|residueAudMs").p90)}), audible clipping ${g0(t("phrase|clipAudMs").mean)} ms (p90 ${g0(t("phrase|clipAudMs").p90)}), clean ${pc(t("phrase|clean").mean)} n=${t("phrase|clean").n}`, + ); + } +} +const rt = perClip.reduce( + (a, c) => ({ + audio: a.audio + c.audioSec, + helper: a.helper + (c.helperSec ?? 0), + wall: a.wall + c.wallMs / MS, + ts: a.ts + c.tsMs / MS, + }), + { audio: 0, helper: 0, wall: 0, ts: 0 }, +); +lines.push( + `\nruntime: ${perClip.length} clips, ${(rt.audio / 60).toFixed(1)} min audio; helper ${rt.helper.toFixed(1)} s (RTF ${(rt.helper / rt.audio).toFixed(3)}), per clip mean ${(rt.helper / perClip.length).toFixed(2)} s; TS post-pass ${(rt.ts * MS).toFixed(0)} ms total (${((rt.ts * MS) / perClip.length).toFixed(1)} ms/clip, both stages)`, +); +writeFileSync(path.join(RESULTS, `${outName}.txt`), lines.join("\n") + "\n"); +if (!quiet) console.log(lines.join("\n")); diff --git a/tools/stt-eval/word-timing/lib.mjs b/tools/stt-eval/word-timing/lib.mjs new file mode 100644 index 000000000..71391e1a3 --- /dev/null +++ b/tools/stt-eval/word-timing/lib.mjs @@ -0,0 +1,243 @@ +// Shared helpers for the word-timing harness: paths, WAV I/O, energy envelope, +// silence runs, reference ends, word normalization and alignment, stats. +// Node stdlib only. +import { spawn } from "node:child_process"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +export const HERE = path.dirname(fileURLToPath(import.meta.url)); +export const REPO = path.resolve(HERE, "../../.."); +/** Corpus, raw helper output and results. Gitignored; OSC_WORD_TIMING_DATA moves it. */ +export const DATA = path.resolve(process.env.OSC_WORD_TIMING_DATA ?? path.join(HERE, "data")); +export const CLIPS = path.join(DATA, "clips"); +export const RESULTS = path.join(DATA, "results"); +export const FFMPEG = + process.env.FFMPEG ?? path.join(REPO, "electron/native/bin/win32-x64/ffmpeg.exe"); +export const MODELS = path.join( + process.env.APPDATA ?? "", + "openscreen", + "stt-models", + "whisper-ggml", +); +export const SR = 16000; + +/** + * Starts whisper-stt-server with the app's models on a random port in + * 20500-20599 and waits until it answers. The process is killed on exit. + */ +export async function startHelper(exe, { cpu = false, env = process.env } = {}) { + const port = 20500 + Math.floor(Math.random() * 100); + const args = [ + "--model", + path.join(MODELS, "ggml-small-q8_0.bin"), + "--vad-model", + path.join(MODELS, "ggml-silero-v6.2.0.bin"), + "--port", + String(port), + "--host", + "127.0.0.1", + "--threads", + "16", + ...(cpu ? ["--cpu"] : []), + ]; + const child = spawn(exe, args, { stdio: ["ignore", "ignore", "pipe"], env }); + let stderr = ""; + child.stderr.on("data", (d) => { + stderr = (stderr + d).slice(-4000); + }); + const stop = () => child.kill(); + process.on("exit", stop); + process.on("SIGINT", () => process.exit(130)); + const base = `http://127.0.0.1:${port}`; + for (let t = 0; ; t++) { + if (child.exitCode !== null) throw new Error(`helper exited ${child.exitCode}: ${stderr}`); + try { + if ((await fetch(base)).ok) return { base, stop }; + } catch { + // not listening yet + } + if (t > 600) { + stop(); + throw new Error(`helper not ready: ${stderr}`); + } + await new Promise((r) => setTimeout(r, 200)); + } +} + +/** POSTs a WAV to /inference the way the app does (verbose_json, language "auto"). */ +export async function transcribe(base, wav, language = "auto") { + const form = new FormData(); + form.set("file", new Blob([readFileSync(wav)], { type: "audio/wav" }), path.basename(wav)); + form.set("response_format", "verbose_json"); + form.set("language", language); + const t0 = performance.now(); + const res = await fetch(`${base}/inference`, { method: "POST", body: form }); + const wallMs = performance.now() - t0; + const json = await res.json(); + if (!res.ok) throw new Error(`${wav}: HTTP ${res.status} ${JSON.stringify(json)}`); + return { json, wallMs }; +} + +/** 16-bit PCM mono 16 kHz WAV -> Float32Array in [-1, 1]. Walks chunks. */ +export function readWav(file) { + const b = readFileSync(file); + let off = 12; + let fmt = null; + while (off + 8 <= b.length) { + const id = b.toString("ascii", off, off + 4); + const size = b.readUInt32LE(off + 4); + if (id === "fmt ") + fmt = { + channels: b.readUInt16LE(off + 10), + rate: b.readUInt32LE(off + 12), + bits: b.readUInt16LE(off + 22), + }; + if (id === "data") { + if (!fmt || fmt.channels !== 1 || fmt.bits !== 16 || fmt.rate !== SR) + throw new Error(`${file}: need 16 kHz mono s16, got ${JSON.stringify(fmt)}`); + const n = Math.min(size, b.length - off - 8) >> 1; + const out = new Float32Array(n); + for (let i = 0; i < n; i++) out[i] = b.readInt16LE(off + 8 + i * 2) / 32768; + return out; + } + off += 8 + size + (size & 1); + } + throw new Error(`${file}: no data chunk`); +} + +/** RMS in dBFS per `frameSec` frame. */ +export function envelopeDb(samples, frameSec = 0.005) { + const L = Math.round(SR * frameSec); + const n = Math.floor(samples.length / L); + const out = new Float32Array(n); + for (let f = 0; f < n; f++) { + let s = 0; + for (let i = f * L; i < (f + 1) * L; i++) s += samples[i] * samples[i]; + out[f] = 10 * Math.log10(s / L + 1e-12); + } + return out; +} + +/** Runs of frames below `thrDb` lasting at least `minSec`: [{ start, end }] seconds. */ +export function silenceRuns(env, frameSec, thrDb, minSec) { + const runs = []; + let s = -1; + for (let f = 0; f <= env.length; f++) { + const quiet = f < env.length && env[f] < thrDb; + if (quiet && s < 0) s = f; + if (!quiet && s >= 0) { + if ((f - s) * frameSec >= minSec) runs.push({ start: s * frameSec, end: f * frameSec }); + s = -1; + } + } + return runs; +} + +// Reference ends. SAPI gives them (last non-silent phoneme). OneCore gives only +// starts: a word runs to the next word's start unless a pause (>= MIN_PAUSE of +// energy below THR_DB) sits between them, in which case it ends where the pause +// starts. Checked against SAPI's phoneme ends in validate-ref.mjs. +export const THR_DB = -50; +export const MIN_PAUSE = 0.1; +export function referenceWords(tts, samples) { + const FRAME = 0.005; + const runs = silenceRuns(envelopeDb(samples, FRAME), FRAME, THR_DB, MIN_PAUSE); + const dur = samples.length / SR; + return tts.words.map((w, i) => { + const next = tts.words[i + 1]?.start ?? dur; + const pause = runs.find((r) => r.start > w.start + 0.03 && r.start < next); + const derivedEnd = pause ? pause.start : next; + return { + text: w.text, + start: w.start, + end: w.end ?? derivedEnd, + derivedEnd, + ttsEnd: w.end ?? null, + }; + }); +} + +/** Lowercase, keep letters, digits and inner apostrophes only. */ +export function norm(w) { + return w + .toLowerCase() + .normalize("NFC") + .replace(/[’']/g, "'") + .replace(/[^\p{L}\p{N}']/gu, "") + .replace(/^'+|'+$/g, ""); +} + +/** Levenshtein alignment of two token lists; returns matched pairs [i, j] and sub/ins/del counts. */ +export function align(ref, hyp) { + const n = ref.length; + const m = hyp.length; + const W = m + 1; + const D = new Uint32Array((n + 1) * W); + const B = new Uint8Array((n + 1) * W); // 0 diag, 1 up (del), 2 left (ins) + for (let i = 1; i <= n; i++) { + D[i * W] = i; + B[i * W] = 1; + } + for (let j = 1; j <= m; j++) { + D[j] = j; + B[j] = 2; + } + for (let i = 1; i <= n; i++) { + for (let j = 1; j <= m; j++) { + const c = D[(i - 1) * W + j - 1] + (ref[i - 1] === hyp[j - 1] ? 0 : 1); + const u = D[(i - 1) * W + j] + 1; + const l = D[i * W + j - 1] + 1; + if (c <= u && c <= l) { + D[i * W + j] = c; + B[i * W + j] = 0; + } else if (u <= l) { + D[i * W + j] = u; + B[i * W + j] = 1; + } else { + D[i * W + j] = l; + B[i * W + j] = 2; + } + } + } + const pairs = []; + let sub = 0, + ins = 0, + del = 0; + let i = n, + j = m; + while (i > 0 || j > 0) { + const b = B[i * W + j]; + if (i > 0 && j > 0 && b === 0) { + if (ref[i - 1] === hyp[j - 1]) pairs.push([i - 1, j - 1]); + else sub++; + i--; + j--; + } else if (i > 0 && (j === 0 || b === 1)) { + del++; + i--; + } else { + ins++; + j--; + } + } + pairs.reverse(); + return { pairs, sub, ins, del }; +} + +export function stats(xs) { + if (xs.length === 0) return { n: 0 }; + const a = [...xs].sort((p, q) => p - q); + const q = (p) => a[Math.min(a.length - 1, Math.floor(p * a.length))]; + const mean = a.reduce((s, x) => s + x, 0) / a.length; + const within = (t) => a.filter((x) => x <= t).length / a.length; + return { + n: a.length, + mean, + median: q(0.5), + p90: q(0.9), + w20: within(0.02), + w50: within(0.05), + w100: within(0.1), + }; +} diff --git a/tools/stt-eval/word-timing/make-corpus.mjs b/tools/stt-eval/word-timing/make-corpus.mjs new file mode 100644 index 000000000..5bc031ed3 --- /dev/null +++ b/tools/stt-eval/word-timing/make-corpus.mjs @@ -0,0 +1,137 @@ +// Builds the synthetic corpus: SSML -> Windows TTS (tts.ps1) -> 16 kHz mono WAV +// + reference word times, plus a degraded copy (pink noise + light reverb). +// Usage: node make-corpus.mjs (skips clips already on disk) +import { execFileSync } from "node:child_process"; +import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { RENDERS, TEXTS } from "./corpus-texts.mjs"; +import { CLIPS, DATA, FFMPEG, HERE, readWav, referenceWords } from "./lib.mjs"; + +mkdirSync(CLIPS, { recursive: true }); + +const esc = (s) => s.replace(/&/g, "&").replace(//g, ">"); +function ssml(text, lang, rate) { + const body = text + .split(/(\{\d+\})/) + .map((part) => { + const m = /^\{(\d+)\}$/.exec(part); + return m ? `` : esc(part); + }) + .join(""); + // OneCore rejects a without attributes: always write the rate. + const pct = Math.round((rate - 1) * 100); + const r = ` rate="${pct >= 0 ? "+" : ""}${pct}%"`; + return `${body}`; +} + +function render(id, text, lang, r) { + const base = path.join(CLIPS, id); + if (existsSync(`${base}.ref.json`)) return JSON.parse(readFileSync(`${base}.ref.json`, "utf8")); + const ssmlFile = `${base}.ssml`; + writeFileSync(ssmlFile, ssml(text, lang, r.rate)); + execFileSync( + "powershell", + [ + "-NoProfile", + "-ExecutionPolicy", + "Bypass", + "-File", + path.join(HERE, "tts.ps1"), + "-Engine", + r.engine, + "-Voice", + r.voice, + "-SsmlFile", + ssmlFile, + "-Out", + base, + ], + { stdio: "inherit" }, + ); + // Canonical 16 kHz mono s16 with a plain header (both engines already emit + // 16 kHz mono, so this is a container rewrite, no resampling). + execFileSync(FFMPEG, [ + "-hide_banner", + "-loglevel", + "error", + "-y", + "-i", + `${base}.raw.wav`, + "-ar", + "16000", + "-ac", + "1", + "-c:a", + "pcm_s16le", + "-map_metadata", + "-1", + "-fflags", + "+bitexact", + `${base}.wav`, + ]); + rmSync(`${base}.raw.wav`); + // Degraded copy: pink noise ~25 dB under the speech, two early reflections. + execFileSync(FFMPEG, [ + "-hide_banner", + "-loglevel", + "error", + "-y", + "-i", + `${base}.wav`, + "-filter_complex", + "[0:a]aecho=0.9:0.6:23|41:0.25|0.15[r];anoisesrc=color=pink:amplitude=0.012:sample_rate=16000:seed=7[n];[r][n]amix=inputs=2:duration=first:normalize=0[o]", + "-map", + "[o]", + "-ar", + "16000", + "-ac", + "1", + "-c:a", + "pcm_s16le", + "-fflags", + "+bitexact", + `${base}.noisy.wav`, + ]); + const tts = JSON.parse(readFileSync(`${base}.tts.json`, "utf8").replace(/^/, "")); + const samples = readWav(`${base}.wav`); + const ref = { + id, + lang, + ...r, + durationSec: samples.length / 16000, + words: referenceWords(tts, samples), + }; + writeFileSync(`${base}.ref.json`, JSON.stringify(ref, null, 1)); + return ref; +} + +const manifest = []; +for (const lang of ["fr", "en"]) { + TEXTS[lang].forEach((text, i) => { + for (const r of RENDERS[lang]) { + const id = `${lang}${i + 1}-${r.voice.split(" ")[1].toLowerCase()}-${Math.round(r.rate * 100)}`; + const ref = render(id, text, lang, r); + manifest.push({ + id, + lang, + engine: r.engine, + voice: r.voice, + rate: r.rate, + script: i + 1, + durationSec: ref.durationSec, + words: ref.words.length, + }); + console.log(id, ref.durationSec.toFixed(1), "s", ref.words.length, "words"); + } + }); +} +const total = manifest.reduce((s, c) => s + c.durationSec, 0); +writeFileSync( + path.join(DATA, "corpus-manifest.json"), + JSON.stringify( + { totalSec: total, degraded: "pink noise amplitude 0.012 + aecho 23/41 ms", clips: manifest }, + null, + 1, + ), +); +console.log(`total ${(total / 60).toFixed(1)} min clean (+ the same again degraded)`); diff --git a/tools/stt-eval/word-timing/real-check.mjs b/tools/stt-eval/word-timing/real-check.mjs new file mode 100644 index 000000000..09a3d7865 --- /dev/null +++ b/tools/stt-eval/word-timing/real-check.mjs @@ -0,0 +1,65 @@ +// Sanity check on real speech, which has no ground truth: the VAD onset is the +// only boundary we can trust there. +// Usage: node real-check.mjs [--cpu] [--snap ] +// (16 kHz mono s16 WAV, e.g. ffmpeg -i take.webm -ar 16000 -ac 1 take.wav) +// Prints how far each word's first-token time (`anchor`) lies after its start +// (the one-token lag the helper corrects), and, per VAD stretch, where its first +// word starts relative to the onset: as the helper reports it, and after the +// post-pass. Writes the raw response to /results/real-.json. +import { mkdirSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { pathToFileURL } from "node:url"; +import { REPO, RESULTS, startHelper, transcribe } from "./lib.mjs"; + +const [exe, wav, ...rest] = process.argv.slice(2); +if (!exe || !wav) + throw new Error( + "usage: node real-check.mjs [--cpu] [--snap ]", + ); +const snapAt = rest.indexOf("--snap"); +const snapPath = path.resolve( + snapAt >= 0 ? rest[snapAt + 1] : path.join(REPO, "electron/stt/snapWordBoundaries.ts"), +); +const { anchorWordsOnSpeech } = await import(pathToFileURL(snapPath).href); + +const { base, stop } = await startHelper(exe, { cpu: rest.includes("--cpu") }); +let json; +try { + ({ json } = await transcribe(base, wav)); +} finally { + stop(); +} +mkdirSync(RESULTS, { recursive: true }); +writeFileSync(path.join(RESULTS, `real-${path.basename(wav, ".wav")}.json`), JSON.stringify(json)); + +const isP = (w) => /^[\p{P}\p{S}]+$/u.test(w.word); +const raw = json.segments + .flatMap((s) => s.words) + .map((w) => ({ + word: w.word.trim(), + startSec: w.start, + endSec: Math.max(w.start + 0.02, w.end), + anchorSec: w.anchor ?? w.start, + })) + .filter((w) => w.word); +const speech = json.speech.map((s) => ({ startSec: s.start, endSec: s.end })); +const post = anchorWordsOnSpeech(raw, speech); +const lag = raw + .filter((w) => !isP(w)) + .map((w) => w.anchorSec - w.startSec) + .sort((a, b) => a - b); +const q = (p) => (lag[Math.floor(lag.length * p)] * 1000).toFixed(0); +console.log( + `${json.detected_language}, ${raw.length} words; first-token time minus start: median ${q(0.5)} ms, p10 ${q(0.1)}, p90 ${q(0.9)}`, +); +const ms = (x) => `${x >= 0 ? "+" : ""}${(x * 1000).toFixed(0)} ms`; +let k = 0; +for (const [i, s] of speech.entries()) { + const tail = Math.min(s.endSec + 0.1, speech[i + 1]?.startSec ?? Number.POSITIVE_INFINITY); + while (k < raw.length && isP(raw[k])) k++; + if (k >= raw.length || raw[k].anchorSec >= tail) continue; + console.log( + `stretch ${s.startSec.toFixed(2)}-${s.endSec.toFixed(2)}: "${raw[k].word}" starts ${ms(raw[k].startSec - s.startSec)} from the onset raw, ${ms(post[k].startSec - s.startSec)} after the post-pass`, + ); + while (k < raw.length && raw[k].anchorSec < tail) k++; +} diff --git a/tools/stt-eval/word-timing/run-helper.mjs b/tools/stt-eval/word-timing/run-helper.mjs new file mode 100644 index 000000000..291acd5d4 --- /dev/null +++ b/tools/stt-eval/word-timing/run-helper.mjs @@ -0,0 +1,53 @@ +// Runs whisper-stt-server over the corpus and saves its raw verbose_json. +// Usage: node run-helper.mjs [--cpu] [--condition clean|noisy|both] +// [--language auto|fr|en] [--only ] [--env KEY=VAL] +// -> /results/raw//..json (+ wallMs) +// Clips already done are skipped, so an interrupted run resumes. +import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { CLIPS, DATA, RESULTS, startHelper, transcribe } from "./lib.mjs"; + +const [tag, exe, ...rest] = process.argv.slice(2); +if (!tag || !exe) + throw new Error( + "usage: node run-helper.mjs [--cpu] [--condition both] [--language auto] [--only ] [--env K=V]", + ); +const flag = (name, dflt) => { + const i = rest.indexOf(name); + return i >= 0 ? rest[i + 1] : dflt; +}; +const condition = flag("--condition", "both"); +const language = flag("--language", "auto"); // the app sends "auto" +const only = flag("--only", ""); +const env = { ...process.env }; +for (let i = 0; i < rest.length; i++) + if (rest[i] === "--env") { + const [k, v] = rest[i + 1].split("="); + env[k] = v; + } + +const { base, stop } = await startHelper(exe, { cpu: rest.includes("--cpu"), env }); +try { + const outDir = path.join(RESULTS, "raw", tag); + mkdirSync(outDir, { recursive: true }); + const manifest = JSON.parse(readFileSync(path.join(DATA, "corpus-manifest.json"), "utf8")); + const conds = condition === "both" ? ["clean", "noisy"] : [condition]; + for (const c of manifest.clips) { + if (only && !c.id.includes(only)) continue; + for (const cond of conds) { + const out = path.join(outDir, `${c.id}.${cond}.json`); + if (existsSync(out)) continue; + const wav = path.join(CLIPS, cond === "clean" ? `${c.id}.wav` : `${c.id}.noisy.wav`); + const { json, wallMs } = await transcribe(base, wav, language); + writeFileSync(out, JSON.stringify({ ...json, wallMs })); + console.log( + `${c.id}.${cond}`, + `${(wallMs / 1000).toFixed(2)} s`, + json.detected_language, + json.backend, + ); + } + } +} finally { + stop(); +} diff --git a/tools/stt-eval/word-timing/summarize.mjs b/tools/stt-eval/word-timing/summarize.mjs new file mode 100644 index 000000000..849673c27 --- /dev/null +++ b/tools/stt-eval/word-timing/summarize.mjs @@ -0,0 +1,39 @@ +// Markdown table of evaluated configurations, from /results/.json. +// Usage: node summarize.mjs [