From 2c3e70e2f51ae104081e6a2449d5d502872b3826 Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Thu, 1 Oct 2026 02:31:31 +0200 Subject: [PATCH 1/3] feat(stt): align words with character-level DTW whisper.cpp's DTW pass teacher-forces the decoded BPE tokens; ours feeds the text back one character per decoder row (arXiv 2509.09987), averages the alignment heads with per-frame L2 normalisation, and drops one silent final consonant per French word. It is spliced into a build-tree copy of whisper.cpp at configure time, so the fetched source stays untouched. Word-timing harness, inner starts: median 31 -> 16 ms, within 50 ms 64% -> 86% (FR 83%, EN 89%); clean single-word cuts 19% -> 39%. Refs #948 --- electron/native/whisper-stt/CMakeLists.txt | 32 +++ .../whisper-stt/whisper-patches/char-dtw.cpp | 227 ++++++++++++++++++ .../transcription-and-captions.md | 36 ++- tools/stt-eval/word-timing/README.md | 3 +- tools/stt-eval/word-timing/lib.mjs | 4 +- 5 files changed, 295 insertions(+), 7 deletions(-) create mode 100644 electron/native/whisper-stt/whisper-patches/char-dtw.cpp diff --git a/electron/native/whisper-stt/CMakeLists.txt b/electron/native/whisper-stt/CMakeLists.txt index b95baf7d1..c634316ce 100644 --- a/electron/native/whisper-stt/CMakeLists.txt +++ b/electron/native/whisper-stt/CMakeLists.txt @@ -144,6 +144,38 @@ endif() FetchContent_MakeAvailable(whisper httplib json) +# Character-level DTW (issue #948): whisper.cpp's word-timing pass, +# whisper_exp_compute_token_level_timestamps_dtw, and the median filter only it +# used, are replaced by whisper-patches/char-dtw.cpp. The whisper target +# compiles that copy of whisper.cpp from the build tree, so the fetched source +# is never modified: a shared FETCHCONTENT_SOURCE_DIR_WHISPER and the Nix build +# get the same result. Bumping WHISPER_REF fails here, loudly, if the code moved. +set(OSC_DTW_FRAGMENT "${CMAKE_CURRENT_SOURCE_DIR}/whisper-patches/char-dtw.cpp") +set(OSC_WHISPER_CPP "${whisper_SOURCE_DIR}/src/whisper.cpp") +set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS "${OSC_DTW_FRAGMENT}" "${OSC_WHISPER_CPP}") +file(READ "${OSC_WHISPER_CPP}" osc_src) +string(REPLACE "\r\n" "\n" osc_src "${osc_src}") +string(FIND "${osc_src}" "\nstruct median_filter_user_data {" osc_begin) +string(FIND "${osc_src}" "\nvoid whisper_log_set(" osc_end) +if(osc_begin GREATER -1 AND osc_end GREATER osc_begin) + math(EXPR osc_len "${osc_end} - ${osc_begin}") + string(SUBSTRING "${osc_src}" ${osc_begin} ${osc_len} osc_old) + string(FIND "${osc_old}" "static void whisper_exp_compute_token_level_timestamps_dtw(" osc_fn) +endif() +if(osc_begin EQUAL -1 OR NOT osc_end GREATER osc_begin OR osc_fn EQUAL -1) + message(FATAL_ERROR "char-dtw: the DTW pass was not found where expected in ${OSC_WHISPER_CPP}") +endif() +string(SUBSTRING "${osc_src}" 0 ${osc_begin} osc_head) +string(SUBSTRING "${osc_src}" ${osc_end} -1 osc_tail) +file(READ "${OSC_DTW_FRAGMENT}" osc_fragment) +file(WRITE "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp.tmp" + "${osc_head}\n${osc_fragment}${osc_tail}") +configure_file("${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp.tmp" + "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp" COPYONLY) +get_target_property(osc_whisper_srcs whisper SOURCES) +list(TRANSFORM osc_whisper_srcs REPLACE "^whisper\\.cpp$" "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp") +set_property(TARGET whisper PROPERTY SOURCES ${osc_whisper_srcs}) + add_executable(whisper-stt-server src/main.cpp) target_link_libraries(whisper-stt-server PRIVATE whisper diff --git a/electron/native/whisper-stt/whisper-patches/char-dtw.cpp b/electron/native/whisper-stt/whisper-patches/char-dtw.cpp new file mode 100644 index 000000000..15a024930 --- /dev/null +++ b/electron/native/whisper-stt/whisper-patches/char-dtw.cpp @@ -0,0 +1,227 @@ +// OpenScreen (issue #948): character-level teacher-forced DTW. +// +// Spliced into whisper.cpp v1.9.1 by ../CMakeLists.txt in place of everything +// from `struct median_filter_user_data` to the end of +// whisper_exp_compute_token_level_timestamps_dtw. Same signature, same output: +// a `t_dtw` on every text token of the window, in centiseconds. +// +// Upstream teacher-forces the decoded BPE tokens once more and runs DTW on the +// cross-attention of the alignment heads. This does the same with the text cut +// into single characters ("Whisper Has an Internal Word Aligner", +// arXiv 2509.09987): one decoder row per character, so a word boundary is +// where the path reaches the row that predicts the space or the letter after +// it, not wherever a multi-letter token happens to start. On the word-timing +// harness (tools/stt-eval/word-timing) it takes the inner-boundary median from +// 31 to about 17 ms. +// +// t_dtw keeps its upstream meaning: the time the path enters the row that +// predicts whatever follows the token, i.e. the END of the token. +// +// Departures from upstream, all measured on the harness: +// - the attention of the heads is averaged and each audio frame's column is +// scaled to unit L2 norm, as in the paper. Upstream's z-score + median filter +// does worse on characters (+8 ms median); +// - French drops one silent final consonant per word before aligning (vais, +// vous, plaît): its row otherwise eats the start of the next word, +60 to +// +140 ms on the words after it; +// - the decoder has n_text_ctx positions. A window too long as characters is +// aligned in several runs: one stretch of it as characters, the rest of the +// window around it as BPE tokens; +// - runs never step back in time: each t_dtw is at least the previous one. +static void whisper_exp_compute_token_level_timestamps_dtw( + struct whisper_context * ctx, + struct whisper_state * state, + struct whisper_full_params params, + int i_segment, + size_t n_segments, + int seek, + int n_frames, + int /*medfilt_width*/, + int n_threads) +{ + const int n_audio_ctx = state->exp_n_audio_ctx > 0 ? state->exp_n_audio_ctx : ctx->model.hparams.n_audio_ctx; + WHISPER_ASSERT(n_frames <= n_audio_ctx * 2); + WHISPER_ASSERT(ctx->params.dtw_aheads_preset != WHISPER_AHEADS_NONE); + + const whisper_token eot = whisper_token_eot(ctx); + const auto & vocab = ctx->vocab; + + std::vector text; + for (size_t i = i_segment; i < i_segment + n_segments; ++i) { + for (auto & t : state->result_all[i].tokens) { + if (t.id < eot) { + text.push_back(&t); + } + } + } + if (text.empty()) { + return; + } + + std::vector sot_seq = { whisper_token_sot(ctx) }; + if (whisper_is_multilingual(ctx)) { + const int lang_id = whisper_lang_id(params.language); + state->lang_id = lang_id; + sot_seq.push_back(whisper_token_lang(ctx, lang_id)); + } + sot_seq.push_back(whisper_token_not(ctx)); + + // Each text token as one token per character. A character with no token of + // its own goes as its bytes; a token that cannot be split stays whole. + const auto find_id = [&](const std::string & s, whisper_token & id) { + const auto it = vocab.token_to_id.find(s); + if (it == vocab.token_to_id.end()) return false; + id = it->second; + return true; + }; + std::vector> chars(text.size()); + for (size_t k = 0; k < text.size(); ++k) { + const std::string & s = vocab.id_to_token.at(text[k]->id); + std::vector out; + bool ok = true; + for (size_t i = 0; ok && i < s.size();) { + const unsigned char c = s[i]; + const size_t len = std::min(c < 0x80 ? 1 : (c >> 5) == 0x6 ? 2 : (c >> 4) == 0xE ? 3 : (c >> 3) == 0x1E ? 4 : 1, s.size() - i); + const std::string piece = s.substr(i, len); + i += len; + whisper_token id; + if (find_id(piece, id)) { + out.push_back(id); + continue; + } + for (size_t b = 0; ok && b < piece.size(); ++b) { + ok = find_id(piece.substr(b, 1), id); + if (ok) out.push_back(id); + } + } + chars[k] = ok ? std::move(out) : std::vector{ text[k]->id }; + } + + if (state->lang_id == whisper_lang_id("fr")) { + // One silent final consonant per word of two letters or more. A word is a + // run of letters, across tokens; a byte >= 0x80 counts as a letter. + const auto ch = [&](whisper_token id) { + const std::string & s = vocab.id_to_token.at(id); + return s.size() == 1 ? (unsigned char) s[0] : (unsigned char) 0x80; + }; + std::vector> word; // (token, character) + const auto end_word = [&]() { + if (word.size() >= 2 && strchr("stxdz", tolower(ch(chars[word.back().first][word.back().second]))) != nullptr) { + chars[word.back().first][word.back().second] = -1; + } + word.clear(); + }; + for (size_t k = 0; k < chars.size(); ++k) { + for (size_t i = 0; i < chars[k].size(); ++i) { + const unsigned char c = ch(chars[k][i]); + if (c >= 0x80 || isalpha(c)) { + word.push_back({ k, i }); + } else { + end_word(); + } + } + } + end_word(); + for (auto & cs : chars) { + cs.erase(std::remove(cs.begin(), cs.end(), -1), cs.end()); + } + } + + struct ggml_init_params gparams = { + /*.mem_size =*/ ctx->params.dtw_mem_size, + /*.mem_buffer =*/ NULL, + /*.no_alloc =*/ false, + }; + + int64_t t_floor = 0; + for (int i = i_segment - 1; i >= 0 && t_floor == 0; --i) { + for (const auto & t : state->result_all[i].tokens) { + if (t.id < eot && t.t_dtw > t_floor) t_floor = t.t_dtw; + } + } + + const int n_ctx = ctx->model.hparams.n_text_ctx; + const int n_text = (int) text.size(); + for (int k0 = 0; k0 < n_text;) { + // Tokens [k0, k1) as characters, as many as fit. + int len = (int) sot_seq.size() + n_text + 1; + int k1 = k0; + while (k1 < n_text && len - 1 + (int) chars[k1].size() <= n_ctx) { + len += (int) chars[k1].size() - 1; + ++k1; + } + if (k1 == k0) { + chars[k0] = { text[k0]->id }; + continue; + } + + std::vector tokens = sot_seq; + std::vector first(n_text + 1); // each text token's first row after `not` + for (int k = 0; k < n_text; ++k) { + first[k] = (int) (tokens.size() - sot_seq.size()); + if (k >= k0 && k < k1) { + tokens.insert(tokens.end(), chars[k].begin(), chars[k].end()); + } else { + tokens.push_back(text[k]->id); + } + } + first[n_text] = (int) (tokens.size() - sot_seq.size()); + tokens.push_back(eot); + + whisper_kv_cache_clear(state->kv_self); + whisper_batch_prep_legacy(state->batch, tokens.data(), tokens.size(), 0, 0); + whisper_kv_cache_seq_rm(state->kv_self, 0, 0, -1); + if (!whisper_decode_internal(*ctx, *state, state->batch, n_threads, true, nullptr, nullptr)) { + WHISPER_LOG_INFO("DECODER FAILED\n"); + WHISPER_ASSERT(0); + } + const ggml_tensor * qk = state->aheads_cross_QKs; + WHISPER_ASSERT(qk != nullptr && qk->type == GGML_TYPE_F32 && ggml_is_contiguous(qk)); + const int64_t n_tok = qk->ne[0]; + const int64_t n_actx = qk->ne[1]; + const int64_t n_head = qk->ne[2]; + const int64_t F = n_frames / 2; // 20 ms per frame + WHISPER_ASSERT(F <= n_actx); + auto & data = state->aheads_cross_QKs_data; + data.resize(n_tok * n_actx * n_head); + ggml_backend_tensor_get(qk, data.data(), 0, sizeof(float) * data.size()); + + // Rows: `not`, then every token after it but eot. Row r predicts the + // token after it, so entering row r is where that token starts. + const int64_t row0 = (int64_t) sot_seq.size() - 1; + const int64_t R = (int64_t) tokens.size() - 1 - row0; + + struct ggml_context * gctx = ggml_init(gparams); + ggml_tensor * x = ggml_new_tensor_2d(gctx, GGML_TYPE_F32, R, F); + float * xd = (float *) x->data; + for (int64_t f = 0; f < F; ++f) { + double norm = 0.0; + for (int64_t r = 0; r < R; ++r) { + float a = 0.0f; + for (int64_t h = 0; h < n_head; ++h) { + a += data[(row0 + r) + f * n_tok + h * n_tok * n_actx]; + } + xd[r + f * R] = a; + norm += (double) a * a; + } + norm = std::sqrt(norm) + 1e-9; + for (int64_t r = 0; r < R; ++r) { + xd[r + f * R] = (float) (-xd[r + f * R] / norm); + } + } + + const ggml_tensor * path = dtw_and_backtrace(gctx, x); + std::vector entry(R, 0); + for (int64_t i = path->ne[1] - 1; i >= 0; --i) { + const int32_t r = whisper_get_i32_nd(path, 0, i, 0, 0); + if (r >= 0 && r < R) entry[r] = whisper_get_i32_nd(path, 1, i, 0, 0); + } + ggml_free(gctx); + + for (int k = k0; k < k1; ++k) { + t_floor = std::max(t_floor, (int64_t) entry[first[k + 1]] * 2 + seek); + text[k]->t_dtw = t_floor; + } + k0 = k1; + } +} diff --git a/technical-documentation/architecture/transcription-and-captions.md b/technical-documentation/architecture/transcription-and-captions.md index 65c058b2b..41645e23a 100644 --- a/technical-documentation/architecture/transcription-and-captions.md +++ b/technical-documentation/architecture/transcription-and-captions.md @@ -212,6 +212,29 @@ it verbatim in the response. `flash_attn=false`, which together are the prerequisites for DTW to actually run). `t_dtw == -1` is the DTW-inactive guardrail: the helper fails the request rather than emit zero-quality timestamps. + **The DTW pass is ours, not upstream's.** whisper.cpp teacher-forces the + decoded BPE tokens a second time and aligns their cross-attention on the + audio. [`whisper-patches/char-dtw.cpp`](../../electron/native/whisper-stt/whisper-patches/char-dtw.cpp) + replaces that pass and feeds the text back **one character per decoder + row** instead ("Whisper Has an Internal Word Aligner", arXiv 2509.09987), + so a boundary falls on the row of the space or letter where the next word + starts, not on the edge of a multi-letter token. `CMakeLists.txt` splices it + into a build-tree copy of `whisper.cpp`; the fetched source is never edited, + and bumping `WHISPER_REF` fails the configure if the code it replaces moved. + What it does differently, each measured on the harness below: + - the heads' attention is averaged and each audio frame scaled to unit L2 + norm, as in the paper. Upstream's z-score and median filter do worse on + characters, and the median filter was most of the pass's CPU time; + - French drops one silent final consonant per word (*vais*, *vous*, + *plaît*) before aligning. Its row otherwise holds the start of the next + word: +60 to +140 ms on the word after it; + - the decoder has 448 positions. A window too long as characters is + aligned in several passes, each with one stretch as characters and the + rest of the window as BPE tokens; + - the alignment heads stay `WHISPER_AHEADS_SMALL`. Picking the ten heads + with the most concentrated attention per window out of all 144, as the + paper does, chose nearly the same ten every time and scored worse + (inner median 24 ms, phrase starts 82% within 50 ms). 3. **Word grouping** — BPE tokens join into a single word whenever the detokenized text begins with a space, or at the first token of the segment. @@ -235,7 +258,7 @@ it verbatim in the response. **end** (`tools/stt-eval/whispercpp-dtw-poc/REPORT.md`). 5. **Anchor phrase edges on the speech** — [`electron/stt/snapWordBoundaries.ts`](../../electron/stt/snapWordBoundaries.ts) - leaves boundaries inside a phrase alone: they are within ~30 ms of the audio + leaves boundaries inside a phrase alone: they are within ~16 ms of the audio already, and an energy snap on top of them drags correct boundaries early. The edges of a phrase are different. The token before a phrase's first word is the previous phrase's last one, so that word starts in the pause, as @@ -269,6 +292,10 @@ one word or one phrase. Issue #948 has the method, the baseline and the plan. |---|---|---|---|---| | First-token start + 150 ms RMS snap + VAD edges (before #948) | 105 / 275 ms / 32% | 83% | 5% | 84% | | Previous-token start + two-way VAD edges | 31 / 125 ms / 64% | 89% | 19% | 88% | +| Same, character-level DTW | 16 / 60 ms / 86% | 89% | 39% | 93% | + +Per language, character-level DTW gives French 19 ms median and 83% within +50 ms (was 26 ms, 70%) and English 15 ms and 89% (was 40 ms, 57%). A +15 ms calibration offset on every boundary gained 4 points of inner boundaries within 50 ms but dropped noisy phrase deletes to 82%, so it is not @@ -793,6 +820,7 @@ it deletes data runtime. - **Word timing inside a phrase.** Phrase edges sit on the VAD (step 5), but a word in the middle of continuous speech is only as good as whisper-small's - DTW: about 30 ms median and 125 ms P90, so one single-word delete in five is - clean. Character-level DTW, then a CTC forced aligner, are the upgrade path - (issue #948 Phases 2 and 3); `tools/stt-eval/word-timing` measures them. + character-level DTW: about 16 ms median and 60 ms P90 on synthetic speech, + so about two single-word deletes in five are clean. A CTC forced aligner is + the upgrade path (issue #948 Phase 3); `tools/stt-eval/word-timing` + measures it. diff --git a/tools/stt-eval/word-timing/README.md b/tools/stt-eval/word-timing/README.md index b12ec4649..b49938d6f 100644 --- a/tools/stt-eval/word-timing/README.md +++ b/tools/stt-eval/word-timing/README.md @@ -37,7 +37,8 @@ node evaluate.mjs [--snap ] [--out ] node summarize.mjs "Before=" "After=" ``` -- `run-helper.mjs` starts its own helper on a port in 20500-20599 with the +- `run-helper.mjs` starts its own helper on a port in 20500-20599 (or the + 100 ports from `OSC_WORD_TIMING_PORT`, to run beside another agent) with the app's models (`%APPDATA%/openscreen/stt-models/whisper-ggml`: `ggml-small-q8_0.bin` and `ggml-silero-v6.2.0.bin`), sends every clip like the app does, and stops it. Vulkan takes about 2 min for the 55 min of audio, CPU about 17. diff --git a/tools/stt-eval/word-timing/lib.mjs b/tools/stt-eval/word-timing/lib.mjs index 71391e1a3..85cda97f9 100644 --- a/tools/stt-eval/word-timing/lib.mjs +++ b/tools/stt-eval/word-timing/lib.mjs @@ -24,10 +24,10 @@ export const SR = 16000; /** * Starts whisper-stt-server with the app's models on a random port in - * 20500-20599 and waits until it answers. The process is killed on exit. + * 20500-20599 (OSC_WORD_TIMING_PORT moves the range) and waits until it answers. The process is killed on exit. */ export async function startHelper(exe, { cpu = false, env = process.env } = {}) { - const port = 20500 + Math.floor(Math.random() * 100); + const port = Number(process.env.OSC_WORD_TIMING_PORT ?? 20500) + Math.floor(Math.random() * 100); const args = [ "--model", path.join(MODELS, "ggml-small-q8_0.bin"), From 4ed6a9c7e0d22d99895a0978e884ec5fbc89024c Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Thu, 1 Oct 2026 02:32:19 +0200 Subject: [PATCH 2/3] docs(stt): match the char-dtw comment to the measured numbers --- electron/native/whisper-stt/whisper-patches/char-dtw.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/electron/native/whisper-stt/whisper-patches/char-dtw.cpp b/electron/native/whisper-stt/whisper-patches/char-dtw.cpp index 15a024930..241ac5abd 100644 --- a/electron/native/whisper-stt/whisper-patches/char-dtw.cpp +++ b/electron/native/whisper-stt/whisper-patches/char-dtw.cpp @@ -12,7 +12,7 @@ // where the path reaches the row that predicts the space or the letter after // it, not wherever a multi-letter token happens to start. On the word-timing // harness (tools/stt-eval/word-timing) it takes the inner-boundary median from -// 31 to about 17 ms. +// 31 to 16 ms. // // t_dtw keeps its upstream meaning: the time the path enters the row that // predicts whatever follows the token, i.e. the END of the token. @@ -20,7 +20,7 @@ // Departures from upstream, all measured on the harness: // - the attention of the heads is averaged and each audio frame's column is // scaled to unit L2 norm, as in the paper. Upstream's z-score + median filter -// does worse on characters (+8 ms median); +// does worse on characters (+7 ms median); // - French drops one silent final consonant per word before aligning (vais, // vous, plaît): its row otherwise eats the start of the next word, +60 to // +140 ms on the words after it; From c9bff0f13abd90dee622ca153e84a519a0e26810 Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Thu, 1 Oct 2026 08:40:20 +0200 Subject: [PATCH 3/3] fix(stt): fail configure if the patched whisper.cpp is not compiled; read the harness port from the helper env --- electron/native/whisper-stt/CMakeLists.txt | 3 +++ tools/stt-eval/word-timing/lib.mjs | 2 +- 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/electron/native/whisper-stt/CMakeLists.txt b/electron/native/whisper-stt/CMakeLists.txt index c634316ce..0f6f358da 100644 --- a/electron/native/whisper-stt/CMakeLists.txt +++ b/electron/native/whisper-stt/CMakeLists.txt @@ -174,6 +174,9 @@ configure_file("${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp.tmp" "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp" COPYONLY) get_target_property(osc_whisper_srcs whisper SOURCES) list(TRANSFORM osc_whisper_srcs REPLACE "^whisper\\.cpp$" "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp") +if(NOT "${CMAKE_CURRENT_BINARY_DIR}/whisper-patched/whisper.cpp" IN_LIST osc_whisper_srcs) + message(FATAL_ERROR "char-dtw: the whisper target does not list whisper.cpp as expected (${osc_whisper_srcs})") +endif() set_property(TARGET whisper PROPERTY SOURCES ${osc_whisper_srcs}) add_executable(whisper-stt-server src/main.cpp) diff --git a/tools/stt-eval/word-timing/lib.mjs b/tools/stt-eval/word-timing/lib.mjs index 85cda97f9..5c163a581 100644 --- a/tools/stt-eval/word-timing/lib.mjs +++ b/tools/stt-eval/word-timing/lib.mjs @@ -27,7 +27,7 @@ export const SR = 16000; * 20500-20599 (OSC_WORD_TIMING_PORT moves the range) and waits until it answers. The process is killed on exit. */ export async function startHelper(exe, { cpu = false, env = process.env } = {}) { - const port = Number(process.env.OSC_WORD_TIMING_PORT ?? 20500) + Math.floor(Math.random() * 100); + const port = Number(env.OSC_WORD_TIMING_PORT ?? 20500) + Math.floor(Math.random() * 100); const args = [ "--model", path.join(MODELS, "ggml-small-q8_0.bin"),