From fb340eaea52668ba4d2288be1f41cbfa2cd640e3 Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Fri, 25 Sep 2026 11:54:31 -0400 Subject: [PATCH] fix(vocab): make alignBaseWordsToUTF8Ranges iterative (#961) The exact-alignment search recursed once per base word and passed `ranges + [match]` down each level, so stack depth was linear and live memory quadratic in transcript length. ctcTokenEvaluateCandidates SIGBUSed inside Swift concurrency tasks at ~1-2k words (512 KB stacks), and each level also scanned to the end of the text looking for a second alignment, making time quadratic too. Rewrite as explicit-stack backtracking over a single shared path: - a word can only start inside the delimiter run after the cursor, so candidates per level are those positions, probed with an anchored literal match instead of an unbounded forward search - (wordIndex, cursor) states that produced no complete alignment are memoized, bounding pathological delimiter-only words - uniqueness semantics unchanged: 0 or >=2 alignments -> all nil Differential check against the old recursive implementation: 0 mismatches over 200k randomized texts (quotes, C++, combining marks, repeated words, delimiter-only tokens). 50k words align in 21 ms inside a detached Task; the old code SIGBUSes at 1k in the same harness. --- .../VocabularyRescorer+Utilities.swift | 92 ++++++++++++------- .../VocabularyCandidateEvidenceTests.swift | 17 ++++ 2 files changed, 77 insertions(+), 32 deletions(-) diff --git a/Sources/FluidAudio/ASR/Parakeet/SlidingWindow/CustomVocabulary/Rescorer/VocabularyRescorer+Utilities.swift b/Sources/FluidAudio/ASR/Parakeet/SlidingWindow/CustomVocabulary/Rescorer/VocabularyRescorer+Utilities.swift index 1278bf1f9..0320a4d20 100644 --- a/Sources/FluidAudio/ASR/Parakeet/SlidingWindow/CustomVocabulary/Rescorer/VocabularyRescorer+Utilities.swift +++ b/Sources/FluidAudio/ASR/Parakeet/SlidingWindow/CustomVocabulary/Rescorer/VocabularyRescorer+Utilities.swift @@ -143,43 +143,71 @@ extension VocabularyRescorer { return Array(repeating: nil, count: baseWords.count) } - var alignments: [[Range]] = [] - - func search( - wordIndex: Int, - cursor: String.Index, - ranges: [Range] - ) { - guard alignments.count < 2 else { return } - guard wordIndex < baseWords.count else { - guard isDelimiterOnly(baseText[cursor..] = [] + var path: [Range] = [] + var deadStates = Set() + + func frame(wordIndex: Int, cursor: String.Index) -> Frame { + let limit = baseText[cursor...].firstIndex(where: { !isInterWordDelimiter($0) }) ?? baseText.endIndex + return Frame( + state: State(wordIndex: wordIndex, cursor: cursor), + limit: limit, + next: cursor, + alignmentsAtEntry: alignmentCount + ) + } + + var stack = [frame(wordIndex: 0, cursor: baseText.startIndex)] + path.reserveCapacity(baseWords.count) + + while alignmentCount < 2, let top = stack.last { + let wordIndex = top.state.wordIndex + if wordIndex == baseWords.count { + if top.limit == baseText.endIndex { + alignmentCount += 1 + if alignmentCount == 1 { alignment = path } } - guard alignments.count < 2, match.lowerBound < baseText.endIndex else { return } - searchStart = baseText.index(after: match.lowerBound) + stack.removeLast() + if !path.isEmpty { path.removeLast() } + continue + } + + guard let start = top.next, start < baseText.endIndex else { + if alignmentCount == top.alignmentsAtEntry { deadStates.insert(top.state) } + stack.removeLast() + if !path.isEmpty { path.removeLast() } + continue } + stack[stack.count - 1].next = start < top.limit ? baseText.index(after: start) : nil + + guard + let match = baseText.range( + of: baseWords[wordIndex], + options: [.literal, .anchored], + range: start..