diff --git a/Sources/DoNotTypeCore/AudioChunker.swift b/Sources/DoNotTypeCore/AudioChunker.swift
index bac738b..fa311de 100644
--- a/Sources/DoNotTypeCore/AudioChunker.swift
+++ b/Sources/DoNotTypeCore/AudioChunker.swift
@@ -86,6 +86,15 @@ public enum AudioChunker {
var chunks: [Chunk] = []
var start = 0
+ // One pass over the whole recording, reused for every boundary decision below. Silero is
+ // sequential, so this is also the only correct way to run it here: analysing each tail
+ // separately would restart the model's state mid-utterance.
+ let speech: SpeechActivity.Stream? = {
+ guard let stream = try? SpeechActivity.Stream() else { return nil }
+ guard (try? stream.append(pcm: body)) != nil else { return nil }
+ return stream.speechSegments(includingOpenRun: true).isEmpty ? nil : stream
+ }()
+
while start < body.count {
let remaining = body.count - start
@@ -99,7 +108,12 @@ public enum AudioChunker {
}
let tail = body.subdata(in: start..
[Chunk] {
guard !pcm.isEmpty else { return [] }
pending.append(pcm)
totalBytes += pcm.count
+ // Fed every tick, whether or not a boundary is wanted yet, because the model's state
+ // is sequential: skipping audio would leave it reading a discontinuity.
+ try? speech?.append(pcm: pcm)
var ready: [Chunk] = []
while shouldAnalyse,
let cut = AudioChunker.bestBoundary(
- in: pending, format: format, policy: policy)
+ in: pending, format: format, policy: policy, pauses: currentPauses())
{
let samples = pending.subdata(in: 0.. [Pause]? {
+ guard let speech else { return nil }
+ let windowBytes = SpeechActivity.windowSamples * 2
+ let analysedBytes = speech.analysedSamples * 2
+ guard analysedBytes + windowBytes >= startBytes + pending.count else { return nil }
+
+ // No finalised speech at all, in a minute of audio, means the model is not seeing this
+ // recording — an unusual timbre, or something that is not speech. Refusing to cut on
+ // that basis would let one request grow without bound, so the energy finder takes over.
+ // One segment and no gaps is the opposite: Silero *is* reading it and says the speaker
+ // has not paused, which is a reason to wait rather than to overrule it.
+ let segments = speech.speechSegments(includingOpenRun: true)
+ guard segments.count >= 2 else { return nil }
+ // The open run counts here, and it has to. During capture the gap worth cutting at is
+ // always the most recent one, and the run after it has not ended — the speaker is
+ // still going. Only that run's *start* is used to close the gap, and Silero does not
+ // report a start until 250 ms of speech has confirmed it, which is exactly the
+ // evidence that the pause is over. Excluding it made the newest pause invisible and
+ // the segmenter never cut.
+ return speech.pauses(
+ from: startBytes / 2, format: format, includingOpenRun: true)
+ }
+
private var canConsiderBoundary: Bool {
if !emittedFirst {
return Double(totalBytes) / Double(format.bytesPerSecond) > AudioChunker.threshold
@@ -217,11 +278,25 @@ public enum AudioChunker {
durationSeconds: Double(samples.count) / Double(format.bytesPerSecond))
}
- private struct PauseCandidate {
- let cut: Int
- let seconds: Double
- let duration: Double
- let depth: Double
+ /// A place a cut could go, from whichever detector found it.
+ ///
+ /// Two produce these — the frame-energy finder below and Silero's speech-segment gaps — and one
+ /// scorer ranks them, so the two detectors stay swappable and directly comparable.
+ public struct Pause: Sendable, Equatable {
+ /// Byte offset of the pause's midpoint, relative to the start of the buffer searched.
+ public let cut: Int
+ public let seconds: Double
+ public let duration: Double
+ /// How confidently this is not speech, on a 0–20 scale. Decibels below the speech
+ /// threshold for the energy finder; the model's own confidence for Silero.
+ public let depth: Double
+
+ public init(cut: Int, seconds: Double, duration: Double, depth: Double) {
+ self.cut = cut
+ self.seconds = seconds
+ self.duration = duration
+ self.depth = depth
+ }
}
/// Returns the best safe boundary, or nil when the audio has no energy-qualified pause.
@@ -233,12 +308,39 @@ public enum AudioChunker {
/// A run counts as a boundary only when
/// it is surrounded by speech; uniform noise therefore cannot masquerade as one enormous
/// pause. The middle leaves acoustic context on both sides without duplicating samples.
+ /// Ranks pauses found by either detector and returns the chosen cut.
+ ///
+ /// `pauses` of nil means "find them yourself, from frame energy" — the behaviour every caller
+ /// had before Silero could supply them, and the fallback when the model will not load.
static func bestBoundary(
- in body: Data, format: Format = Format(), policy: BoundaryPolicy = defaultPolicy
+ in body: Data, format: Format = Format(), policy: BoundaryPolicy = defaultPolicy,
+ pauses: [Pause]? = nil
) -> Int? {
+ let found = pauses ?? energyPauses(in: body, format: format, policy: policy)
+ let eligible = found.filter { $0.duration >= policy.minimumPause && $0.seconds >= policy.minimum }
+ let preferred = eligible.filter { $0.seconds <= policy.horizon }
+ if !preferred.isEmpty {
+ return preferred.max { boundaryScore($0, policy: policy) < boundaryScore($1, policy: policy) }?.cut
+ }
+ // Past the decision horizon, use the first real pause instead of waiting for a prettier
+ // one. Still no pause, still no cut.
+ return eligible.min { $0.seconds < $1.seconds }?.cut
+ }
+
+ /// Every energy-qualified pause in the buffer.
+ ///
+ /// This is deliberately only a pause finder; Silero makes every decision about whether a chunk
+ /// contains speech. The floor is a low percentile of the recording's own energy so a train and
+ /// a quiet office are judged relative to themselves. The second percentile is intentional:
+ /// ordinary speech can contain less than ten percent pause, while splitting needs only one
+ /// quiet run. A run counts only when it is surrounded by speech, so uniform noise cannot
+ /// masquerade as one enormous pause.
+ static func energyPauses(
+ in body: Data, format: Format = Format(), policy: BoundaryPolicy = defaultPolicy
+ ) -> [Pause] {
let frameMilliseconds = 20
let frameBytes = format.bytesPerSecond * frameMilliseconds / 1_000
- guard frameBytes > 0, body.count >= frameBytes * 3 else { return nil }
+ guard frameBytes > 0, body.count >= frameBytes * 3 else { return [] }
var levels: [Double] = []
levels.reserveCapacity(body.count / frameBytes)
@@ -263,7 +365,7 @@ public enum AudioChunker {
offset += frameBytes
}
}
- guard !levels.isEmpty else { return nil }
+ guard !levels.isEmpty else { return [] }
let sorted = levels.sorted()
let floor = sorted[min(sorted.count - 1, sorted.count / 50)]
@@ -273,7 +375,7 @@ public enum AudioChunker {
let evidenceFrames = 5 // 100 ms of speech on each side defeats isolated transients.
let evidenceWindow = 100 // two seconds
- var candidates: [PauseCandidate] = []
+ var candidates: [Pause] = []
var frame = 0
while frame < speaking.count {
guard !speaking[frame] else {
@@ -297,21 +399,14 @@ public enum AudioChunker {
let gapLevel = levels[runStart.. Double {
+ private static func boundaryScore(_ candidate: Pause, policy: BoundaryPolicy) -> Double {
let preferredBonus = candidate.duration >= policy.preferredPause ? 3.0 : 0
let duration = min(2, candidate.duration) * 4
let depth = min(20, candidate.depth) / 10
diff --git a/Sources/DoNotTypeCore/SpeechActivity.swift b/Sources/DoNotTypeCore/SpeechActivity.swift
index 930812f..3b6d392 100644
--- a/Sources/DoNotTypeCore/SpeechActivity.swift
+++ b/Sources/DoNotTypeCore/SpeechActivity.swift
@@ -105,17 +105,24 @@ public enum SpeechActivity {
// MARK: - Silero segmentation
- /// The part of upstream `get_speech_timestamps` that decides whether final speech exists.
- /// Padding and maximum-segment splitting do not affect this gate, which needs duration rather
- /// than timestamps; preserving the hysteresis and strict minimum-duration comparison does.
- private static func finalisedSpeechSamples(
- probabilities: [Float], audioLengthSamples: Int
- ) -> Int {
+ /// The part of upstream `get_speech_timestamps` that decides where final speech is.
+ ///
+ /// One implementation, used by both the yes/no gate and the boundary finder, because a
+ /// segmenter that disagreed with the gate about what counts as speech would cut in places the
+ /// gate then refused to send. Padding and maximum-segment splitting do not affect either
+ /// caller; preserving the hysteresis and the strict minimum-duration comparison does.
+ ///
+ /// - Parameter includeOpenSegment: whether a run still open at the end counts. True for a
+ /// finished recording, whose end really is the end. False for a live capture, where the
+ /// speaker has simply not stopped yet and the run's end is not known.
+ static func finalisedSpeechSegments(
+ probabilities: [Float], audioLengthSamples: Int, includeOpenSegment: Bool = true
+ ) -> [Range] {
let minimumSpeechSamples = sampleRate * minimumSpeechMilliseconds / 1_000
let minimumSilenceSamples = sampleRate * minimumSilenceMilliseconds / 1_000
var speechStart: Int?
var possibleEnd: Int?
- var total = 0
+ var segments: [Range] = []
for (index, probability) in probabilities.enumerated() {
let current = windowSamples * index
@@ -130,15 +137,133 @@ public enum SpeechActivity {
if possibleEnd == nil { possibleEnd = current }
guard let end = possibleEnd, current - end >= minimumSilenceSamples else { continue }
- if end - start > minimumSpeechSamples { total += end - start }
+ if end - start > minimumSpeechSamples { segments.append(start.. minimumSpeechSamples {
- total += audioLengthSamples - start
+ if includeOpenSegment, let start = speechStart,
+ audioLengthSamples - start > minimumSpeechSamples
+ {
+ segments.append(start.. Int {
+ finalisedSpeechSegments(
+ probabilities: probabilities, audioLengthSamples: audioLengthSamples
+ ).reduce(0) { $0 + $1.count }
+ }
+
+ // MARK: - Streaming
+
+ /// Silero over a capture that is still running, carrying state instead of re-reading the buffer.
+ ///
+ /// The live segmenter asks for a boundary every 200 ms once a minute of audio is pending. Doing
+ /// that by re-running the model over the whole pending buffer costs about 0.19 s of CPU per
+ /// call at 60 seconds pending — roughly a whole core, sustained, for as long as no qualifying
+ /// pause appears. Feeding only the new samples costs about 0.6 ms.
+ ///
+ /// Probabilities are kept for the whole capture rather than trimmed at each cut. One `Float`
+ /// per 512 samples is 31 floats a second, so an hour of dictation is under half a megabyte, and
+ /// keeping absolute indices means a cut never has to re-align the array against the buffer.
+ public final class Stream: @unchecked Sendable {
+ private let model: SileroModel
+ private var carry: SileroModel.Carry
+ /// Samples arrived but not yet a complete 512-sample window.
+ private var leftover = Data()
+ private var probabilities: [Float] = []
+ /// Samples represented by `probabilities`, so callers can index in absolute samples.
+ public private(set) var analysedSamples = 0
+
+ public init() throws {
+ switch SpeechActivity.modelState {
+ case .ready(let loaded): model = loaded
+ case .failed(let detail): throw DetectorError.unavailable(detail)
+ }
+ carry = SileroModel.Carry()
+ }
+
+ /// Feeds 16 kHz mono 16-bit PCM. Safe to call with any size, including a partial window.
+ public func append(pcm: Data) throws {
+ guard !pcm.isEmpty else { return }
+ leftover.append(pcm)
+ let windowBytes = SpeechActivity.windowSamples * 2
+ let complete = leftover.count / windowBytes
+ guard complete > 0 else { return }
+
+ let consumed = complete * windowBytes
+ let block = leftover.prefix(consumed)
+ leftover.removeSubrange(0.. [Range] {
+ SpeechActivity.finalisedSpeechSegments(
+ probabilities: probabilities, audioLengthSamples: analysedSamples,
+ includeOpenSegment: includingOpenRun)
+ }
+
+ /// Boundary candidates in the gaps between finalised speech, relative to `originSample`.
+ func pauses(
+ from originSample: Int, format: AudioChunker.Format, includingOpenRun: Bool = false
+ ) -> [AudioChunker.Pause] {
+ SpeechActivity.pauses(
+ segments: speechSegments(includingOpenRun: includingOpenRun),
+ probabilities: probabilities, from: originSample, format: format)
+ }
+ }
+
+ /// The gaps between finalised speech runs, as boundary candidates.
+ ///
+ /// A gap is bounded by two runs that each cleared Silero's 250 ms minimum, so it is flanked by
+ /// real speech by construction — the energy finder has to check that separately with a
+ /// five-frames-in-two-seconds heuristic.
+ ///
+ /// `depth` is the model's own confidence that the gap is not speech, scaled to the 0–20 range
+ /// the energy finder's decibel depth uses, so one scorer can rank candidates from either source.
+ static func pauses(
+ segments: [Range], probabilities: [Float], from originSample: Int,
+ format: AudioChunker.Format
+ ) -> [AudioChunker.Pause] {
+ var out: [AudioChunker.Pause] = []
+ for (a, b) in zip(segments, segments.dropFirst()) {
+ let gapStart = a.upperBound
+ let gapEnd = b.lowerBound
+ guard gapEnd > gapStart, gapStart >= originSample else { continue }
+
+ let firstWindow = gapStart / windowSamples
+ let lastWindow = max(firstWindow + 1, gapEnd / windowSamples)
+ let slice = probabilities[
+ min(firstWindow, probabilities.count).. [Float] {
+ var carry = Carry()
+ return try probabilities(pcm: pcm, sampleCount: sampleCount, carry: &carry)
+ }
+
+ /// The recurrent state Silero carries from one 512-sample window to the next.
+ ///
+ /// Split out so a live capture can keep feeding the same session instead of re-running the
+ /// model over the whole buffer on every tick. Upstream's own streaming mode is exactly this:
+ /// the state and the 64-sample context are the entire history the model needs.
+ struct Carry {
+ var state = [Float](repeating: 0, count: 2 * 128)
+ var context = [Float](repeating: 0, count: 64)
+ }
+
+ func probabilities(pcm: Data, sampleCount: Int, carry: inout Carry) throws -> [Float] {
try pcm.withUnsafeBytes { raw in
- var state = [Float](repeating: 0, count: 2 * 128)
- var context = [Float](repeating: 0, count: 64)
+ var state = carry.state
+ var context = carry.context
+ defer {
+ carry.state = state
+ carry.context = context
+ }
var probabilities: [Float] = []
probabilities.reserveCapacity(
(sampleCount + SpeechActivity.windowSamples - 1)
diff --git a/Tests/DoNotTypeCoreTests/SpeechStreamTests.swift b/Tests/DoNotTypeCoreTests/SpeechStreamTests.swift
new file mode 100644
index 0000000..a5dcd8b
--- /dev/null
+++ b/Tests/DoNotTypeCoreTests/SpeechStreamTests.swift
@@ -0,0 +1,142 @@
+import XCTest
+
+@testable import DoNotTypeCore
+
+/// The streaming Silero session, which is the part of the boundary change that can go subtly wrong.
+///
+/// Silero is recurrent: every 512-sample window is evaluated with the state and the 64-sample
+/// context left by the one before it. Feeding a live capture therefore means carrying that state
+/// across calls rather than re-reading the buffer, and a bug in the carry produces probabilities
+/// that are *plausible* rather than wrong-looking — which is exactly the kind of error a synthetic
+/// fixture never catches.
+final class SpeechStreamTests: XCTestCase {
+ private func realSpeech() throws -> Data {
+ let url = URL(fileURLWithPath: #filePath)
+ .deletingLastPathComponent().deletingLastPathComponent()
+ .deletingLastPathComponent()
+ .appendingPathComponent("eval/audio/real-acronym.wav")
+ try XCTSkipUnless(
+ FileManager.default.fileExists(atPath: url.path), "eval fixture not present")
+ let wav = try Data(contentsOf: url)
+ return try XCTUnwrap(AudioChunker.pcmBody(of: wav))
+ }
+
+ /// The carry is correct if arrival size cannot change the answer.
+ ///
+ /// 85 ms is what the macOS recorder's tap actually delivers, and 1 s and 4 s bracket it either
+ /// side. All three must produce identical speech segments, because the audio is identical.
+ func testArrivalSizeDoesNotChangeTheResult() throws {
+ let pcm = try realSpeech()
+ var results: [[Range]] = []
+
+ for chunkSeconds in [0.085, 1.0, 4.0] {
+ let stream = try SpeechActivity.Stream()
+ let step = max(2, Int(chunkSeconds * 16_000) * 2)
+ var offset = 0
+ while offset < pcm.count {
+ let end = min(offset + step, pcm.count)
+ try stream.append(pcm: pcm.subdata(in: offset..= 2, "this clip has no finalised gap to test")
+ let pauses = stream.pauses(from: 0, format: AudioChunker.Format())
+ XCTAssertEqual(pauses.count, segments.count - 1)
+
+ for pause in pauses {
+ XCTAssertGreaterThan(pause.duration, 0)
+ XCTAssertGreaterThanOrEqual(pause.depth, 0)
+ XCTAssertLessThanOrEqual(pause.depth, 20)
+ // The midpoint must land inside a gap, never inside a run.
+ let sample = pause.cut / 2
+ for segment in segments {
+ XCTAssertFalse(segment.contains(sample), "a cut landed inside speech")
+ }
+ }
+ }
+}