From 9781f5d93e6574a994255312f0bd5ab0a0a9f98e Mon Sep 17 00:00:00 2001 From: TheMeinerLP Date: Fri, 21 Aug 2026 09:34:38 +0200 Subject: [PATCH 1/2] perf(worker): transcribe the speech instead of the padded track `WhisperEngine._transcribe` decoded a speaker's recording, found the audible regions with the amplitude gate and then handed faster-whisper the *whole* array, marking the speech with `clip_timestamps`. The library shrinks the array only inside its `vad_filter` branch -- `if vad_filter and clip_timestamps == "0"`, then `audio = np.concatenate(audio_chunks)` -- and setting clips is exactly what skips that branch, so log-mel extraction ran over the padding too and scaled with the padded length of a track rather than with what was said. That is why the worker's memory limit had to be raised from 6Gi to 12Gi in production. The gated speech is now concatenated here and handed over as one short array, with `clip_timestamps` re-expressed on that timeline and the returned segment times mapped back onto the recording's afterwards. Measured end to end through the real `_transcribe`, on a 100-minute mostly-padding track of the shape the recording that started this has -- 187 clips, 41.2 minutes of speech, bit-exact zero between them. Both runs back to back on the same idle machine, `tiny`/int8, beam 8: peak RSS wall clock CPU before 5025 MB 331 s 1276 s after 2441 MB 318 s 1255 s Peak resident memory falls by 2.58 GB, 51%, and it now scales with speech rather than with session length -- a four-hour meeting no longer costs more than a one-hour one for the same amount of talking. The `before` peak moves a few percent between runs (an earlier pair gave 5214 MB); the `after` peak did not move at all across four runs, because it is set by the size of the speech and that is the same array every time. Wall clock is unchanged within run-to-run noise, which is the expected result and not a disappointment: the extraction that was allocating those gigabytes is a small share of the time, and this change deliberately keeps the encoder work identical (see below). `tiny` rather than `large-v3` because those weights are not on the measuring machine. The term that moves is the log-mel extraction, which does not depend on the model at all; large-v3's ~1.55 GB of resident weights adds to both rows equally, putting the old path at ~6.5 GB against a 6Gi (6.44 GB) limit and the new one at ~3.9 GB. Segment counts differ a little between runs of identical code, so they are not a signal either way -- CTranslate2 is not bit-reproducible across threads. The clip boundaries stay, on the new timeline. They are what stops an encoder window spanning two utterances spoken minutes apart: `segment_size = min(nb_max_frames, content_frames - seek, seek_clip_end - seek)` caps a window at the clip it began in. Dropping them -- plain concatenation plus `restore_speech_timestamps`, the library's own helper -- was measured emitting a single 258-second segment covering four utterances spoken 9 s, 47 s, 96 s and 150 s into a recording, because that helper resolves a segment's `start` and `end` independently. So the restore is our own arithmetic, and **a segment is attributed to the clip its encoder window came from, never to the clip its reported times fall in**. `Segment.seek` is that window's first log-mel frame, and the seek loop never decodes a window from outside the clip it is walking, so the frame names one clip and only one. Comparing it against our own clip boundaries on the library's `round(ts * frames_per_second)` grid, then clamping the result into that clip, gives the guarantee the change needs: a restored segment always lies inside the file-timeline extent of the clip whose window produced it. Reading the clip off the times cannot give that. A decoded `end` is `time_offset + end_timestamp_position * time_precision` and `_split_segments_by_timestamps` caps neither it nor the seek positions it derives at the window's content, so a tail segment routinely claims audio the window never held. On the 187-clip track above, with the times-based rule, 57 of 462 segments reported an `end` past their clip and 5 of them overran by more than half the segment -- coming back 20 to 31 seconds from where they were spoken, one of them carrying 14 seconds of text. The error is always exactly one removed gap, so on a session with three utterances forty minutes apart it is forty minutes. That is the same silent wrong timestamp the library's helper produces; relocating it into our own file would not have made it acceptable. Nothing at any test tier observed the library behaviour all of this rests on, and a fake model cannot: a fake breaks the way its author imagined. `test_a_window_never_spans_two_clips` therefore drives the real `generate_segments` with a stand-in `self` -- no weights, no download, so it stays out of the `slow` marker -- and asserts that the windows the real seek loop schedules tile exactly the frame ranges this adapter computes. It fails if `segment_size` stops being capped at the clip, if `Segment.seek` stops being the window's first frame, or if the boundary rounding changes. `pyproject.toml` pins `faster-whisper>=1.1` with no upper bound and Renovate merges on a green build, so that test is the only upper bound there is. The per-clip encoder cost is deliberately not fixed here. 187 clips still pay a 30-second window each, and that is the trade: the 2.2x of encoder work is bought with segment timestamps wrong by up to 258 seconds, and a meeting protocol whose entire output is "who said what, when" cannot spend correctness on CPU time. Making clips longer (`_MERGE_GAP_SECONDS`) recovers it later without touching any of this. `charts/sturnus/values.yaml` is fixed in the same commit: the comment above `resources` still claimed `vad_filter` was what kept the padded length out of the decoder, and it never costed the extraction term at all. The numbers are unchanged -- requests 4Gi, limits 6Gi -- because the point is that they are true again, so the 12Gi override in the deployment repo can be reverted rather than replaced. --- charts/sturnus/values.yaml | 34 +- src/sturnus/infrastructure/whisper.py | 256 +++++++++- tests/infrastructure/test_whisper.py | 676 ++++++++++++++++++++++++-- 3 files changed, 911 insertions(+), 55 deletions(-) diff --git a/charts/sturnus/values.yaml b/charts/sturnus/values.yaml index 2f8e590..b0c9b4b 100644 --- a/charts/sturnus/values.yaml +++ b/charts/sturnus/values.yaml @@ -286,7 +286,8 @@ worker: tmpSizeLimit: 4Gi # Sized for large-v3 at int8_float32 with beam_size 8, which is a # different budget from the large-v3-turbo/beam 5 these numbers were set - # for. Three parts, added up rather than guessed at: + # for. Four parts, added up rather than guessed at -- except the last, + # which peaks before the second one starts and so overlaps it: # # * The weights, resident for the life of the process. 1.55B # parameters quantised to one byte each is ~1.55GB, against ~0.81GB @@ -302,11 +303,28 @@ worker: # that share by 8/5, so ~2.2GB. # * The process itself -- interpreter, numpy, soxr, boto3, SQLAlchemy, # one decrypted WAV read through it. A few hundred MB; call it 0.5GB. + # * The log-mel extraction, which is transient and peaks *before* any + # decoding starts, so it overlaps with the second item rather than + # adding to it. Linear in the array handed to the model, at roughly + # 50MB per minute of it: 2.0GB for the 40.2 minutes of speech in a + # 100-minute session. This is the term nobody costed and the reason + # this limit had to be overridden to 12Gi in the deployment repo -- + # `WhisperEngine` used to hand faster-whisper the *whole* padded + # track and mark the speech with `clip_timestamps`, and the library + # shrinks the array only in the `vad_filter` branch that setting clips + # skips, so extraction scaled with session length and cost 4.94GB for + # that same session. It now concatenates the gated speech before the + # call (`sturnus.infrastructure.whisper`), which is what makes the + # arithmetic here true again and lets the 12Gi override go back to the + # 6Gi below. For scale: on the old path a four-hour session + # extrapolates to ~11.9GB of extraction alone, i.e. it would have + # burst even the 12Gi stopgap. # # ~4.25GB at peak, and measured at **4.11GB** on an x86 CPU pinned to # four threads, transcribing 4.7 minutes of German speech -- close enough # to the arithmetic above to trust it for longer jobs, since only the - # second item grows with the audio. For comparison the previous + # second and fourth items grow with the audio and the fourth peaks first. + # For comparison the previous # configuration (large-v3-turbo, int8, beam 5) peaked at 1.96GB, which is # why the old 2560Mi limit held. # @@ -321,9 +339,15 @@ worker: # # Measured, that ceiling is further away than it looks. The same run # transcribed 282s of speech in 145s -- 1.94x faster than real time, - # against 4.6x for the old configuration. `vad_filter` means only the - # *speech* in a track is decoded, not its padded length, so 1800s of - # lease covers roughly 58 minutes of one person actually talking. Beyond + # against 4.6x for the old configuration. The gate finds the speech and + # `WhisperEngine` concatenates it before the model sees it, so both + # extraction and decoding scale with what was said and not with the padded + # length of a track: 1800s of lease covers roughly 58 minutes of one + # person actually talking. That is also what keeps the memory limit off + # the critical path -- 6Gi covers roughly 60 minutes of one speaker's + # actual speech (60 x 50MB of extraction + 1.55 + 0.5 ~= 5.1GB) and the + # lease expires just before that, so for a four-hour session the lease, + # not this limit, is the number to revisit. Beyond # that the lease expires mid-job; with `replicas: 1` (hardcoded in the # Deployment template) and the worker processing one job at a time, # nothing else can claim it, so the only consequence today is that a diff --git a/src/sturnus/infrastructure/whisper.py b/src/sturnus/infrastructure/whisper.py index 75568fb..0dd20a2 100644 --- a/src/sturnus/infrastructure/whisper.py +++ b/src/sturnus/infrastructure/whisper.py @@ -19,13 +19,44 @@ every transcript this project produced came back empty or hallucinated. `sturnus.infrastructure.speech_gate` does the same job with a stateless amplitude test; its module docstring carries the full reasoning. + +What the gate finds is then concatenated here and handed to the model as one +short array, with `clip_timestamps` re-expressed on that timeline and the +segment times mapped back onto the recording's afterwards. Handing over the +padded file and only marking the speech with `clip_timestamps` — which is what +this did until now — makes faster-whisper run its log-mel extraction over the +padding as well, because the array is shrunk only in the `vad_filter` branch +that setting clips skips: 4.94 GB of frames for a 100-minute track against +1.99 GB for the 41 minutes of speech in it, and the worker's memory limit had +to be raised from 6Gi to 12Gi in production because of it. The clip boundaries +stay, on the new timeline, because they are what keeps an encoder window from +spanning two utterances spoken minutes apart; `_on_the_original_timeline` +below carries the rest of that reasoning. + +This adapter is not a pure consumer of documented API and should not be read +as one. `clip_timestamps` is a documented argument, but what it is used *for* +here — and how a returned segment is put back on the recording's timeline — +rests on three internals of `generate_segments`: that `segment_size = +min(nb_max_frames, content_frames - seek, seek_clip_end - seek)` keeps an +encoder window inside one clip, that `Segment.seek` is that window's first +log-mel frame, and that clip boundaries reach the seek loop as `round(ts * +frames_per_second)`. None of the three is documented, the loop around them +carries faster-whisper's own note that it should be rewritten as a nested +loop, and `pyproject.toml` pins `faster-whisper>=1.1` with no upper bound +while Renovate merges on a green build. +`tests/infrastructure/test_whisper.py::test_a_window_never_spans_two_clips` +drives the real `generate_segments` — no weights, no download — and fails if +any of the three stops holding. It is the only upper bound there is. """ from __future__ import annotations import asyncio +from bisect import bisect_right +from itertools import accumulate from pathlib import Path +import numpy as np from faster_whisper import WhisperModel # type: ignore[import-untyped] from faster_whisper.audio import decode_audio # type: ignore[import-untyped] @@ -38,6 +69,87 @@ _SAMPLE_RATE = 16_000 +def _on_the_frame_grid(times: list[float], frames_per_second: int) -> list[int]: + """Seconds, as the log-mel frame numbers faster-whisper's seek loop counts in. + + `generate_segments` converts `clip_timestamps` with `seek_points = + [round(ts * self.frames_per_second) for ts in options.clip_timestamps]` + and every seek position it reports afterwards is a number on that grid. + Attributing a segment to a clip is therefore a comparison of frames, and + this is the one place the conversion is written down. Doing it in seconds + instead would put a window that begins exactly on a boundary that rounded + *up* into the clip before it, and being one clip out is being one removed + silence out — minutes, not milliseconds. + """ + return [round(time * frames_per_second) for time in times] + + +def _on_the_original_timeline( + start: float, + end: float, + seek: int, + bounds: list[tuple[int, int]], + clip_starts_in_frames: list[int], + offsets: list[float], +) -> tuple[float, float]: + """One segment's times, moved from the concatenated speech back to the file. + + The clip is decided by **where the segment was decoded from, not by the + times it reports**, and that is the whole of the difference between a + working restore and the one faster-whisper ships. `seek` is the segment's + own `Segment.seek`: the first log-mel frame of the encoder window it came + out of (`yield Segment(seek=previous_seek, ...)`, with `previous_seek` + set to the window's start immediately before decoding). The seek loop + enters a clip at `seek = seek_clips[clip_idx][0]` and leaves it the + moment `seek >= seek_clip_end`, so that frame names one clip and only + one, whatever the decoder then says about the audio in it. + + Reading it from the times cannot be made safe. A decoded `end` is + `time_offset + end_timestamp_position * time_precision` and + `_split_segments_by_timestamps` caps neither it nor the seek positions it + derives at the window's content, so a tail segment routinely claims a few + tenths of a second the window never held. Resolve the clip from that + `end` — or from a midpoint, when the overrun is more than half the + segment's own length — and the *next* clip's offset is added, moving the + text across the entire removed silence. Measured on the real engine with + this branch's own pre-fix code, on the shape the recording that started + this has — 100 minutes, 187 clips, 41.2 minutes of speech: 57 of 462 + segments reported an `end` past their clip, and for 5 of them the overrun + was more than half the segment, so a midpoint landed in the next clip. + Those 5 came back 20 to 31 seconds from where they were spoken, one of + them carrying 14 seconds of text. 20 to 31 seconds because that is the + silence removed between two clips *on this shape*; the error is always + exactly one gap, so on a session with three utterances forty minutes + apart it is forty minutes. That is the same silently wrong timestamp + `restore_speech_timestamps` produces, which is the reason this project + does the restore itself — and moving the arithmetic into our own file is + not what made it wrong there. + + The guarantee this now gives: **a restored segment always lies inside the + file-timeline extent of the clip whose encoder window produced it.** + `seek` is what picks the clip; the clamp is what holds the times to it + whatever the decoder claimed, at a cost of an overrun's worth off a + segment that only ever feeds a global sort, a 15-second merge window and + an %H:%M:%S render. It rests on three properties of `generate_segments`, + none of them documented and all of them observed on the real library by + `tests/infrastructure/test_whisper.py::test_a_window_never_spans_two_clips`: + that `Segment.seek` is the window's first frame, that the loop never + decodes a window from outside the clip it is walking, and that clip + boundaries reach it as `round(ts * frames_per_second)`. + """ + # `bisect_right` on the clip *starts*: the clips are adjacent on the + # concatenated timeline, so the starts partition it and the window's + # frame falls in exactly one of them. `max(..., 0)` is for a `seek` below + # the first boundary, which the loop cannot produce and which would + # otherwise index the list from the wrong end. + index = max(bisect_right(clip_starts_in_frames, seek) - 1, 0) + first, last = bounds[index] + low, high = first / _SAMPLE_RATE, last / _SAMPLE_RATE + restored_start = min(max(start + offsets[index], low), high) + restored_end = min(max(end + offsets[index], restored_start), high) + return restored_start, restored_end + + class WhisperEngine: def __init__( self, @@ -75,10 +187,84 @@ def _transcribe( # entire recording of nothing but padding through the decoder — # slow, and the single most reliable way to make Whisper invent # text. A silent participant must produce no segments at all. + # + # It guards a second thing since the speech is concatenated here: + # `np.concatenate([])` raises `ValueError`, and a zero-length + # array is an input `FeatureExtractor` has never been exercised + # against. This one line is what stands between an all-padding + # track and both failures. return TranscriptionResult(segments=(), language=self._default_language) + # Seconds are what `speech_clips` speaks in and samples are the only + # unit in which the arithmetic below is exact, so the conversion + # happens once, here. + bounds = [(round(start * _SAMPLE_RATE), round(end * _SAMPLE_RATE)) for start, end in clips] + # `speech_clips` promises clips inside the array it was handed, + # ascending, non-overlapping, and none shorter than about 0.37 s + # (`_MIN_RUN_FRAMES` plus a quarter-second hangover at each end). + # Every line below is built on those promises and none of the + # failures would be visible in a transcript: a clip past the end of + # the array slices short, so every offset after it is wrong by the + # difference; out-of-order clips make `offsets` decrease, so a + # speaker's lines come back shuffled and the protocol interleaves two + # speakers wrongly; and a clip that does not end strictly after it + # starts contributes a length of zero or less to the cumulative sum + # below, which is what the concatenated timeline *is* — the + # boundaries stop increasing, so `clip_timestamps` describes a clip + # the seek loop can never enter and `clip_starts_in_frames` stops + # being sorted, at which point the `bisect_right` that attributes a + # segment to a clip answers by accident. Asserted rather than + # repaired, and asserted here rather than + # trusted, because the constants the promises rest on live in another + # module and this is where breaking them would surface. + assert bounds[0][0] >= 0 and bounds[-1][1] <= audio.shape[0], ( + "speech_clips returned a clip outside the audio it was given" + ) + assert all(first < last for first, last in bounds), ( + "speech_clips returned a clip that does not end after it starts" + ) + assert all( + before[1] <= after[0] for before, after in zip(bounds, bounds[1:], strict=False) + ), "speech_clips returned overlapping or out-of-order clips" + + # One allocation for the whole of the speech. Deliberately not + # `faster_whisper.vad.collect_chunks`, which concatenates inside its + # own loop and is therefore quadratic in the clip count: 1.71 s + # against 0.01 s for a bit-identical array on the 100-minute + # recording, and worse again on a four-hour session. + speech = np.concatenate([audio[first:last] for first, last in bounds]) + # The padded array is dead the moment those slices are copied, and + # dropping it *here* rather than letting the frame hold it is worth + # 271 MB of peak: `WhisperModel.transcribe` allocates the log-mel + # features while this frame is still alive, so anything still + # referenced adds to the peak instead of overlapping with it. + del audio + + # Where each clip lands once the silence between them is gone. + # `concat_ends` doubles as the boundary list `clip_timestamps` wants: + # the clips are adjacent on this timeline, so the flat + # [start, end, start, end, ...] form is just the cumulative + # boundaries repeated. + lengths = [last - first for first, last in bounds] + counted = list(accumulate(lengths)) + concat_ends = [count / _SAMPLE_RATE for count in counted] + concat_starts = [0.0, *concat_ends[:-1]] + # Silence removed before each clip, in seconds — a difference of two + # integer sample counts, so it is exact, and adding it to a segment + # decoded from that clip is the whole of the restore. + offsets = [ + first / _SAMPLE_RATE - concat_start + for (first, _), concat_start in zip(bounds, concat_starts, strict=True) + ] + # Which clip a returned segment belongs to is read off the seek + # position it reports, so the seek loop's own frame grid has to be + # reproduced here (`_on_the_frame_grid`). Asked for before the call + # rather than while collecting: a library that stopped publishing + # `frames_per_second` should fail now, not after a 100-minute decode. + clip_starts_in_frames = _on_the_frame_grid(concat_starts, self._model.frames_per_second) + segments, info = self._model.transcribe( - audio, + speech, language=language, # Biases the decoder towards the vocabulary and the style of # this text. It is the only lever Sturnus has on proper nouns, @@ -93,7 +279,19 @@ def _transcribe( # A flat list of seconds — [start0, end0, start1, end1, ...] — not # a list of pairs and not the dict form, which belongs to # `BatchedInferencePipeline.transcribe`, a different API. - clip_timestamps=[value for clip in clips for value in clip], + # + # Still passed, now that the array is already only speech, for + # the one thing it does that concatenating cannot: `segment_size + # = min(nb_max_frames, content_frames - seek, seek_clip_end - + # seek)` caps an encoder window at the clip it began in, so the + # decoder is never shown two utterances at once and cannot emit a + # segment spanning the join between them. Dropping it — plain + # concatenation, which is what the library's own `vad_filter` + # path does — was measured emitting a single 258-second segment + # covering four utterances spoken minutes apart. + clip_timestamps=[ + value for pair in zip(concat_starts, concat_ends, strict=True) for value in pair + ], # Redundant on paper: faster-whisper's guard is # `if vad_filter and clip_timestamps == "0"`, so setting the clips # already keeps Silero from ever being loaded. Stated anyway, so @@ -113,12 +311,16 @@ def _transcribe( # One hallucinated segment then becomes the context every # following segment is decoded against, and the cascade the # two thresholds above exist to catch is exactly what that - # produces. `vad_filter` makes the default worse here rather - # than better: it cuts one speaker's track into fragments with - # every silence removed, so the "previous text" is routinely - # from minutes earlier and about something else entirely -- + # produces. Cutting the silence out makes the default worse + # here rather than better, whichever way it is cut: one + # speaker's track becomes fragments minutes apart, so the + # "previous text" is routinely about something else entirely -- # per-speaker recordings of a conversation are the case this - # default is least suited to. + # default is least suited to. (faster-whisper also resets the + # prompt at every window boundary of its own accord, so under + # the clip boundaries above this is belt and braces rather than + # the load-bearing part. It stays because it is a decision, and + # because the boundaries are not what it depends on.) condition_on_previous_text=False, # Above the library's default of 5. Beam search cost is # roughly linear in the width and this deployment transcribes @@ -128,18 +330,30 @@ def _transcribe( # document people read instead of having been in the room. beam_size=8, ) - # No offset arithmetic here, deliberately. Unlike the `vad_filter` - # path, which concatenates the kept audio and repairs the timestamps - # afterwards, the `clip_timestamps` path runs the feature extractor - # over the whole array and makes the seek loop jump between clips, so - # `time_offset = seek * time_per_frame` is already on the original - # timeline. These offsets stay file-relative in exactly the sense - # `sturnus.application.transcription.to_absolute` assumes. - collected = tuple( - TranscribedSegment(start=s.start, end=s.end, text=s.text) for s in segments - ) - # With `clip_timestamps` set, detection starts at the first clip - # instead of at second 0, so a speaker whose file opens with twenty - # minutes of padding no longer has their language guessed from it. + # The model was handed concatenated speech, so every time it + # reports is a position in *that* array: an utterance spoken at 02:30 + # comes back at about five seconds. `sturnus.application. + # transcription.to_absolute` adds the speaker's epoch to whatever is + # here and `domain.transcript` sorts all speakers' segments together + # on the result, so leaving these alone would not make a + # slightly-wrong document — it would stack the whole meeting into its + # first seconds and interleave two speakers into nonsense. The + # restored times are file-relative in exactly the sense `to_absolute` + # assumes, to within the 10 ms of frame-grid rounding + # `_on_the_original_timeline` describes. + collected = [] + for segment in segments: + start, end = _on_the_original_timeline( + segment.start, + segment.end, + segment.seek, + bounds, + clip_starts_in_frames, + offsets, + ) + collected.append(TranscribedSegment(start=start, end=end, text=segment.text)) + # The model never sees the padding at all now, so a speaker whose + # file opens with twenty minutes of it no longer has their language + # guessed from silence. detected = getattr(info, "language", None) or self._default_language - return TranscriptionResult(segments=collected, language=detected) + return TranscriptionResult(segments=tuple(collected), language=detected) diff --git a/tests/infrastructure/test_whisper.py b/tests/infrastructure/test_whisper.py index a93b786..49870bf 100644 --- a/tests/infrastructure/test_whisper.py +++ b/tests/infrastructure/test_whisper.py @@ -1,14 +1,32 @@ +import logging import wave +from collections.abc import Callable +from dataclasses import fields from pathlib import Path +from types import SimpleNamespace from typing import Any import numpy as np import pytest +from faster_whisper import WhisperModel # type: ignore[import-untyped] +from faster_whisper.feature_extractor import ( # type: ignore[import-untyped] + FeatureExtractor, +) +from faster_whisper.transcribe import ( # type: ignore[import-untyped] + TranscriptionOptions, +) -from sturnus.infrastructure.whisper import WhisperEngine +from sturnus.infrastructure.whisper import WhisperEngine, _on_the_frame_grid FIXTURE = Path(__file__).parent.parent / "fixtures" / "hello.wav" +#: faster-whisper counts seek positions in log-mel frames, 16 kHz over a +#: 160-sample hop. The fakes below report seeks on that grid because a real +#: `Segment` does; `test_a_window_never_spans_two_clips` checks this number +#: against the library's own `frames_per_second` rather than leaving two +#: hundreds to agree by luck. +_FRAMES_PER_SECOND = 100 + @pytest.fixture(scope="module") def engine() -> WhisperEngine: @@ -61,14 +79,39 @@ async def test_silence_yields_no_segments(engine: WhisperEngine, tmp_path: Path) class _FakeSegment: - """The three attributes `_transcribe` reads off a faster-whisper segment.""" + """The four attributes `_transcribe` reads off a faster-whisper segment. + + `seek` is the load-bearing one and the reason this fake is not just three + floats: it is the first log-mel frame of the encoder window the segment + was decoded from, and it is what says which clip the segment belongs to. + A fake that carried only times could not tell an engine that reads it + apart from one that guesses the clip from the times -- which is exactly + the defect `test_a_segment_that_overruns_its_clip_does_not_jump_the_gap` + exists for. + """ - def __init__(self, start: float, end: float, text: str) -> None: + def __init__(self, seek: int, start: float, end: float, text: str) -> None: + self.seek = seek self.start = start self.end = end self.text = text +def _as_decoded_from(window_start: float, start: float, end: float, text: str) -> _FakeSegment: + """A segment reported from the encoder window beginning at `window_start`. + + faster-whisper yields `Segment(seek=previous_seek, ...)` with + `previous_seek` the frame the window started at, so every fake segment + here has to name its window in the frames the library counts in. Writing + it as "which window did this come out of" rather than as a raw number is + what keeps these tests readable once the point is that the times a + segment reports are *not* what decides where it lands. + """ + return _FakeSegment( + seek=round(window_start * _FRAMES_PER_SECOND), start=start, end=end, text=text + ) + + class _FakeInfo: def __init__(self, language: str | None) -> None: self.language = language @@ -85,18 +128,38 @@ class _RecordingModel: """ def __init__( - self, segments: tuple[_FakeSegment, ...] = (), language: str | None = "de" + self, + segments: tuple[_FakeSegment, ...] = (), + language: str | None = "de", + segments_from: Callable[[list[float]], tuple[_FakeSegment, ...]] | None = None, ) -> None: self.segments = segments self.language = language + #: Builds the segments from the `clip_timestamps` the fake was handed, + #: for tests about *where* a segment ends up. A real model can only + #: speak in the coordinates it was given, and since the engine now + #: hands over concatenated speech, those coordinates are a timeline + #: the test cannot know until the engine has computed it. Hardcoding + #: a guess at it would be a test that pins the arithmetic against + #: itself; asking for the middle of whatever clip the engine declared + #: pins it against the audio the test actually built. + self.segments_from = segments_from + #: What `WhisperModel` publishes as the resolution of the seek grid. + #: `_transcribe` reads it to put its own clip boundaries on that grid, + #: so a fake that omitted it would not be exercising the conversion + #: at all. + self.frames_per_second = _FRAMES_PER_SECOND self.calls: list[dict[str, Any]] = [] def transcribe(self, audio: Any, **kwargs: Any) -> tuple[Any, _FakeInfo]: self.calls.append({"audio": audio, **kwargs}) + segments = self.segments + if self.segments_from is not None: + segments = self.segments_from(kwargs["clip_timestamps"]) # faster-whisper returns a generator, so a caller that forgot to # consume it would see no segments; returning an iterator keeps the # fake honest about that. - return iter(self.segments), _FakeInfo(self.language) + return iter(segments), _FakeInfo(self.language) def _engine_with(model: _RecordingModel, default_language: str = "de") -> WhisperEngine: @@ -130,6 +193,37 @@ def _tone(seconds: float, amplitude: float = 0.2) -> np.ndarray: return (amplitude * np.sin(2.0 * np.pi * 220.0 * t)).astype(np.float32) +def _track_with_speech_at(*spoken_at: float, total: float) -> np.ndarray: + """A padded track holding a two-second tone starting at each of `spoken_at`. + + The gaps are bit-exact zero, which is what `SpeakerWriter` writes between + Discord packets and the only thing the gate is aimed at. Building the + track from times the test names is what makes the timestamp assertions + below independent: every second in them comes from this function, not + from the code under test. + """ + track = np.zeros(round(total * 16_000), dtype=np.float32) + for start in spoken_at: + tone = _tone(2.0) + first = round(start * 16_000) + track[first : first + tone.shape[0]] = tone + return track + + +def _holds_a_silent_second(audio: np.ndarray) -> bool: + """Whether a whole second of `audio` is bit-exact zero, as padding is. + + Checked instead of trusting a length, because it is the property that + matters: the feature extractor allocates log-mel frames for whatever it + is handed, and a second of digital zero in there is a second of frames + computed for audio that was already known to be silence. + """ + if audio.shape[0] < 16_000: + return False + silent = np.cumsum(np.concatenate(([0], (audio == 0.0).astype(np.int64)))) + return bool((silent[16_000:] - silent[:-16_000] == 16_000).any()) + + async def test_a_file_with_only_padding_never_reaches_the_model(tmp_path: Path) -> None: """The short circuit, which is the single most important line in the change. @@ -153,16 +247,25 @@ async def test_a_file_with_only_padding_never_reaches_the_model(tmp_path: Path) assert result.language == "de" -async def test_the_model_is_given_the_whole_array_and_the_clips_the_gate_found( +async def test_the_model_is_handed_the_speech_and_not_the_padded_track( tmp_path: Path, ) -> None: - """Decode once, measure that array, and hand the model the same array. - - Passing the path instead would decode a 100-minute file twice and, worse, - would let the clips be computed on a different copy than the one the - model seeks through. The array must also be the *whole* file, not the - speech spliced together: that is what makes the returned offsets absolute - on the original timeline, so `to_absolute` keeps working untouched. + """The array reaching the model is the concatenated speech, nothing else. + + 0.5.0 handed over the whole decoded file and marked the speech with + `clip_timestamps`. That fixed the transcripts and cost the memory: + faster-whisper shrinks the array only in its `vad_filter` branch + (`if vad_filter and clip_timestamps == "0"`, then + `audio = np.concatenate(audio_chunks)`), and setting clips is precisely + what skips that branch, so `FeatureExtractor` ran over the padding too -- + 4.94 GB of log-mel frames for a 100-minute track, against 1.99 GB for the + 41 minutes of speech in it. The worker's limit had to go from 6Gi to 12Gi + in production because of this line. + + Asserted three ways, because handing the whole array over again is + invisible in a transcript: the array is far shorter than the file, it + holds none of the padding that was cut out, and its length is exactly the + duration `clip_timestamps` now describes. """ recording = tmp_path / "speech.wav" _write_wav( @@ -184,14 +287,200 @@ async def test_the_model_is_given_the_whole_array_and_the_clips_the_gate_found( call = model.calls[0] audio = call["audio"] assert isinstance(audio, np.ndarray) - assert audio.shape[0] == pytest.approx(16_000 * 10, abs=16_000 * 0.05) + # Two of the file's ten seconds are the tone, and the gate widens a clip + # by a fraction of a second at each end. Ten seconds' worth of samples + # here is the whole file again. + assert 16_000 * 2 <= audio.shape[0] <= 16_000 * 4 + assert not _holds_a_silent_second(audio) clips = call["clip_timestamps"] + # Still the flat [start, end, ...] list of seconds faster-whisper wants, + # but on the concatenated timeline: it begins at 0.0 and ends at the + # length of the array handed over. Leaving it in original-file seconds + # would point the seek loop past the end of a two-and-a-half-second + # array, and every clip after the first would decode nothing. assert [type(value) for value in clips] == [float, float] - start, end = clips - assert start <= 2.0 - assert end >= 4.0 - assert 0.0 <= start < end <= audio.shape[0] / 16_000 + assert clips[0] == 0.0 + assert clips[1] == pytest.approx(audio.shape[0] / 16_000, abs=1e-9) + + +async def test_the_clip_list_is_the_concatenated_timelines_own_boundaries( + tmp_path: Path, +) -> None: + """Three utterances become three adjacent clips, and the joins are marked. + + Marking them is what keeps this change safe. `segment_size = min( + nb_max_frames, content_frames - seek, seek_clip_end - seek)` caps an + encoder window at the clip it started in, so no window is ever shown + audio from two utterances at once and no decoded segment can span a join. + Dropping the boundaries -- passing plain concatenated audio, which is + what the library's own `vad_filter` path does -- was measured emitting + one 258-second segment covering four utterances spoken minutes apart. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(10.0, 60.0, 150.0, total=200.0)) + model = _RecordingModel() + engine = _engine_with(model) + + await engine.transcribe(recording, language="de", initial_prompt=None) + + call = model.calls[0] + clips = call["clip_timestamps"] + assert len(clips) == 6 + starts, ends = clips[0::2], clips[1::2] + assert starts[0] == 0.0 + # Adjacent, exactly: clip n+1 begins where clip n ends, because in the + # array handed over there is nothing between them any more. + assert starts[1:] == ends[:-1] + assert ends[-1] == pytest.approx(call["audio"].shape[0] / 16_000, abs=1e-9) + # Each clip is the two-second tone plus a little hangover, and none of + # them carries the fifty- and ninety-second gaps that separated them. + assert all(2.0 <= end - start <= 3.0 for start, end in zip(starts, ends, strict=True)) + + +def _a_model_without_weights() -> Any: + """A real `WhisperModel` with everything the seek loop reads and no weights. + + The property the test below observes belongs to faster-whisper, not to + us, so a fake model cannot observe it -- a fake breaks the way its author + imagined, and the failure that matters is the one nobody imagined. What + makes it observable without a HuggingFace download is that + `generate_segments` computes the whole window schedule itself and only + hands the *result* to the encoder: stub `encode`, `get_prompt` and + `generate_with_fallback` and the seek loop, the clip walk and + `segment_size` are all still the library's own code, running for real. + + Constructed through `object.__new__` because `WhisperModel.__init__` + loads CTranslate2 weights. The attributes below are the ones the loop + reads off `self`; if a release adds another, this raises `AttributeError` + here, which is the loud failure this whole test exists to produce. + """ + model = object.__new__(WhisperModel) + model.feature_extractor = FeatureExtractor() + model.frames_per_second = ( + model.feature_extractor.sampling_rate // model.feature_extractor.hop_length + ) + model.time_precision = 0.02 + model.input_stride = 2 + model.logger = logging.getLogger("sturnus.tests.faster_whisper") + model.encode = lambda *_args, **_kwargs: None + model.get_prompt = lambda *_args, **_kwargs: [] + # No tokens at all, so `_split_segments_by_timestamps` takes its + # no-timestamps branch and reports the window itself: `start` is the + # window's `time_offset` and `end` is `time_offset + segment_duration`. + # That is what makes each emitted segment a readable record of one + # window, which is the thing under observation. + model.generate_with_fallback = lambda *_args, **_kwargs: ( + SimpleNamespace(sequences_ids=[[]], no_speech_prob=0.0), + 0.0, + 0.0, + 1.0, + ) + return model + + +def _decoding_options(clip_timestamps: list[float]) -> Any: + """`TranscriptionOptions` with the fields the seek loop reads, and nothing else. + + Built from `dataclasses.fields` rather than by listing every argument, so + an option added upstream defaults to `None` here instead of breaking a + test that has no opinion about it. The names this test *does* have an + opinion about are checked against the real field list, so a rename is an + immediate failure rather than a silently ignored keyword. + """ + chosen: dict[str, Any] = { + "clip_timestamps": clip_timestamps, + "multilingual": False, + "without_timestamps": False, + "prefix": None, + "hotwords": None, + "initial_prompt": None, + "word_timestamps": False, + "condition_on_previous_text": False, + "prompt_reset_on_temperature": 0.5, + # `None` skips the no-speech fast-forward, which would otherwise + # decide what to do with a stubbed decoder's made-up probability. + "no_speech_threshold": None, + "log_prob_threshold": None, + } + names = {field.name for field in fields(TranscriptionOptions)} + assert not set(chosen) - names, "faster-whisper renamed a decoding option" + return TranscriptionOptions(**{name: chosen.get(name) for name in names}) + + +def test_a_window_never_spans_two_clips() -> None: + """The library behaviour this whole design rests on, observed on the library. + + `WhisperEngine` hands the model concatenated speech, so two utterances an + hour apart are adjacent samples in the array. The only thing keeping an + encoder window from covering both is `segment_size = min(nb_max_frames, + content_frames - seek, seek_clip_end - seek)` -- an undocumented internal + of `generate_segments`, in a loop that carries the library's own note + that it should be rewritten as a nested loop, in a dependency pinned + `faster-whisper>=1.1` with no upper bound and merged by Renovate on a + green build. Nothing else in this suite would notice it going away: the + clips would still be passed, the transcript would still be produced, and + the timestamps would be quietly wrong by minutes. This test is that + upper bound. + + It pins the three properties `_on_the_original_timeline` names, all + against windows the real loop scheduled: + + * a segment's `seek` **is** the first frame of its window, since `start` + is the window's `time_offset` and nothing else; + * every window lies inside one clip, and the windows of a clip tile it + exactly -- from its first frame to its last, with no gap and no + overlap; + * the frames a clip occupies are the ones `_on_the_frame_grid` computes, + which is what makes attributing by `seek` land in the right clip. The + clip boundaries here are deliberately off the 10 ms grid (2.507 s + rounds *up* to frame 251, 5.117 s to 512), so a library that truncated + where it now rounds would start its windows one frame early and the + tiling would not close. + + A clip longer than the 30 s window is included so both terms of the + `min` are exercised; `widest` asserts it really was split. + """ + model = _a_model_without_weights() + assert model.frames_per_second == _FRAMES_PER_SECOND + + clip_timestamps = [0.0, 2.507, 2.507, 5.117, 5.117, 41.048] + # 80 mel bins by `content_frames + 1`, which is how `generate_segments` + # reads a length. 45 seconds of them, comfortably past the last clip, so + # `content_frames - seek` never binds and the clip cap is the term under + # observation. The values are never looked at: `encode` is stubbed. + features = np.zeros((80, 4_500 + 1), dtype=np.float32) + tokenizer = SimpleNamespace(timestamp_begin=50_364, decode=lambda *_args: " ja") + + decoded = list( + model.generate_segments(features, tokenizer, _decoding_options(clip_timestamps), False) + ) + + per_frame = model.feature_extractor.time_per_frame + windows = [] + for segment in decoded: + assert segment.start == pytest.approx(segment.seek * per_frame, abs=1e-9) + windows.append((segment.seek, round(segment.end / per_frame))) + + boundaries = _on_the_frame_grid(clip_timestamps, model.frames_per_second) + covered, widest = 0, 0 + for clip_start, clip_end in zip(boundaries[0::2], boundaries[1::2], strict=True): + inside = [window for window in windows if clip_start <= window[0] < clip_end] + assert inside, "the seek loop decoded nothing at all in a clip it was given" + cursor = clip_start + for window_start, window_end in inside: + assert window_start == cursor + assert window_end <= clip_end + assert window_end - window_start <= model.feature_extractor.nb_max_frames + cursor = window_end + assert cursor == clip_end + covered += len(inside) + widest = max(widest, len(inside)) + # No window belonged to no clip, and the 30 s cap was reached at least + # once -- without which the last clip would prove nothing the first two + # do not. + assert covered == len(windows) + assert widest > 1 async def test_silero_is_never_asked(tmp_path: Path) -> None: @@ -230,24 +519,345 @@ async def test_the_decoder_side_hallucination_guards_stay_set(tmp_path: Path) -> assert model.calls[0]["no_speech_threshold"] == 0.6 -async def test_offsets_are_returned_unchanged(tmp_path: Path) -> None: - """No offset arithmetic, deliberately. +def _gate_clips(recording: Path) -> tuple[tuple[float, float], ...]: + """What the gate finds in `recording`, as the engine itself would ask it. + + The clips are an *input* to the arithmetic under test, not something that + arithmetic defines, so recomputing an expected answer from them checks + the offset addition rather than restating it. Asking the gate here rather + than hardcoding 9.74 also keeps these tests out of the way of the branch + that is changing the gate's own constants. + """ + from faster_whisper.audio import decode_audio # type: ignore[import-untyped] + + from sturnus.infrastructure.speech_gate import speech_clips + + return speech_clips(decode_audio(str(recording), sampling_rate=16_000)) + + +def _assert_the_model_spoke_in_its_own_coordinates( + call: dict[str, Any], file_seconds: float +) -> None: + """The premise every timestamp assertion in this file rests on. + + Mapping a segment back is only meaningful if the model was speaking in + some other timeline to begin with. 0.5.0 handed over the whole padded + file, so its offsets were already file-relative and *every* "the times + are right" assertion below is satisfied by doing no arithmetic at all. + Each such test therefore states the premise instead of inheriting it from + the test that pins the array, which a future edit could delete on its own. + + A quarter of the file is a deliberately loose ceiling -- the tracks here + are under a twentieth speech -- because the point is to exclude the whole + file, not to re-pin what the gate found. + """ + audio = call["audio"] + assert audio.shape[0] < round(file_seconds * 16_000) // 4 + - Unlike the `vad_filter` path, which concatenates the kept audio and - repairs the offsets afterwards, the `clip_timestamps` path runs the - feature extractor over the whole array and makes the seek loop jump - between clips, so `time_offset = seek * time_per_frame` is already on the - original timeline. Any remapping added here would shift every timestamp - in the finished document by the length of the leading padding. +def _middle_of_each_clip(clips: list[float]) -> tuple[_FakeSegment, ...]: + """One segment in the middle half of every clip the model was handed. + + Each comes out of the window that opens the clip it belongs to, which is + what a real decoder reports for any clip shorter than the 30 s window -- + every clip in these tests. + """ + return tuple( + _as_decoded_from( + start, + start=start + (end - start) * 0.25, + end=start + (end - start) * 0.75, + text=f" utterance {index}", + ) + for index, (start, end) in enumerate(zip(clips[0::2], clips[1::2], strict=True)) + ) + + +async def test_segment_times_come_back_on_the_recordings_own_timeline( + tmp_path: Path, +) -> None: + """Three utterances minutes apart, each reported where it was spoken. + + This is the assertion the change stands on. The model is handed a few + seconds of concatenated speech, so the times it returns are positions in + *that* array -- the utterance spoken at 02:30 comes back at about five + seconds. `sturnus.application.transcription.to_absolute` adds the + speaker's epoch to whatever is here and `domain.transcript` sorts every + speaker's segments together on the result, so leaving them unmapped does + not make a slightly-wrong document: it stacks the whole meeting into its + first ten seconds and interleaves two speakers into nonsense. A test that + only checked the text would pass with every timestamp wrong. """ recording = tmp_path / "speech.wav" - _write_wav(recording, np.concatenate([np.zeros(16_000 * 5, dtype=np.float32), _tone(2.0)])) - model = _RecordingModel(segments=(_FakeSegment(5.25, 6.75, " hallo"),)) + _write_wav(recording, _track_with_speech_at(10.0, 60.0, 150.0, total=200.0)) + model = _RecordingModel(segments_from=_middle_of_each_clip) engine = _engine_with(model) result = await engine.transcribe(recording, language="de", initial_prompt=None) - assert [(s.start, s.end, s.text) for s in result.segments] == [(5.25, 6.75, " hallo")] + _assert_the_model_spoke_in_its_own_coordinates(model.calls[0], file_seconds=200.0) + assert [s.text for s in result.segments] == [ + " utterance 0", + " utterance 1", + " utterance 2", + ] + # Each segment covers the middle half of its clip, and each clip is one + # two-second tone plus hangover, so a correctly mapped segment lies + # inside the two seconds the test placed that tone in. + for segment, spoken_at in zip(result.segments, (10.0, 60.0, 150.0), strict=True): + assert spoken_at <= segment.start < segment.end <= spoken_at + 2.0 + # And the silence between them is still silence: the gaps in the result + # are the gaps in the recording, not the two-and-a-half seconds they + # shrank to in the array the model saw. + assert result.segments[1].start - result.segments[0].start == pytest.approx(50.0, abs=1.0) + assert result.segments[2].start - result.segments[1].start == pytest.approx(90.0, abs=1.0) + + +async def test_a_segment_filling_its_clip_round_trips_within_ten_milliseconds( + tmp_path: Path, +) -> None: + """The mapping is exact, not approximate, and the tolerance says so. + + The only error the design admits is faster-whisper rounding each clip + boundary onto its 10 ms frame grid (`seek_points = round(ts * + frames_per_second)`); our own arithmetic is a difference of two integer + sample counts and contributes nothing. Ten milliseconds is half of + Whisper's own 20 ms timestamp grid and two orders below the "roughly the + low hundreds of milliseconds" that cross-speaker interleave order needs, + so it is worth pinning that the bound is this and not "somewhere in the + right clip". + + The gate is asked for the clips separately here. They are an *input* to + the arithmetic under test, not a constant it defines -- the engine hands + the same array to the same function -- so recomputing the expected answer + from them tests the offset addition rather than restating it. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(30.0, total=60.0)) + ((clip_start, clip_end),) = _gate_clips(recording) + + # Fifty milliseconds in from each end of the one clip, expressed on the + # concatenated timeline the engine will hand over. + model = _RecordingModel( + segments_from=lambda clips: ( + _as_decoded_from(clips[0], clips[0] + 0.05, clips[1] - 0.05, " hallo"), + ) + ) + + result = await _engine_with(model).transcribe(recording, language="de", initial_prompt=None) + + _assert_the_model_spoke_in_its_own_coordinates(model.calls[0], file_seconds=60.0) + (segment,) = result.segments + assert segment.start == pytest.approx(clip_start + 0.05, abs=0.01) + assert segment.end == pytest.approx(clip_end - 0.05, abs=0.01) + + +async def test_a_segment_from_a_later_window_of_a_long_clip_stays_in_that_clip( + tmp_path: Path, +) -> None: + """A clip longer than one encoder window, which every other test here lacks. + + The clips in this file are all a couple of seconds long, so every window + in them begins exactly where its clip does -- and a mapping that looked + the seek position up as a clip boundary rather than searching for the + clip containing it would pass all of them. A speaker who talks for + three-quarters of a minute without a two-second pause is one clip and + several windows, and the second of those windows begins nowhere near a + boundary. + """ + recording = tmp_path / "speech.wav" + track = _track_with_speech_at(300.0, total=400.0) + speaking = _tone(45.0) + track[round(10.0 * 16_000) : round(10.0 * 16_000) + speaking.shape[0]] = speaking + _write_wav(recording, track) + (first_clip_start, first_clip_end), (last_clip_start, _) = _gate_clips(recording) + assert first_clip_end - first_clip_start > 30.0, "the fixture no longer needs two windows" + + # Half a second into the second 30-second window of the first clip. + model = _RecordingModel( + segments_from=lambda clips: ( + _as_decoded_from(clips[0] + 30.0, clips[0] + 30.5, clips[0] + 31.5, " later"), + ) + ) + + result = await _engine_with(model).transcribe(recording, "de", None) + + _assert_the_model_spoke_in_its_own_coordinates(model.calls[0], file_seconds=400.0) + (segment,) = result.segments + assert segment.start == pytest.approx(first_clip_start + 30.5, abs=0.02) + assert segment.end == pytest.approx(first_clip_start + 31.5, abs=0.02) + # Which is well before the second utterance, four minutes further on. + assert segment.end < last_clip_start + + +async def test_a_segment_that_overruns_its_clip_does_not_jump_the_gap( + tmp_path: Path, +) -> None: + """The near miss, which is the one way this change fails silently. + + A decoded segment can report an `end` well past the audio its window + actually held. `_split_segments_by_timestamps` builds it as `end_time = + time_offset + end_timestamp_position * time_precision` and caps neither + that nor the seek it derives at the window's content, so the tail segment + of a clip routinely runs a few tenths of a second into nothing. Resolve + the clip from the times -- from `end`, or from the midpoint of a segment + that overruns by more than half its own length -- and the *next* clip's + offset is added instead, so the text lands on the far side of all the + removed silence. Measured on the real engine, with this branch's own + pre-fix code, on the shape of the recording that started all this -- 100 + minutes, 187 clips, 41.2 minutes of speech: 57 of 462 segments reported + an `end` past their clip, and 5 of those overran by more than half the + segment, so their midpoints landed in the next clip. They came back 20 to + 31 seconds from where they were spoken, one of them carrying 14 seconds + of text. The error is always exactly one removed gap, so on a session + with three utterances forty minutes apart it is forty minutes. It is the + same silent-wrong-timestamp failure the library's own + `restore_speech_timestamps` produces, relocated into our code. + + So the clip is not resolved from the times at all: it is read off the + window the segment came out of. The overrun here is 0.7 s against a clip + of about 2.5 s, which puts the midpoint past the boundary -- the case a + rule based on the times cannot survive. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(10.0, 300.0, total=320.0)) + (first_clip_start, first_clip_end), _ = _gate_clips(recording) + # Out of the window that opens the first clip -- the only window a clip + # of two and a half seconds has -- but reported ending 0.7 s past that + # clip, so the midpoint sits 0.2 s beyond the boundary and the far side + # of the boundary is five minutes of removed silence. + model = _RecordingModel( + segments_from=lambda clips: ( + _as_decoded_from(clips[0], clips[1] - 0.3, clips[1] + 0.7, " word"), + ) + ) + + result = await _engine_with(model).transcribe(recording, language="de", initial_prompt=None) + + _assert_the_model_spoke_in_its_own_coordinates(model.calls[0], file_seconds=320.0) + (segment,) = result.segments + # Where the first utterance is, not where the second one is: the gap + # between them is 4.8 minutes, so a wrong clip is off by minutes and not + # by a tolerance anyone could widen this assertion to absorb. + assert first_clip_start <= segment.start < segment.end <= first_clip_end + assert segment.start == pytest.approx(first_clip_end - 0.3, abs=0.01) + # And it stops at the end of the clip it came from. Those 0.7 s are audio + # the decoder was never shown -- they are on the far side of the join, in + # silence that was cut out -- and a segment is a claim about audio that + # exists. + assert segment.end == pytest.approx(first_clip_end, abs=0.01) + + +async def test_a_segment_reported_before_its_clip_starts_is_still_inside_it( + tmp_path: Path, +) -> None: + """The other edge of the clip, and the other half of the clamp. + + A clip boundary reaches the seek loop as `round(ts * frames_per_second)`, + so a boundary that rounds *down* puts the clip's first frame up to 5 ms + before where this code says the clip begins -- and `start = time_offset + + start_timestamp_position * time_precision` is free to be exactly that + frame. Restored without a floor the segment claims audio from before its + clip, which on this timeline is silence that was cut out and on the + original one is the tail of a gap. The guarantee is that a segment lies + inside the clip it was decoded from; this is the edge that needs the + floor for it to hold. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(10.0, 300.0, total=320.0)) + _, (last_clip_start, last_clip_end) = _gate_clips(recording) + # The second one lies *entirely* before the clip, which is what a + # zero-length segment on a boundary that rounded down looks like. It is + # the only shape in which the restored `end` would come out below the + # restored `start`, and a segment whose end precedes its start is not + # something any consumer of `TranscribedSegment` is prepared for. + model = _RecordingModel( + segments_from=lambda clips: ( + _as_decoded_from(clips[2], clips[2] - 0.005, clips[2] + 0.5, " hallo"), + _as_decoded_from(clips[2], clips[2] - 0.005, clips[2] - 0.002, " hm"), + ) + ) + + result = await _engine_with(model).transcribe(recording, "de", None) + + _assert_the_model_spoke_in_its_own_coordinates(model.calls[0], file_seconds=320.0) + straddling, wholly_before = result.segments + assert last_clip_start <= straddling.start < straddling.end <= last_clip_end + assert straddling.end == pytest.approx(last_clip_start + 0.5, abs=0.01) + assert last_clip_start <= wholly_before.start <= wholly_before.end <= last_clip_end + + +async def test_a_segment_reported_at_the_very_end_lands_in_the_last_clip( + tmp_path: Path, +) -> None: + """A segment with no duration, sitting exactly on the final boundary. + + The decoder can emit one at the tail of a clip, and it is the shape that + breaks every rule based on the times a segment reports: `start`, `end` + and their midpoint all sit *on* a boundary, so a boundary search answers + "the clip after this one" -- which past the last clip is an `IndexError` + in a worker thread, failing a job that had already been transcribed. Read + off the window instead, it is simply the last clip, and the clamp keeps + the zero-length segment inside it. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(10.0, 300.0, total=320.0)) + _, (last_clip_start, last_clip_end) = _gate_clips(recording) + model = _RecordingModel( + segments_from=lambda clips: (_as_decoded_from(clips[-2], clips[-1], clips[-1], " ja"),) + ) + + result = await _engine_with(model).transcribe(recording, "de", None) + + (segment,) = result.segments + # In the second utterance's clip, five minutes into the recording, not + # in the first one and not at second zero of the concatenated array. + assert last_clip_start <= segment.start <= segment.end <= last_clip_end + + +async def test_a_clip_list_that_breaks_the_gates_promise_stops_the_transcription( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The three properties every line of the mapping is built on, asserted. + + `speech_clips` promises clips inside the array it was handed, ascending, + non-overlapping, and none shorter than about 0.37 s. All three are + load-bearing here and none of the failures is visible in a transcript: a + clip past the end of the array slices short, so every offset after it is + wrong by the difference; out-of-order clips make the offsets decrease, so + segments come back in the wrong order and the protocol interleaves two + speakers wrongly; a clip that does not end after it starts contributes a + length of zero or less to the cumulative sum that *is* the concatenated + timeline, so the boundaries stop increasing and both the clip list handed + to the library and the sorted list a segment is attributed against turn + to nonsense. The constants those promises rest on live in another module, + so the check belongs here, where the breakage would otherwise surface. + """ + recording = tmp_path / "speech.wav" + _write_wav(recording, _track_with_speech_at(10.0, 60.0, total=80.0)) + + def _clips(source: tuple[tuple[float, float], ...]) -> Callable[..., Any]: + return lambda _audio, **_kwargs: source + + monkeypatch.setattr( + "sturnus.infrastructure.whisper.speech_clips", _clips(((30.0, 32.0), (10.0, 12.0))) + ) + with pytest.raises(AssertionError): + await _engine_with(_RecordingModel()).transcribe(recording, "de", None) + + # Shorter than one sample at 16 kHz, so both bounds round to the same + # integer: a clip of no length at all, and a boundary pair with no audio + # behind it. + monkeypatch.setattr("sturnus.infrastructure.whisper.speech_clips", _clips(((10.0, 10.00001),))) + with pytest.raises(AssertionError): + await _engine_with(_RecordingModel()).transcribe(recording, "de", None) + + # Past the end of an eighty-second file. numpy slices short rather than + # raising, so this is the failure that would otherwise reach production + # as nothing worse-looking than timestamps drifting after the first clip. + monkeypatch.setattr("sturnus.infrastructure.whisper.speech_clips", _clips(((10.0, 120.0),))) + with pytest.raises(AssertionError): + await _engine_with(_RecordingModel()).transcribe(recording, "de", None) async def test_an_undetected_language_falls_back_to_the_default(tmp_path: Path) -> None: @@ -285,6 +895,9 @@ def __init__(self) -> None: #: What `info.language` reports; `None` stands for detection that #: came up empty, which is what the adapter's own default is for. self.detected: str | None = "de" + #: As on `WhisperModel`; `_transcribe` reads it before it calls the + #: model, so a spy without it never gets as far as recording anything. + self.frames_per_second = _FRAMES_PER_SECOND def __call__(self, model_size: str, **kwargs: Any) -> "_ModelSpy": self.construction = {"model_size": model_size, **kwargs} @@ -393,6 +1006,11 @@ async def test_the_hallucination_guards_are_still_in_place(spy: _ModelSpy) -> No # padding back in front of the decoder. clips = spy.transcription["clip_timestamps"] assert clips and len(clips) % 2 == 0 + # On the concatenated timeline, so the first clip starts at 0.0. What the + # gate found in the fixture belongs to the gate's own tests; that the + # list is expressed in the coordinates of the array actually handed over + # is this adapter's decision and is pinned here. + assert clips[0] == 0.0 assert spy.transcription["compression_ratio_threshold"] == 2.4 assert spy.transcription["no_speech_threshold"] == 0.6 From d954abfb855bc3b66bcbe009fdef3aaec312dbfe Mon Sep 17 00:00:00 2001 From: TheMeinerLP Date: Fri, 21 Aug 2026 10:54:37 +0200 Subject: [PATCH 2/2] style: wrap the line the merge left over the limit --- tests/infrastructure/test_whisper.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/infrastructure/test_whisper.py b/tests/infrastructure/test_whisper.py index 43e6aba..f7ff45a 100644 --- a/tests/infrastructure/test_whisper.py +++ b/tests/infrastructure/test_whisper.py @@ -1123,7 +1123,9 @@ async def test_a_track_that_produced_text_is_not_reported_as_a_loss( """ recording = tmp_path / "speech.wav" _write_wav(recording, np.concatenate([_tone(2.0), np.zeros(16_000, dtype=np.float32)])) - engine = _engine_with(_RecordingModel(segments=(_as_decoded_from(0.0, 0.0, 1.5, " Guten Morgen."),))) + engine = _engine_with( + _RecordingModel(segments=(_as_decoded_from(0.0, 0.0, 1.5, " Guten Morgen."),)) + ) with caplog.at_level(logging.WARNING, logger="sturnus.infrastructure.whisper"): result = await engine.transcribe(recording, language="de", initial_prompt=None)