Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 76 additions & 4 deletions murmurflow/dictate.py
Original file line number Diff line number Diff line change
Expand Up @@ -782,7 +782,16 @@ def start_partial_server(*, wait: float = 60.0) -> bool:
this existed. See :func:`whisper.partial_model`.
"""
model = whisper.partial_model()
return bool(model) and start_server(wait=wait, model=model, at=partial_port())
if not model:
# AND STOP THE ONE A PREVIOUS RUN LEFT, which is not tidiness. `partial_at` sends the
# partials to whatever is answering on the live port, so a small server still up from
# before `livePass` changed keeps on answering them — the setting would appear to do
# nothing until the machine was restarted, and 488 MB would sit there while it did.
# Only one that is up AND ours: this must never be a blind pkill on a port we do not hold.
if ours(partial_port()) and server_up(partial_port()):
stop_server(partial_port())
return False
return start_server(wait=wait, model=model, at=partial_port())


def stop_server(at: int = 0) -> int:
Expand Down Expand Up @@ -1629,6 +1638,12 @@ class Stream:
blank: int = 0
#: Chunks actually pasted at the cursor.
typed: int = 0
#: The last pass that cleared every gate, and the seconds of audio it had read — what lets
#: `finish` skip its own transcription (see :func:`whole_clip_read`). ONE attribute holding
#: both, because they are read from another thread while this one is still writing them: as
#: two fields a reader could take the new length beside the older transcript, decide the clip
#: was fully read and drop whatever was said in between. One store, one read, no window.
read: tuple[Heard, float] | None = None


#: In-flight streams, keyed by the wav they are transcribing. A dict and not an attribute on
Expand Down Expand Up @@ -1989,6 +2004,10 @@ def _stream_loop(rec: Recording, stream: Stream) -> None:
while not stream.done.is_set():
started = time.monotonic()
stream.passes += 1
# BEFORE the pass, and off the file SIZE: every byte past the header is a sample, and the
# header of a file ffmpeg is still writing says the clip is empty. Under-reporting here is
# safe — it only makes `whole_clip_read` refuse — where over-reporting would drop speech.
covered = seconds_on_disk(rec.wav)
found = _partial(rec.wav, snapshot, pinned)
heard = found.text
# A pass that read nothing — still below the quiet floor, a hallucination thrown out, the
Expand All @@ -2001,6 +2020,7 @@ def _stream_loop(rec: Recording, stream: Stream) -> None:
continue
if not pinned and found.language and (not spoken or found.language in spoken):
pinned = found.language
stream.read = (found, covered)
settled = stable_prefix(previous, heard)
previous = heard
chunk = stream_tail(stream.text, settled) if settled else ""
Expand All @@ -2027,6 +2047,41 @@ def _stream_loop(rec: Recording, stream: Stream) -> None:
stream.done.wait(max(0.0, STREAM_EVERY_SECONDS - (time.monotonic() - started)))


def whole_clip_read(stream: Stream | None, wav: Path, captured: float) -> Heard | None:
"""The live pass's own transcript, when it already read the whole clip. Else ``None``.

**This is the 8 seconds at the end of a long dictation.** Measured from the operator's own log:
a 73.7s clip spent 3.2s in the final transcription, a 69.0s clip 6.0s, a 79.1s clip 7.8s — and
the row for each says ``→ streamed``, meaning that pass produced NOTHING that was not already
on screen. Worse than wasted: he had stopped, read his sentence, sent it, and the tail then
landed eight seconds later in whatever he was looking at by then ("it just added a lot of
gibberish ... after I already sent the message").

A partial only ever reads the audio captured SO FAR, so the question is what was said after the
last one. If that is silence, the last partial read the whole clip and its transcript is the
transcript — the same model, the same audio, the same gates. `covered` is measured from the
file's SIZE before the pass rather than after, so it under-reports and this refuses more often
than it strictly must, which is the right way for a shortcut to be wrong.

``AHEAD_SECONDS`` is not a tolerance for missing speech: below it there is not enough audio for
:func:`tail_dbfs` to have an opinion, and a quarter of a second cannot hold a word.
"""
if stream is None or stream.read is None:
return None
heard, covered = stream.read
if not heard.text:
return None
ahead = max(0.0, captured - covered)
if ahead > AHEAD_SECONDS and tail_dbfs(wav, ahead) >= quiet_floor():
return None # something was said after the last pass looked
return heard


#: How much unread audio at the end of a clip is too little to hold a word. See
#: :func:`whole_clip_read`.
AHEAD_SECONDS = 0.25


def stop_streaming(wav: Path) -> Stream | None:
"""Stop the stream for ``wav`` starting any new pass. Returns it, for :func:`streamed`.

Expand Down Expand Up @@ -2193,6 +2248,19 @@ def _peak_dbfs(frames: bytes) -> float:
return 20 * math.log10(min(peak, 32768) / 32768.0)


def seconds_on_disk(wav: Path) -> float:
"""How many seconds of audio a clip STILL BEING RECORDED holds. ``0.0`` if it cannot be read.

The same arithmetic as :func:`tail_dbfs` and for the same reason: while ffmpeg is appending,
the RIFF header still says the file is empty, so :func:`audio_seconds` answers 0 for a clip
that is minutes long.
"""
try:
return max(0.0, (wav.stat().st_size - 44) / BYTES_PER_SECOND)
except OSError:
return 0.0


def tail_dbfs(wav: Path, seconds: float) -> float:
"""Peak of the LAST ``seconds`` of a clip that is STILL BEING RECORDED. ``0.0`` = no opinion.

Expand Down Expand Up @@ -2382,7 +2450,10 @@ def retire() -> None:

started = time.monotonic()
try:
heard = transcribe(wav)
# The live pass may already have read this exact audio with this exact model. If nothing
# was said after it looked, transcribing again buys a copy of what is on screen for the
# seconds the operator is standing there waiting — see :func:`whole_clip_read`.
heard = whole_clip_read(stream, wav, captured) or transcribe(wav)
raw = heard.text
finally:
retire()
Expand Down Expand Up @@ -2955,8 +3026,9 @@ def emit(line: str) -> None:
)
else:
emit(
"the words arrive while you talk — in ~2s lumps, because the big model is answering "
"the live pass too. `murmurflow setup small` gets the fast one (~488 MB, ~5x quicker)"
f"the words arrive while you talk ({Path(whisper.model()).stem} on the live pass, so "
"they come in bigger lumps the longer you talk). `murmurflow config set livePass small`"
" is the fast one, with thinner punctuation"
)

#: Consecutive clips that took the cold path while a warm server was supposed to be answering.
Expand Down
19 changes: 19 additions & 0 deletions murmurflow/whisper.py
Original file line number Diff line number Diff line change
Expand Up @@ -127,7 +127,26 @@ def partial_model() -> str:
Deliberately ignores the ``model`` config override, which names the model that writes what you
keep. Pinning that to a small model is a choice about the transcript; it must not also silently
become the choice about the partials, and vice versa.

**The default is now the BIG model, i.e. no small model at all, and that is a measurement.**
Once a live pass could type PUNCTUATION (`dictate.missing_mark`), the model answering the live
pass stopped deciding only the words and started deciding the marks the operator keeps — and
the two models are not close there. Replayed through the whole stream loop on one real 38s
clip, against the big model's own whole-clip transcript:

live = small 13.9% different "...in the end like when I just stopped my control it
just added a lot of gibberish I'm not sure and that came"
live = big 11.4% different "...in the end, like when I just stopped my control, it
just added a lot of gibberish, I'm not sure. And that
came after a few seconds, after I already sent..."

The percentages understate it; the sentences are the finding. The operator's call, given both
("I would rather have it qualitatively high rather than faster"). ``livePass: "small"`` puts
the fast one back, and the cost of the default is real: a pass over a long clip costs seconds
rather than ~1, so the words arrive in bigger lumps the longer you talk.
"""
if str(config.load().get("livePass", "big")).strip().lower() != "small":
return ""
for name in PARTIAL_PREFERENCE:
for directory in _search_dirs():
candidate = directory / name
Expand Down
41 changes: 38 additions & 3 deletions tests/test_murmurflow.py
Original file line number Diff line number Diff line change
Expand Up @@ -1681,6 +1681,37 @@ def test_a_forgotten_key_still_gets_its_words(monkeypatch):
assert starts == [1] # the daemon started its server and then rescued the clip on its own


def test_the_end_is_not_transcribed_twice_when_the_live_pass_already_read_it(monkeypatch, tmp_path):
"""The 8 seconds at the end of a long dictation, and the tail that landed after he had sent.

From the operator's own log: a 73.7s clip spent 3.2s in the final transcription, 69.0s spent
6.0s, 79.1s spent 7.8s — and each of those rows says `→ streamed`, meaning that pass produced
nothing that was not already on screen. He had stopped, read his sentence and sent it; the
tail then arrived eight seconds later in whatever he was looking at by then.
"""
clip = tmp_path / "c.wav"
clip.write_bytes(b"\x00" * (44 + dictate.BYTES_PER_SECOND * 30))
live = dictate.Heard("what the live pass already read", 0.99, "en", warm=True)
stream = dictate.Stream(threading.Event())

# Nothing read yet: there is nothing to reuse.
assert dictate.whole_clip_read(stream, clip, 30.0) is None
assert dictate.whole_clip_read(None, clip, 30.0) is None

# It read 29.9 of the 30 seconds — too little left to hold a word.
stream.read = (live, 29.9)
assert dictate.whole_clip_read(stream, clip, 30.0) is live

# It read 20 of 30, and the ten seconds it never saw are SILENT: still the whole transcript.
stream.read = (live, 20.0)
monkeypatch.setattr(dictate, "tail_dbfs", lambda _wav, _seconds: -90.0)
assert dictate.whole_clip_read(stream, clip, 30.0) is live

# Same ten seconds, but somebody was talking in them. Now the shortcut must refuse.
monkeypatch.setattr(dictate, "tail_dbfs", lambda _wav, _seconds: -12.0)
assert dictate.whole_clip_read(stream, clip, 30.0) is None


def test_the_last_seconds_of_a_clip_still_being_recorded_can_be_read(tmp_path):
"""`wave` cannot answer this and that is the whole reason it exists.

Expand Down Expand Up @@ -2108,17 +2139,21 @@ def test_the_live_model_is_a_small_one_and_never_the_transcript_model(tmp_path,
models = config.home_root() / "models"
models.mkdir(parents=True, exist_ok=True)
(models / "ggml-large-v3-turbo.bin").write_bytes(b"x")
assert whisper.partial_model() == "" # a big model is not a live model
(models / "ggml-base.bin").write_bytes(b"x")
assert whisper.partial_model().endswith("ggml-base.bin")
(models / "ggml-small.bin").write_bytes(b"x")
# THE DEFAULT IS THE BIG MODEL, i.e. no live model at all — the live pass types the
# punctuation the operator keeps, and the two models are not close there.
assert whisper.partial_model() == ""
config.set_value("livePass", "small")
assert whisper.partial_model().endswith("ggml-small.bin") # small beats base: measured German
(models / "ggml-small.bin").unlink()
assert whisper.partial_model().endswith("ggml-base.bin") # base is the fallback below small
assert whisper.model().endswith("ggml-large-v3-turbo.bin") # and the transcript is unmoved

# An explicit `model` override still names the transcript model, and only that one.
config.set_value("model", str(models / "ggml-base.bin"))
assert whisper.model().endswith("ggml-base.bin")
assert whisper.partial_model().endswith("ggml-small.bin")
assert whisper.partial_model().endswith("ggml-base.bin")


def test_the_live_server_gets_its_own_model_and_its_own_port(monkeypatch):
Expand Down
Loading