Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
55 changes: 50 additions & 5 deletions murmurflow/dictate.py
Original file line number Diff line number Diff line change
Expand Up @@ -1790,21 +1790,63 @@ def stream_tail(pasted: str, final: str) -> str:
if not already:
return final
words = final.split()
keys = [_key(word) for word in words]
return " ".join(words[_reached(already, [_key(word) for word in words]) :])


def _reached(already: list[str], keys: list[str]) -> int:
"""The index in ``keys`` just past everything that is already on screen.

Split out of :func:`stream_tail` because :func:`missing_mark` asks the same question about the
same two sequences — where does the screen END inside this transcript — and two answers to that
would drift apart word by word.
"""
if keys[: len(already)] == already:
return " ".join(words[len(already) :])
return len(already)
matcher = difflib.SequenceMatcher(a=already, b=keys, autojunk=False)
matched = [block for block in matcher.get_matching_blocks() if block.size]
if not matched: # nothing corresponds: trust the count, lose nothing
return " ".join(words[len(already) :])
return len(already)
reached = matched[-1]
# Words on screen PAST the alignment are ones the final pass said differently. The tail that
# follows them is not new text, it is the same words again in the better model's wording, and
# typing it puts both on screen — reported as "it just adds another word at the end", which is
# exactly what one reworded last word looks like ("the design" + "designs"). One dropped for
# one left over: the rewording is skipped and anything genuinely beyond the screen still lands.
reworded = len(already) - (reached.a + reached.size)
return " ".join(words[reached.b + reached.size + reworded :])
return reached.b + reached.size + reworded


def missing_mark(pasted: str, settled: str) -> str:
"""The mark that belongs directly after ``pasted``, once a later word has confirmed it.

**This is the punctuation streaming used to lose, and it is the last of it.** A mark rides on
the word in front of it, and :func:`stable_prefix` will not type the mark on a word that is
still touching the end of the audio — a pause is how whisper decides a sentence ended, and it
takes that decision back the moment the speaker carries on. So the word lands bare, and
nothing could ever put the mark on afterwards: the word is already on screen and there is no
un-type. Reported as "there was a break before, but no punctuation".

A LATER pass answers the question the earlier one could not. If the word carrying the mark is
no longer at the end of the transcript — real speech follows it, and whisper still ends the
sentence there — the mark is a decision made WITH the following audio, which is the same test
every other word passes before it is typed. It goes on with no space in front of it, joined to
the word it belongs to.

Returns ``""`` unless a word after it has also settled, so this can never be the lone full stop
of :func:`end_mark`'s docstring: the mark is only ever typed in the same breath as the word
that proves it.
"""
already = [_key(word) for word in pasted.split()]
words = settled.split()
if not already or not words:
return ""
index = _reached(already, [_key(word) for word in words])
if index <= 0 or index >= len(words):
return "" # nothing before it, or nothing after it to confirm it
mark = _TRAILING_MARK.search(words[index - 1])
if not mark or pasted.rstrip().endswith(mark.group()):
return ""
return mark.group()


def _partial(live: Path, snapshot: Path, language: str = "") -> Heard:
Expand Down Expand Up @@ -1948,6 +1990,9 @@ def _stream_loop(rec: Recording, stream: Stream) -> None:
settled = stable_prefix(previous, heard)
previous = heard
chunk = stream_tail(stream.text, settled) if settled else ""
# The mark on the word already at the end of the screen, now that a later word has
# settled behind it. Only ever together with that word — see :func:`missing_mark`.
mark = missing_mark(stream.text, settled) if chunk else ""
if chunk:
with _INJECT_LOCK:
# Inside the lock, because `stop_streaming` sets this and then takes the lock: past
Expand All @@ -1961,7 +2006,7 @@ def _stream_loop(rec: Recording, stream: Stream) -> None:
# `stream.text` is literally what is on the screen, built from what LANDED rather
# than from what was asked for — see :func:`place`. The leading space travels with
# the chunk, so this is a concatenation and never a re-join.
landed = place(f" {chunk}" if stream.text else chunk)
landed = place(f"{mark} {chunk}" if stream.text else chunk)
if landed:
stream.text = f"{stream.text}{landed}".strip()
stream.typed += 1
Expand Down
84 changes: 66 additions & 18 deletions tests/test_murmurflow.py
Original file line number Diff line number Diff line change
Expand Up @@ -1695,6 +1695,57 @@ def test_the_microphone_closes_itself_when_the_second_tap_never_comes(monkeypatc
assert dictate.MAX_CLIP_SECONDS == 600


def _drive_stream(passes):
"""What lands on screen when the live pass reads ``passes``, one after another.

The same four calls `_stream_loop` makes, in the same order, so a sequence that broke a real
dictation can be replayed as a test. Kept beside the tests that use it rather than inside them:
four copies of the loop drift, and a copy that drifts stops testing the loop.
"""
screen = previous = ""
for heard in passes:
settled = dictate.stable_prefix(previous, heard)
previous = heard
chunk = dictate.stream_tail(screen, settled) if settled else ""
mark = dictate.missing_mark(screen, settled) if chunk else ""
if chunk:
screen = f"{screen}{mark} {chunk}".strip() if screen else chunk
return screen


def test_the_mark_lands_once_a_later_word_confirms_it():
"""Reported as "there was a break before, but no punctuation".

A mark rides on the word in front of it, and that word is never typed with its mark while it
still touches the end of the audio — a pause is how whisper decides a sentence ended, and it
takes that back the moment the speaker carries on. So the word landed bare and nothing could
put the mark on afterwards. A later pass answers it: real speech follows and whisper STILL
ends the sentence there, so the mark goes on, joined to the word it belongs to.
"""
reference = "I did not say the three times. I did, however, say really three times."
screen = _drive_stream(
[
"I did not say the three times",
"I did not say the three times", # the break
"I did not say the three times.", # whisper ends the sentence
"I did not say the three times.",
"I did not say the three times. I", # he carries on
"I did not say the three times. I did however say",
"I did not say the three times. I did, however, say really",
reference,
reference,
]
)
# ...plus the one thing only the key release can know: the mark that ends the clip.
landed = screen + dictate.end_mark(screen, reference)
assert landed == reference # streamed, and identical to the whole-clip transcript
# And the fixes it must not undo, driven through the same loop.
assert _drive_stream(["Could you please work on my", "Could you please work on my..."] * 2) == (
"Could you please work on my"
)
assert "The The" not in _drive_stream(["Well, yeah. The The The"] * 3)


def test_a_pause_never_types_a_lone_full_stop_where_the_next_word_goes():
"""Reported as "it puts a period instead of the word" after a short break.

Expand All @@ -1704,24 +1755,21 @@ def test_a_pause_never_types_a_lone_full_stop_where_the_next_word_goes():
screen with no letters in it, so the next alignment read it as something the final pass had
reworded and dropped a real word to pay for it. The word this ate, in the report, was "But".
"""
screen, previous = "", ""
for heard in (
"and then I ran the command",
"and then I ran the command", # the pause: the transcript stops growing
"and then I ran the command.", # whisper decides the sentence ended
"and then I ran the command.",
"and then I ran the command. But", # he speaks again
"and then I ran the command. But when I say",
"and then I ran the command. But when I say",
):
settled = dictate.stable_prefix(previous, heard)
previous = heard
chunk = dictate.stream_tail(screen, settled) if settled else ""
if chunk:
screen = f"{screen} {chunk}".strip() if screen else chunk
assert " ." not in screen
assert "But" in screen
assert screen == "and then I ran the command But when I say"
screen = _drive_stream(
[
"and then I ran the command",
"and then I ran the command", # the pause: the transcript stops growing
"and then I ran the command.", # whisper decides the sentence ended
"and then I ran the command.",
"and then I ran the command. But", # he speaks again
"and then I ran the command. But when I say",
"and then I ran the command. But when I say",
]
)
assert " ." not in screen # never a mark standing on its own
assert "But" in screen # and never a word paid to the alignment for one
# The full stop DOES land, because "But" settled behind it and confirmed it.
assert screen == "and then I ran the command. But when I say"


def test_a_pause_does_not_put_a_full_stop_in_the_middle_of_the_sentence():
Expand Down
Loading