From e97f93234b1c9b720a45e87d0240758e3f48f0b5 Mon Sep 17 00:00:00 2001 From: hannesreinsch Date: Wed, 9 Sep 2026 15:17:15 +0200 Subject: [PATCH] fix(stream): the mark lands once a later word confirms it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reported as "there was a break before, but no punctuation", and it is the cost the last commit named out loud rather than a new defect. A mark rides on the word in front of it, and `stable_prefix` will not type the mark on a word still touching the end of the audio — a pause is how whisper decides a sentence ended, and it takes that decision back the moment the speaker carries on. So the word landed bare and nothing could ever put the mark on afterwards: the word is on screen and there is no un-type. A LATER pass answers the question the earlier one could not. If the word carrying the mark is no longer at the end of the transcript — real speech follows it, and whisper STILL ends the sentence there — the mark was decided WITH the following audio, which is the same test every other word passes before it is typed. `missing_mark` returns it, joined to its word with no space in front, and only ever in the same breath as the word that proves it. So it can never be the lone full stop of #55: no confirming word, no mark. `_reached` is split out of `stream_tail` because both now ask the same question about the same two sequences — where does the screen END inside this transcript — and two answers to that would drift apart word by word. Driven through the reported sequence, the streamed text is now identical to the whole-clip transcript, punctuation included: I did not say the three times. I did, however, say really three times. `_drive_stream` in the tests is the loop's four calls in its own order, so the three earlier reports are replayed against it too and none of them regressed. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01P7oJz8M9QzsdJimoM318cj --- murmurflow/dictate.py | 55 +++++++++++++++++++++++--- tests/test_murmurflow.py | 84 +++++++++++++++++++++++++++++++--------- 2 files changed, 116 insertions(+), 23 deletions(-) diff --git a/murmurflow/dictate.py b/murmurflow/dictate.py index 06976e6..4ae9638 100644 --- a/murmurflow/dictate.py +++ b/murmurflow/dictate.py @@ -1790,13 +1790,22 @@ def stream_tail(pasted: str, final: str) -> str: if not already: return final words = final.split() - keys = [_key(word) for word in words] + return " ".join(words[_reached(already, [_key(word) for word in words]) :]) + + +def _reached(already: list[str], keys: list[str]) -> int: + """The index in ``keys`` just past everything that is already on screen. + + Split out of :func:`stream_tail` because :func:`missing_mark` asks the same question about the + same two sequences — where does the screen END inside this transcript — and two answers to that + would drift apart word by word. + """ if keys[: len(already)] == already: - return " ".join(words[len(already) :]) + return len(already) matcher = difflib.SequenceMatcher(a=already, b=keys, autojunk=False) matched = [block for block in matcher.get_matching_blocks() if block.size] if not matched: # nothing corresponds: trust the count, lose nothing - return " ".join(words[len(already) :]) + return len(already) reached = matched[-1] # Words on screen PAST the alignment are ones the final pass said differently. The tail that # follows them is not new text, it is the same words again in the better model's wording, and @@ -1804,7 +1813,40 @@ def stream_tail(pasted: str, final: str) -> str: # exactly what one reworded last word looks like ("the design" + "designs"). One dropped for # one left over: the rewording is skipped and anything genuinely beyond the screen still lands. reworded = len(already) - (reached.a + reached.size) - return " ".join(words[reached.b + reached.size + reworded :]) + return reached.b + reached.size + reworded + + +def missing_mark(pasted: str, settled: str) -> str: + """The mark that belongs directly after ``pasted``, once a later word has confirmed it. + + **This is the punctuation streaming used to lose, and it is the last of it.** A mark rides on + the word in front of it, and :func:`stable_prefix` will not type the mark on a word that is + still touching the end of the audio — a pause is how whisper decides a sentence ended, and it + takes that decision back the moment the speaker carries on. So the word lands bare, and + nothing could ever put the mark on afterwards: the word is already on screen and there is no + un-type. Reported as "there was a break before, but no punctuation". + + A LATER pass answers the question the earlier one could not. If the word carrying the mark is + no longer at the end of the transcript — real speech follows it, and whisper still ends the + sentence there — the mark is a decision made WITH the following audio, which is the same test + every other word passes before it is typed. It goes on with no space in front of it, joined to + the word it belongs to. + + Returns ``""`` unless a word after it has also settled, so this can never be the lone full stop + of :func:`end_mark`'s docstring: the mark is only ever typed in the same breath as the word + that proves it. + """ + already = [_key(word) for word in pasted.split()] + words = settled.split() + if not already or not words: + return "" + index = _reached(already, [_key(word) for word in words]) + if index <= 0 or index >= len(words): + return "" # nothing before it, or nothing after it to confirm it + mark = _TRAILING_MARK.search(words[index - 1]) + if not mark or pasted.rstrip().endswith(mark.group()): + return "" + return mark.group() def _partial(live: Path, snapshot: Path, language: str = "") -> Heard: @@ -1948,6 +1990,9 @@ def _stream_loop(rec: Recording, stream: Stream) -> None: settled = stable_prefix(previous, heard) previous = heard chunk = stream_tail(stream.text, settled) if settled else "" + # The mark on the word already at the end of the screen, now that a later word has + # settled behind it. Only ever together with that word — see :func:`missing_mark`. + mark = missing_mark(stream.text, settled) if chunk else "" if chunk: with _INJECT_LOCK: # Inside the lock, because `stop_streaming` sets this and then takes the lock: past @@ -1961,7 +2006,7 @@ def _stream_loop(rec: Recording, stream: Stream) -> None: # `stream.text` is literally what is on the screen, built from what LANDED rather # than from what was asked for — see :func:`place`. The leading space travels with # the chunk, so this is a concatenation and never a re-join. - landed = place(f" {chunk}" if stream.text else chunk) + landed = place(f"{mark} {chunk}" if stream.text else chunk) if landed: stream.text = f"{stream.text}{landed}".strip() stream.typed += 1 diff --git a/tests/test_murmurflow.py b/tests/test_murmurflow.py index 504c241..6ae7a0d 100644 --- a/tests/test_murmurflow.py +++ b/tests/test_murmurflow.py @@ -1695,6 +1695,57 @@ def test_the_microphone_closes_itself_when_the_second_tap_never_comes(monkeypatc assert dictate.MAX_CLIP_SECONDS == 600 +def _drive_stream(passes): + """What lands on screen when the live pass reads ``passes``, one after another. + + The same four calls `_stream_loop` makes, in the same order, so a sequence that broke a real + dictation can be replayed as a test. Kept beside the tests that use it rather than inside them: + four copies of the loop drift, and a copy that drifts stops testing the loop. + """ + screen = previous = "" + for heard in passes: + settled = dictate.stable_prefix(previous, heard) + previous = heard + chunk = dictate.stream_tail(screen, settled) if settled else "" + mark = dictate.missing_mark(screen, settled) if chunk else "" + if chunk: + screen = f"{screen}{mark} {chunk}".strip() if screen else chunk + return screen + + +def test_the_mark_lands_once_a_later_word_confirms_it(): + """Reported as "there was a break before, but no punctuation". + + A mark rides on the word in front of it, and that word is never typed with its mark while it + still touches the end of the audio — a pause is how whisper decides a sentence ended, and it + takes that back the moment the speaker carries on. So the word landed bare and nothing could + put the mark on afterwards. A later pass answers it: real speech follows and whisper STILL + ends the sentence there, so the mark goes on, joined to the word it belongs to. + """ + reference = "I did not say the three times. I did, however, say really three times." + screen = _drive_stream( + [ + "I did not say the three times", + "I did not say the three times", # the break + "I did not say the three times.", # whisper ends the sentence + "I did not say the three times.", + "I did not say the three times. I", # he carries on + "I did not say the three times. I did however say", + "I did not say the three times. I did, however, say really", + reference, + reference, + ] + ) + # ...plus the one thing only the key release can know: the mark that ends the clip. + landed = screen + dictate.end_mark(screen, reference) + assert landed == reference # streamed, and identical to the whole-clip transcript + # And the fixes it must not undo, driven through the same loop. + assert _drive_stream(["Could you please work on my", "Could you please work on my..."] * 2) == ( + "Could you please work on my" + ) + assert "The The" not in _drive_stream(["Well, yeah. The The The"] * 3) + + def test_a_pause_never_types_a_lone_full_stop_where_the_next_word_goes(): """Reported as "it puts a period instead of the word" after a short break. @@ -1704,24 +1755,21 @@ def test_a_pause_never_types_a_lone_full_stop_where_the_next_word_goes(): screen with no letters in it, so the next alignment read it as something the final pass had reworded and dropped a real word to pay for it. The word this ate, in the report, was "But". """ - screen, previous = "", "" - for heard in ( - "and then I ran the command", - "and then I ran the command", # the pause: the transcript stops growing - "and then I ran the command.", # whisper decides the sentence ended - "and then I ran the command.", - "and then I ran the command. But", # he speaks again - "and then I ran the command. But when I say", - "and then I ran the command. But when I say", - ): - settled = dictate.stable_prefix(previous, heard) - previous = heard - chunk = dictate.stream_tail(screen, settled) if settled else "" - if chunk: - screen = f"{screen} {chunk}".strip() if screen else chunk - assert " ." not in screen - assert "But" in screen - assert screen == "and then I ran the command But when I say" + screen = _drive_stream( + [ + "and then I ran the command", + "and then I ran the command", # the pause: the transcript stops growing + "and then I ran the command.", # whisper decides the sentence ended + "and then I ran the command.", + "and then I ran the command. But", # he speaks again + "and then I ran the command. But when I say", + "and then I ran the command. But when I say", + ] + ) + assert " ." not in screen # never a mark standing on its own + assert "But" in screen # and never a word paid to the alignment for one + # The full stop DOES land, because "But" settled behind it and confirmed it. + assert screen == "and then I ran the command. But when I say" def test_a_pause_does_not_put_a_full_stop_in_the_middle_of_the_sentence():