From fcccae5ae077ea56f6f36abac66c9cd1e7646d2f Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Thu, 17 Sep 2026 05:54:32 -0400 Subject: [PATCH 1/2] fix(state): count Todo headings the way the region writer classifies them MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `loopx/state_projection.py` carried its own copy of the Todo header marker tuples, and `loopx/control_plane/goals/active_state_metadata.py` carried another. Both classify a heading from the same active-state document into the same `user` / `agent` / `None` roles, so this is one contract with two implementations -- and they had drifted. Measured over 23 realistic headings, the two disagree on 15. Three of those disagreements are defects, not preferences: - `Agent Todo Archive` counted as **live** agent Todos. The writer already guards archives through `TODO_ARCHIVE_HEADER_MARKERS`; the counter had no such guard. - `Codex Todo` counted as nothing, although the region writer creates exactly that heading, so a whole class of agent Todos was invisible to the count. - A bare `owner` marker made prose sections match: a section named `Ownership` or `Downstream Owner Notes` counted its bullets as user Todos. The consequence is not a wrong number on a screen. The counts feed `state_projection_gap_warning`, which reports that a Next Action is executable while no agent Todo stands behind it. An agent count inflated by archived work suppresses that warning, so the failure mode is silence. On a state document with two live Todos, two archived ones, one `Codex Todo` and one `Ownership` prose section, the counter returned `{'user': 1, 'agent': 5}` where the truth is `{'user': 0, 'agent': 3}`. Delete both copies and call `todo_role_for_heading`, so the counter and the region writer agree by construction. The markers this drops -- `agent backlog`, `agent action`, `user action`, `用户`, `人工`, bare `owner` -- appear in no shipped template or fixture as a heading, and the writer never creates them, so counting them reported work the machinery could not see or act on. Two undeclared multi-value forks disappear with the copies: `multi_value_forks` 4 -> 2, `multi_value_forks_semantic` 3 -> 1, `multi_value_fork_definitions` 10 -> 6, lowered in this diff on both the registry and `BUDGET_ANCHOR` sides as the equality check requires. Refs #4447 Co-Authored-By: Claude Opus 5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- examples/semantic-vocabulary-drift-smoke.py | 6 +-- loopx/semantics/vocabulary_v0.json | 6 +-- loopx/state_projection.py | 34 +++++------- .../control_plane/test_todo_machine_region.py | 54 +++++++++++++++++++ 4 files changed, 72 insertions(+), 28 deletions(-) diff --git a/examples/semantic-vocabulary-drift-smoke.py b/examples/semantic-vocabulary-drift-smoke.py index d23bad3061..f448662dd1 100755 --- a/examples/semantic-vocabulary-drift-smoke.py +++ b/examples/semantic-vocabulary-drift-smoke.py @@ -151,9 +151,9 @@ "conflicting_definitions": 59, "schema_version_same_runtime_forks": 7, "multi_value_twins": 13, - "multi_value_forks": 4, - "multi_value_forks_semantic": 3, - "multi_value_fork_definitions": 10, + "multi_value_forks": 2, + "multi_value_forks_semantic": 1, + "multi_value_fork_definitions": 6, "same_runtime_forks_semantic": 13, "conflicting_values_semantic": 0, } diff --git a/loopx/semantics/vocabulary_v0.json b/loopx/semantics/vocabulary_v0.json index bc8a41835b..d3753d82c9 100644 --- a/loopx/semantics/vocabulary_v0.json +++ b/loopx/semantics/vocabulary_v0.json @@ -892,15 +892,15 @@ "conflicting_definitions": 59, "schema_version_same_runtime_forks": 7, "multi_value_twins": 13, - "multi_value_forks": 4, - "multi_value_fork_definitions": 10, + "multi_value_forks": 2, + "multi_value_fork_definitions": 6, "same_runtime_forks_semantic": 13, "conflicting_values_semantic": 0, "multi_value_meaning": "Enums, named closed sets, Literal aliases, and TypeScript as-const arrays are vocabulary exactly as a NAME = \"value\" constant is, so they get the same collision rule. One name defined in two modules with identical values is a twin; with different values it is a fork. The semantic multi-value-fork budget excludes only names declared in scope_declarations.", "multi_value_forks_note": "The 4 counted forks include SOURCE_SURFACES, whose four definitions are four CLI commands each listing its own data sources; that is bounded-context reuse of one name, not drift. It stays in the budget until M0.5 adds a scope field (RFC Section 5) and must not be removed by renaming.", "multi_value_twins_note": "A twin whose two definitions are the Python and TypeScript owners of one registered cross_runtime vocabulary is required by I3, not drift, and the TypeScript migration RFC governs any reduction of that pair. Every other counted twin is duplicate knowledge: single-source it from its owner instead of raising this budget.", "semantic_meaning": "same_runtime_forks_semantic and conflicting_values_semantic exclude module-local convention names such as SCHEMA_VERSION, COMMAND, or *_LABEL, which every module legitimately names for itself. The remaining names are shared vocabulary, where a duplicate is real drift rather than local naming; the unfiltered totals stay visible in the generated inventory summary.", - "multi_value_forks_semantic": 3 + "multi_value_forks_semantic": 1 }, "scope_declarations": { "SOURCE_SURFACES": { diff --git a/loopx/state_projection.py b/loopx/state_projection.py index 438bb88dbb..0cbd5b6831 100644 --- a/loopx/state_projection.py +++ b/loopx/state_projection.py @@ -3,6 +3,7 @@ import re from typing import Any +from .control_plane.goals.active_state_metadata import todo_role_for_heading from .control_plane.todos.contract import ( TODO_TASK_PATTERN, build_todo_id, @@ -45,22 +46,6 @@ SECTION_HEADING_PATTERN = re.compile(r"^##+\s+(.+?)\s*$") BULLET_PATTERN = re.compile(r"^\s*(?:[-*]|\d+[.)])\s+(.+?)\s*$") PRIORITY_PATTERN = re.compile(r"^\[(P[0-4])\]\s+(.+)$", re.IGNORECASE) -USER_TODO_HEADER_MARKERS = ( - "user todo", - "owner review", - "owner todo", - "user action", - "用户", - "人工", - "owner", -) -AGENT_TODO_HEADER_MARKERS = ( - "agent todo", - "agent backlog", - "agent action", - "项目 agent", - "agent 待办", -) NEXT_ACTION_EXECUTABLE_PATTERN = re.compile( r"(?i)\b(?:run|repair|fix|implement|add|update|write|record|validate|" r"rerun|debug|inspect|analy[sz]e|sync|refresh|test|benchmark|trace|" @@ -355,12 +340,17 @@ def is_user_wait_text(value: Any) -> bool: def _role_for_heading(heading: str) -> str | None: - normalized = heading.strip().lower() - if any(marker in normalized for marker in USER_TODO_HEADER_MARKERS): - return "user" - if any(marker in normalized for marker in AGENT_TODO_HEADER_MARKERS): - return "agent" - return None + """Classify a state heading exactly as the Todo region writer does. + + This module used to carry its own copy of the marker tuples. The copies had + drifted: they matched bare ``owner``, so a prose section named ``Ownership`` + counted as user Todos; they missed ``codex todo``, which the writer creates; + and they had no archive guard, so ``Agent Todo Archive`` counted as live + agent Todos. The open counts feed ``state_projection_gap_warning``, where an + inflated agent count suppresses the very warning that says a Next Action is + executable with no agent Todo behind it. + """ + return todo_role_for_heading(heading) def _open_count(summary: dict[str, Any] | None) -> int: diff --git a/tests/control_plane/test_todo_machine_region.py b/tests/control_plane/test_todo_machine_region.py index 8fdb9f2d5b..a2ff949f65 100644 --- a/tests/control_plane/test_todo_machine_region.py +++ b/tests/control_plane/test_todo_machine_region.py @@ -94,3 +94,57 @@ def test_mixed_legacy_boundary_does_not_skip_next_heading() -> None: parsed = parse_active_state_todos(source, item_limit=None) assert [item["text"] for item in parsed["user_todos"]["items"]] == ["User task."] assert [item["text"] for item in parsed["agent_todos"]["items"]] == ["Agent task."] + + +def test_state_counts_classify_headings_exactly_as_the_region_writer() -> None: + """The open counts and the Todo region writer must read one marker set. + + ``loopx/state_projection.py`` carried its own copy of the marker tuples and + they had drifted three ways: a bare ``owner`` marker made a prose section + named ``Ownership`` count as user Todos, ``codex todo`` was missing although + the writer creates it, and there was no archive guard so an archived section + counted as live work. The counts feed ``state_projection_gap_warning``, so an + inflated agent count suppresses the warning that a Next Action is executable + with no agent Todo behind it -- the failure is silence, not a wrong number on + a screen. + """ + from loopx.control_plane.goals.active_state_metadata import todo_role_for_heading + from loopx.state_projection import summarize_state_todo_open_counts + + state_text = "\n".join([ + "# Goal state", + "", + "## Agent Todo", + "", + "- [ ] [P1] implement the reader metric", + "- [ ] [P2] classify the candidate groups", + "", + "## Agent Todo Archive", + "", + "- [ ] [P1] archived one", + "- [ ] [P2] archived two", + "", + "## Codex Todo", + "", + "- [ ] [P1] a heading the region writer creates", + "", + "## Ownership", + "", + "- [ ] prose about who owns what, not a Todo", + "", + ]) + + counts = summarize_state_todo_open_counts(state_text) + assert counts == {"user": 0, "agent": 3}, ( + "archived Todos must not count as live, a Codex Todo heading must count, " + "and an Ownership prose section must not count as user Todos" + ) + + for heading, expected in [ + ("Agent Todo Archive", None), + ("Codex Todo", "agent"), + ("Ownership", None), + ("Agent Todo", "agent"), + ("Owner Reading Queue", "user"), + ]: + assert todo_role_for_heading(heading) == expected, heading From 0538faa3c70fbe7985ad8d150eda3fd0c9bbfc9e Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Thu, 17 Sep 2026 06:57:48 -0400 Subject: [PATCH 2/2] test(semantics): pin rename laundering on an injected fork, not a rented one PR #4643 retired the AGENT/USER_TODO_HEADER_MARKERS forks, which four drift tests from #4614 had rented as samples for the rename-laundering limit. Track A retires real forks one by one, so renting them makes every retirement a test breakage -- this PR included. The pins now inject their own synthetic fork (one name, two modules, two disagreeing value sets) and measure against the live baseline instead of hardcoded counts, so they pin the machinery -- the RFC Section 9 known limit -- independently of the debt population: - the laundering test measures base / base+1 / base instead of 3/2; - the declaration test proves exclusion by removing SOURCE_SURFACES from the registry and watching the count rise by exactly one, instead of asserting today's undeclared count; - the three-case rename boundary runs its undeclared cases on the synthetic fork; Case 1 (declared rename refused) still uses SOURCE_SURFACES, which is still declared and still 4-way divergent; - the advisory test keeps one real assertion: RAW_MATERIAL_KEY_HINTS, the only undeclared fork surviving this PR, must stay listed, with a comment that the assertion retires with the fork. Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- .../test_semantic_vocabulary_drift.py | 128 ++++++++++++------ 1 file changed, 88 insertions(+), 40 deletions(-) diff --git a/tests/architecture/test_semantic_vocabulary_drift.py b/tests/architecture/test_semantic_vocabulary_drift.py index 9855b8257e..5cd01c7d8c 100644 --- a/tests/architecture/test_semantic_vocabulary_drift.py +++ b/tests/architecture/test_semantic_vocabulary_drift.py @@ -185,12 +185,65 @@ def test_registry_cannot_add_unanchored_output_selectors(name, metadata, selecti smoke['check_coverage_floor'](registry) +SYNTHETIC_FORK_NAME = "SYNTHETIC_RENAME_SAMPLE_MARKERS" +SYNTHETIC_FORK_MODULES = ( + "loopx/state_projection.py", + "loopx/control_plane/goals/active_state_metadata.py", +) + + +def _with_synthetic_fork(smoke, sources): + """Inject a synthetic multi-value fork into two in-memory modules. + + The rename-laundering limit is a property of the inventory machinery, not + of whichever real fork happens to be undeclared today. Track A retires real + forks one by one (PR #4643 retired two of the three this file used to rent), + so these pins carry their own sample instead: two modules, one name, two + disagreeing value sets -- exactly what makes a multi-value fork. + """ + texts = { + SYNTHETIC_FORK_MODULES[0]: f'\n{SYNTHETIC_FORK_NAME} = ("synthetic_left", "left_two")\n', + SYNTHETIC_FORK_MODULES[1]: f'\n{SYNTHETIC_FORK_NAME} = ("synthetic_right", "right_two")\n', + } + return [ + smoke["SourceFile"](source.path, source.suffix, source.text + texts[source.path]) + if source.path in texts + else source + for source in sources + ] + + +def _rename_synthetic_side(smoke, sources, path): + return [ + smoke["SourceFile"]( + source.path, + source.suffix, + source.text.replace(SYNTHETIC_FORK_NAME, SYNTHETIC_FORK_NAME + "_RENAMED", 1), + ) + if source.path == path + else source + for source in sources + ] + + def test_bounded_context_scope_excludes_only_declared_multi_value_fork() -> None: + """A declaration is what removes a fork from the budget; prove it by removal. + + The undeclared count itself is debt population, not a pin -- Track A lowers + it PR by PR. What must hold is the exclusion: dropping a declaration puts + its fork back into the semantic budget, exactly one. + """ smoke = runpy.run_path(str(SMOKE)) registry = smoke["load_registry"]() sources = smoke["load_sources"](REPO_ROOT) inventory = smoke["build_inventory"](REPO_ROOT, sources=sources) - assert smoke["check_scope_declarations"](registry, inventory) == 3 + with_declaration = smoke["check_scope_declarations"](registry, inventory) + + without = copy.deepcopy(registry) + del without["scope_declarations"]["SOURCE_SURFACES"] + assert ( + smoke["check_scope_declarations"](without, inventory) == with_declaration + 1 + ), "a declared fork is excluded from the semantic budget only while declared" def test_renaming_one_side_of_a_fork_launders_the_semantic_budget() -> None: @@ -204,20 +257,16 @@ def test_renaming_one_side_of_a_fork_launders_the_semantic_budget() -> None: """ smoke = runpy.run_path(str(SMOKE)) registry = smoke["load_registry"]() - sources = smoke["load_sources"](REPO_ROOT) - before = smoke["build_inventory"](REPO_ROOT, sources=sources) - assert smoke["check_scope_declarations"](registry, before) == 3 + plain = smoke["load_sources"](REPO_ROOT) + base = smoke["check_scope_declarations"](registry, smoke["build_inventory"](REPO_ROOT, sources=plain)) - path = "loopx/state_projection.py" - name = "AGENT_TODO_HEADER_MARKERS" - renamed = [ - smoke["SourceFile"](source.path, source.suffix, source.text.replace(name, name + "_RENAMED", 1)) - if source.path == path - else source - for source in sources - ] + forked = _with_synthetic_fork(smoke, plain) + before = smoke["build_inventory"](REPO_ROOT, sources=forked) + assert smoke["check_scope_declarations"](registry, before) == base + 1 + + renamed = _rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0]) after = smoke["build_inventory"](REPO_ROOT, sources=renamed) - assert smoke["check_scope_declarations"](registry, after) == 2, ( + assert smoke["check_scope_declarations"](registry, after) == base, ( "a rename no longer lowers the semantic budget; the RFC known-limits entry " "('Renames launder a collision') is now stale and must be revised" ) @@ -237,9 +286,14 @@ def test_divergent_value_sets_lists_the_names_a_rename_would_hide() -> None: smoke = runpy.run_path(str(SMOKE)) sources = smoke["load_sources"](REPO_ROOT) - inventory = smoke["build_inventory"](REPO_ROOT, sources=sources) - listed = {row["name"] for row in divergent_value_sets(inventory)} - assert {"AGENT_TODO_HEADER_MARKERS", "USER_TODO_HEADER_MARKERS", "RAW_MATERIAL_KEY_HINTS"} <= listed + inventory = smoke["build_inventory"](REPO_ROOT, sources=_with_synthetic_fork(smoke, sources)) + rows = divergent_value_sets(inventory) + listed = {row["name"] for row in rows} + assert SYNTHETIC_FORK_NAME in listed + assert {row["value_sets"] for row in rows if row["name"] == SYNTHETIC_FORK_NAME} == {2} + # The one real undeclared fork that survives PR #4643; when Track A retires + # it, this assertion retires with it. Until then the advisory must name it. + assert "RAW_MATERIAL_KEY_HINTS" in listed # The advisory is not a budget input: it must not appear in the committed # inventory, which stays the single computed authority. @@ -273,43 +327,37 @@ def test_rename_visibility_splits_into_three_cases() -> None: smoke = runpy.run_path(str(SMOKE)) registry = smoke["load_registry"]() - sources = smoke["load_sources"](REPO_ROOT) - name = "AGENT_TODO_HEADER_MARKERS" - - def renamed_in(paths): - out = sources - for path in paths: - out = [ - smoke["SourceFile"](source.path, source.suffix, source.text.replace(name, name + "_RENAMED", 1)) - if source.path == path - else source - for source in out - ] - return out + plain = smoke["load_sources"](REPO_ROOT) + forked = _with_synthetic_fork(smoke, plain) # Case 1: declared name, one side renamed -> the declaration no longer resolves. - declared = _restate(smoke, sources, "loopx/global_todos.py", "SOURCE_SURFACES", "GT_SOURCE_SURFACES") + declared = _restate(smoke, plain, "loopx/global_todos.py", "SOURCE_SURFACES", "GT_SOURCE_SURFACES") with pytest.raises(smoke["Drift"], match="every defining module"): smoke["check_scope_declarations"](registry, smoke["build_inventory"](REPO_ROOT, sources=declared)) # Case 2: undeclared name, one side renamed -> gone from the budget AND the advisory. - partial = smoke["build_inventory"](REPO_ROOT, sources=renamed_in(["loopx/state_projection.py"])) - assert name not in {entry["name"] for entry in partial["duplicate_definitions"]["multi_value_forks"]} - assert name not in {row["name"] for row in divergent_value_sets(partial)} + partial = smoke["build_inventory"]( + REPO_ROOT, sources=_rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0]) + ) + assert SYNTHETIC_FORK_NAME not in { + entry["name"] for entry in partial["duplicate_definitions"]["multi_value_forks"] + } + assert SYNTHETIC_FORK_NAME not in {row["name"] for row in divergent_value_sets(partial)} # Case 3: every side renamed -> also invisible; indistinguishable from an honest rename. whole = smoke["build_inventory"]( REPO_ROOT, - sources=renamed_in([ - "loopx/state_projection.py", - "loopx/control_plane/goals/active_state_metadata.py", - ]), + sources=_rename_synthetic_side( + smoke, _rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0]), SYNTHETIC_FORK_MODULES[1] + ), ) - assert name not in {entry["name"] for entry in whole["duplicate_definitions"]["multi_value_forks"]} - assert name not in {row["name"] for row in divergent_value_sets(whole)} + assert SYNTHETIC_FORK_NAME not in { + entry["name"] for entry in whole["duplicate_definitions"]["multi_value_forks"] + } + assert SYNTHETIC_FORK_NAME not in {row["name"] for row in divergent_value_sets(whole)} # The surviving forks are what the advisory does list, by name. - assert "USER_TODO_HEADER_MARKERS" in {row["name"] for row in divergent_value_sets(partial)} + assert "RAW_MATERIAL_KEY_HINTS" in {row["name"] for row in divergent_value_sets(partial)} def test_bounded_context_scope_requires_every_distinct_defining_module() -> None: