Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions examples/semantic-vocabulary-drift-smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -151,9 +151,9 @@
"conflicting_definitions": 55,
"schema_version_same_runtime_forks": 7,
"multi_value_twins": 13,
"multi_value_forks": 4,
"multi_value_forks_semantic": 3,
"multi_value_fork_definitions": 10,
"multi_value_forks": 2,
"multi_value_forks_semantic": 1,
"multi_value_fork_definitions": 6,
"same_runtime_forks_semantic": 11,
"conflicting_values_semantic": 0,
}
Expand Down
6 changes: 3 additions & 3 deletions loopx/semantics/vocabulary_v0.json
Original file line number Diff line number Diff line change
Expand Up @@ -953,15 +953,15 @@
"conflicting_definitions": 55,
"schema_version_same_runtime_forks": 7,
"multi_value_twins": 13,
"multi_value_forks": 4,
"multi_value_fork_definitions": 10,
"multi_value_forks": 2,
"multi_value_fork_definitions": 6,
"same_runtime_forks_semantic": 11,
"conflicting_values_semantic": 0,
"multi_value_meaning": "Enums, named closed sets, Literal aliases, and TypeScript as-const arrays are vocabulary exactly as a NAME = \"value\" constant is, so they get the same collision rule. One name defined in two modules with identical values is a twin; with different values it is a fork. The semantic multi-value-fork budget excludes only names declared in scope_declarations.",
"multi_value_forks_note": "The 4 counted forks include SOURCE_SURFACES, whose four definitions are four CLI commands each listing its own data sources; that is bounded-context reuse of one name, not drift. It stays in the budget until M0.5 adds a scope field (RFC Section 5) and must not be removed by renaming.",
"multi_value_twins_note": "A twin whose two definitions are the Python and TypeScript owners of one registered cross_runtime vocabulary is required by I3, not drift, and the TypeScript migration RFC governs any reduction of that pair. Every other counted twin is duplicate knowledge: single-source it from its owner instead of raising this budget.",
"semantic_meaning": "same_runtime_forks_semantic and conflicting_values_semantic exclude module-local convention names such as SCHEMA_VERSION, COMMAND, or *_LABEL, which every module legitimately names for itself. The remaining names are shared vocabulary, where a duplicate is real drift rather than local naming; the unfiltered totals stay visible in the generated inventory summary.",
"multi_value_forks_semantic": 3
"multi_value_forks_semantic": 1
},
"scope_declarations": {
"SOURCE_SURFACES": {
Expand Down
34 changes: 12 additions & 22 deletions loopx/state_projection.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import re
from typing import Any

from .control_plane.goals.active_state_metadata import todo_role_for_heading
from .control_plane.todos.contract import (
TODO_TASK_PATTERN,
build_todo_id,
Expand Down Expand Up @@ -45,22 +46,6 @@
SECTION_HEADING_PATTERN = re.compile(r"^##+\s+(.+?)\s*$")
BULLET_PATTERN = re.compile(r"^\s*(?:[-*]|\d+[.)])\s+(.+?)\s*$")
PRIORITY_PATTERN = re.compile(r"^\[(P[0-4])\]\s+(.+)$", re.IGNORECASE)
USER_TODO_HEADER_MARKERS = (
"user todo",
"owner review",
"owner todo",
"user action",
"用户",
"人工",
"owner",
)
AGENT_TODO_HEADER_MARKERS = (
"agent todo",
"agent backlog",
"agent action",
"项目 agent",
"agent 待办",
)
NEXT_ACTION_EXECUTABLE_PATTERN = re.compile(
r"(?i)\b(?:run|repair|fix|implement|add|update|write|record|validate|"
r"rerun|debug|inspect|analy[sz]e|sync|refresh|test|benchmark|trace|"
Expand Down Expand Up @@ -355,12 +340,17 @@ def is_user_wait_text(value: Any) -> bool:


def _role_for_heading(heading: str) -> str | None:
normalized = heading.strip().lower()
if any(marker in normalized for marker in USER_TODO_HEADER_MARKERS):
return "user"
if any(marker in normalized for marker in AGENT_TODO_HEADER_MARKERS):
return "agent"
return None
"""Classify a state heading exactly as the Todo region writer does.

This module used to carry its own copy of the marker tuples. The copies had
drifted: they matched bare ``owner``, so a prose section named ``Ownership``
counted as user Todos; they missed ``codex todo``, which the writer creates;
and they had no archive guard, so ``Agent Todo Archive`` counted as live
agent Todos. The open counts feed ``state_projection_gap_warning``, where an
inflated agent count suppresses the very warning that says a Next Action is
executable with no agent Todo behind it.
"""
return todo_role_for_heading(heading)


def _open_count(summary: dict[str, Any] | None) -> int:
Expand Down
128 changes: 88 additions & 40 deletions tests/architecture/test_semantic_vocabulary_drift.py
Original file line number Diff line number Diff line change
Expand Up @@ -185,12 +185,65 @@ def test_registry_cannot_add_unanchored_output_selectors(name, metadata, selecti
smoke['check_coverage_floor'](registry)


SYNTHETIC_FORK_NAME = "SYNTHETIC_RENAME_SAMPLE_MARKERS"
SYNTHETIC_FORK_MODULES = (
"loopx/state_projection.py",
"loopx/control_plane/goals/active_state_metadata.py",
)


def _with_synthetic_fork(smoke, sources):
"""Inject a synthetic multi-value fork into two in-memory modules.

The rename-laundering limit is a property of the inventory machinery, not
of whichever real fork happens to be undeclared today. Track A retires real
forks one by one (PR #4643 retired two of the three this file used to rent),
so these pins carry their own sample instead: two modules, one name, two
disagreeing value sets -- exactly what makes a multi-value fork.
"""
texts = {
SYNTHETIC_FORK_MODULES[0]: f'\n{SYNTHETIC_FORK_NAME} = ("synthetic_left", "left_two")\n',
SYNTHETIC_FORK_MODULES[1]: f'\n{SYNTHETIC_FORK_NAME} = ("synthetic_right", "right_two")\n',
}
return [
smoke["SourceFile"](source.path, source.suffix, source.text + texts[source.path])
if source.path in texts
else source
for source in sources
]


def _rename_synthetic_side(smoke, sources, path):
return [
smoke["SourceFile"](
source.path,
source.suffix,
source.text.replace(SYNTHETIC_FORK_NAME, SYNTHETIC_FORK_NAME + "_RENAMED", 1),
)
if source.path == path
else source
for source in sources
]


def test_bounded_context_scope_excludes_only_declared_multi_value_fork() -> None:
"""A declaration is what removes a fork from the budget; prove it by removal.

The undeclared count itself is debt population, not a pin -- Track A lowers
it PR by PR. What must hold is the exclusion: dropping a declaration puts
its fork back into the semantic budget, exactly one.
"""
smoke = runpy.run_path(str(SMOKE))
registry = smoke["load_registry"]()
sources = smoke["load_sources"](REPO_ROOT)
inventory = smoke["build_inventory"](REPO_ROOT, sources=sources)
assert smoke["check_scope_declarations"](registry, inventory) == 3
with_declaration = smoke["check_scope_declarations"](registry, inventory)

without = copy.deepcopy(registry)
del without["scope_declarations"]["SOURCE_SURFACES"]
assert (
smoke["check_scope_declarations"](without, inventory) == with_declaration + 1
), "a declared fork is excluded from the semantic budget only while declared"


def test_renaming_one_side_of_a_fork_launders_the_semantic_budget() -> None:
Expand All @@ -204,20 +257,16 @@ def test_renaming_one_side_of_a_fork_launders_the_semantic_budget() -> None:
"""
smoke = runpy.run_path(str(SMOKE))
registry = smoke["load_registry"]()
sources = smoke["load_sources"](REPO_ROOT)
before = smoke["build_inventory"](REPO_ROOT, sources=sources)
assert smoke["check_scope_declarations"](registry, before) == 3
plain = smoke["load_sources"](REPO_ROOT)
base = smoke["check_scope_declarations"](registry, smoke["build_inventory"](REPO_ROOT, sources=plain))

path = "loopx/state_projection.py"
name = "AGENT_TODO_HEADER_MARKERS"
renamed = [
smoke["SourceFile"](source.path, source.suffix, source.text.replace(name, name + "_RENAMED", 1))
if source.path == path
else source
for source in sources
]
forked = _with_synthetic_fork(smoke, plain)
before = smoke["build_inventory"](REPO_ROOT, sources=forked)
assert smoke["check_scope_declarations"](registry, before) == base + 1

renamed = _rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0])
after = smoke["build_inventory"](REPO_ROOT, sources=renamed)
assert smoke["check_scope_declarations"](registry, after) == 2, (
assert smoke["check_scope_declarations"](registry, after) == base, (
"a rename no longer lowers the semantic budget; the RFC known-limits entry "
"('Renames launder a collision') is now stale and must be revised"
)
Expand All @@ -237,9 +286,14 @@ def test_divergent_value_sets_lists_the_names_a_rename_would_hide() -> None:

smoke = runpy.run_path(str(SMOKE))
sources = smoke["load_sources"](REPO_ROOT)
inventory = smoke["build_inventory"](REPO_ROOT, sources=sources)
listed = {row["name"] for row in divergent_value_sets(inventory)}
assert {"AGENT_TODO_HEADER_MARKERS", "USER_TODO_HEADER_MARKERS", "RAW_MATERIAL_KEY_HINTS"} <= listed
inventory = smoke["build_inventory"](REPO_ROOT, sources=_with_synthetic_fork(smoke, sources))
rows = divergent_value_sets(inventory)
listed = {row["name"] for row in rows}
assert SYNTHETIC_FORK_NAME in listed
assert {row["value_sets"] for row in rows if row["name"] == SYNTHETIC_FORK_NAME} == {2}
# The one real undeclared fork that survives PR #4643; when Track A retires
# it, this assertion retires with it. Until then the advisory must name it.
assert "RAW_MATERIAL_KEY_HINTS" in listed

# The advisory is not a budget input: it must not appear in the committed
# inventory, which stays the single computed authority.
Expand Down Expand Up @@ -273,43 +327,37 @@ def test_rename_visibility_splits_into_three_cases() -> None:

smoke = runpy.run_path(str(SMOKE))
registry = smoke["load_registry"]()
sources = smoke["load_sources"](REPO_ROOT)
name = "AGENT_TODO_HEADER_MARKERS"

def renamed_in(paths):
out = sources
for path in paths:
out = [
smoke["SourceFile"](source.path, source.suffix, source.text.replace(name, name + "_RENAMED", 1))
if source.path == path
else source
for source in out
]
return out
plain = smoke["load_sources"](REPO_ROOT)
forked = _with_synthetic_fork(smoke, plain)

# Case 1: declared name, one side renamed -> the declaration no longer resolves.
declared = _restate(smoke, sources, "loopx/global_todos.py", "SOURCE_SURFACES", "GT_SOURCE_SURFACES")
declared = _restate(smoke, plain, "loopx/global_todos.py", "SOURCE_SURFACES", "GT_SOURCE_SURFACES")
with pytest.raises(smoke["Drift"], match="every defining module"):
smoke["check_scope_declarations"](registry, smoke["build_inventory"](REPO_ROOT, sources=declared))

# Case 2: undeclared name, one side renamed -> gone from the budget AND the advisory.
partial = smoke["build_inventory"](REPO_ROOT, sources=renamed_in(["loopx/state_projection.py"]))
assert name not in {entry["name"] for entry in partial["duplicate_definitions"]["multi_value_forks"]}
assert name not in {row["name"] for row in divergent_value_sets(partial)}
partial = smoke["build_inventory"](
REPO_ROOT, sources=_rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0])
)
assert SYNTHETIC_FORK_NAME not in {
entry["name"] for entry in partial["duplicate_definitions"]["multi_value_forks"]
}
assert SYNTHETIC_FORK_NAME not in {row["name"] for row in divergent_value_sets(partial)}

# Case 3: every side renamed -> also invisible; indistinguishable from an honest rename.
whole = smoke["build_inventory"](
REPO_ROOT,
sources=renamed_in([
"loopx/state_projection.py",
"loopx/control_plane/goals/active_state_metadata.py",
]),
sources=_rename_synthetic_side(
smoke, _rename_synthetic_side(smoke, forked, SYNTHETIC_FORK_MODULES[0]), SYNTHETIC_FORK_MODULES[1]
),
)
assert name not in {entry["name"] for entry in whole["duplicate_definitions"]["multi_value_forks"]}
assert name not in {row["name"] for row in divergent_value_sets(whole)}
assert SYNTHETIC_FORK_NAME not in {
entry["name"] for entry in whole["duplicate_definitions"]["multi_value_forks"]
}
assert SYNTHETIC_FORK_NAME not in {row["name"] for row in divergent_value_sets(whole)}

# The surviving forks are what the advisory does list, by name.
assert "USER_TODO_HEADER_MARKERS" in {row["name"] for row in divergent_value_sets(partial)}
assert "RAW_MATERIAL_KEY_HINTS" in {row["name"] for row in divergent_value_sets(partial)}


def test_bounded_context_scope_requires_every_distinct_defining_module() -> None:
Expand Down
54 changes: 54 additions & 0 deletions tests/control_plane/test_todo_machine_region.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,3 +94,57 @@ def test_mixed_legacy_boundary_does_not_skip_next_heading() -> None:
parsed = parse_active_state_todos(source, item_limit=None)
assert [item["text"] for item in parsed["user_todos"]["items"]] == ["User task."]
assert [item["text"] for item in parsed["agent_todos"]["items"]] == ["Agent task."]


def test_state_counts_classify_headings_exactly_as_the_region_writer() -> None:
"""The open counts and the Todo region writer must read one marker set.

``loopx/state_projection.py`` carried its own copy of the marker tuples and
they had drifted three ways: a bare ``owner`` marker made a prose section
named ``Ownership`` count as user Todos, ``codex todo`` was missing although
the writer creates it, and there was no archive guard so an archived section
counted as live work. The counts feed ``state_projection_gap_warning``, so an
inflated agent count suppresses the warning that a Next Action is executable
with no agent Todo behind it -- the failure is silence, not a wrong number on
a screen.
"""
from loopx.control_plane.goals.active_state_metadata import todo_role_for_heading
from loopx.state_projection import summarize_state_todo_open_counts

state_text = "\n".join([
"# Goal state",
"",
"## Agent Todo",
"",
"- [ ] [P1] implement the reader metric",
"- [ ] [P2] classify the candidate groups",
"",
"## Agent Todo Archive",
"",
"- [ ] [P1] archived one",
"- [ ] [P2] archived two",
"",
"## Codex Todo",
"",
"- [ ] [P1] a heading the region writer creates",
"",
"## Ownership",
"",
"- [ ] prose about who owns what, not a Todo",
"",
])

counts = summarize_state_todo_open_counts(state_text)
assert counts == {"user": 0, "agent": 3}, (
"archived Todos must not count as live, a Codex Todo heading must count, "
"and an Ownership prose section must not count as user Todos"
)

for heading, expected in [
("Agent Todo Archive", None),
("Codex Todo", "agent"),
("Ownership", None),
("Agent Todo", "agent"),
("Owner Reading Queue", "user"),
]:
assert todo_role_for_heading(heading) == expected, heading
Loading