Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 12 additions & 3 deletions docs/architecture/rfcs/semantic-vocabulary-convergence-v0.md
Original file line number Diff line number Diff line change
Expand Up @@ -643,10 +643,19 @@ lists Python enums, closed sets, `Literal` aliases, TypeScript `as const`
arrays, and duplicate definitions split into cross-runtime twins, same-runtime
forks, conflicting values, and multi-value twins and forks, one entry per line.
Every multi-value collision carries each defining module and its value set, so
the divergence itself is reviewable rather than only its count. Consumer counts are printed by `--report`; merge-candidate groups are available
through `merge_candidate_groups`; all inventory output is uncommitted.
the divergence itself is reviewable rather than only its count. Consumer counts
and merge-candidate groups are both printed by `--report`;
`merge_candidate_groups` returns the groups; all inventory output is
uncommitted.
Merge candidates are advisory
because an equal value set is not proof of one concept. Single-module string
because an equal value set is not proof of one concept. The printed list drops
the groups whose names are exactly one registered vocabulary's own owner
symbols: `EffectiveAction` and `EFFECTIVE_ACTIONS` are two runtimes spelling one
registered concept, not two concepts to merge. Calling `merge_candidate_groups`
without the registry keeps the unfiltered list. A dropped pair is already
ruled on, so dropping it retires nothing and classifies nothing; each remaining
group is printed with its names, values, modules, and whether its modules span
both runtimes, which is the shape of a registry gap. Single-module string
constants are counted, not listed.

Values are additive. Removing a value, a field, an owner, or a relation is a
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -523,9 +523,14 @@ external_input | compatibility_only | unknown
报告不入库;它每行一条地列出 Python 枚举、闭集、`Literal` 别名、
TypeScript `as const` 数组,以及拆为跨运行时孪生、同运行时分叉、冲突值、多值
孪生与多值分叉四类的重复定义。每个多值冲突都带上全部定义模块及其值集,因此
可评审的是分叉本身而不只是计数。消费者计数由 `--report` 打印,合并候选组通过 `merge_candidate_groups` 获取,
所有清单输出均不提交;合并候选是建议性的,因为值集
相同并不能证明是同一个概念。单模块的字符串常量只计数,不列出。
可评审的是分叉本身而不只是计数。消费者计数与合并候选组都由 `--report` 打印,
`merge_candidate_groups` 返回这些组,所有清单输出均不提交;合并候选是建议性的,因为值集
相同并不能证明是同一个概念。打印的列表会剔除那些名字恰好等于某个已注册词表自身
owner 符号集合的组:`EffectiveAction` 与 `EFFECTIVE_ACTIONS` 是同一个已注册概念在两个
运行时的两种拼法,不是两个待合并的概念。不传注册表调用 `merge_candidate_groups`
仍可得到未过滤列表。被剔除的对已有定论,因此剔除既不退休任何东西也不做任何分类;
留下的每一组都带上名字、值、模块,以及模块是否横跨两个运行时——后者正是注册表
缺口的形状。单模块的字符串常量只计数,不列出。

值是只增的。删除一个值、字段、owner 或关系属于 schema 缩减,遵循 `AGENTS.md`
规则:枚举受影响表面、调研生产者与读者、在同一 diff 中调低下限、记录维护者
Expand Down
46 changes: 43 additions & 3 deletions loopx/semantics/inventory.py
Original file line number Diff line number Diff line change
Expand Up @@ -412,15 +412,52 @@ def one_line(value: Any) -> str:
return "\n".join(lines) + "\n"


def merge_candidate_groups(inventory: dict[str, Any]) -> list[dict[str, Any]]:
def registered_owner_symbol_sets(registry: dict[str, Any]) -> set[frozenset[str]]:
"""Symbol sets that one registered vocabulary already binds into one concept.

A cross-runtime vocabulary names its owner twice because the runtimes spell
it differently: ``EffectiveAction`` in Python and ``EFFECTIVE_ACTIONS`` in
TypeScript are the registered owners of ``effective_action``. The registry
entry is the decision that the two spellings are one vocabulary, so their
equal value sets carry no review question. Only sets with more than one
distinct symbol are returned; a vocabulary with a single owner, or with the
same symbol on both sides, explains nothing.
"""
sets: set[frozenset[str]] = set()
for vocabulary in registry["vocabularies"].values():
symbols = frozenset(
owner.split("::")[-1] for owner in (vocabulary.get("owners") or {}).values() if owner
)
if len(symbols) > 1:
sets.add(symbols)
return sets


def merge_candidate_groups(
inventory: dict[str, Any], registry: dict[str, Any] | None = None
) -> list[dict[str, Any]]:
"""Advisory: distinct names carrying an identical multi-value set.

Value-set equality is a candidate signal, not proof of one concept:
``CONFIDENCE_LEVELS`` and ``EDGE_CASE_COMPLEXITIES`` share ``high/low/medium``
while meaning different things. Each group is for review, never auto-merged,
and is printed on demand rather than committed so it cannot be mistaken for a
ratified decision.

``registry`` drops the groups whose names are exactly the owner symbols of
one registered vocabulary. Those are a naming convention, not duplication:
the registry has already ruled the two spellings one concept, and leaving
them in the list buries the groups that nobody has ruled on. Passing
``None`` keeps the unfiltered list, which is what an audit of the raw
value-set collisions wants. Filtering removes noise from a review list; it
retires nothing and settles no group that stays.

``cross_runtime`` marks a group whose modules span both runtimes. Such a
group is a registry gap by construction, because a value set living in
Python and TypeScript with no vocabulary binding them is exactly what the
registry exists to record.
"""
explained = registered_owner_symbol_sets(registry) if registry is not None else set()
by_values: dict[tuple[str, ...], list[dict[str, Any]]] = defaultdict(list)
for section in ("python_enums", "python_closed_sets", "python_literal_aliases", "typescript_const_arrays"):
for entry in inventory[section]:
Expand All @@ -429,13 +466,16 @@ def merge_candidate_groups(inventory: dict[str, Any]) -> list[dict[str, Any]]:
groups: list[dict[str, Any]] = []
for values, entries in by_values.items():
names = sorted({entry["name"] for entry in entries})
if len(names) < 2:
if len(names) < 2 or frozenset(names) in explained:
continue
modules = sorted({entry["module"] for entry in entries})
groups.append(
{
"names": names,
"values": list(values),
"modules": sorted({entry["module"] for entry in entries}),
"modules": modules,
"cross_runtime": any(module.endswith(".py") for module in modules)
and any(module.endswith(".ts") for module in modules),
}
)
return sorted(groups, key=lambda group: (-len(group["names"]), group["names"]))
Expand Down
39 changes: 37 additions & 2 deletions scripts/generate_semantic_inventory.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
uv run python scripts/generate_semantic_inventory.py # JSON to stdout, no writes
uv run python scripts/generate_semantic_inventory.py --output .local/inventory.json
uv run python scripts/generate_semantic_inventory.py --output .local/inventory.json --check
uv run python scripts/generate_semantic_inventory.py --report # advisory consumer ranking
uv run python scripts/generate_semantic_inventory.py --report # advisory consumer ranking + merge candidates
"""

from __future__ import annotations
Expand All @@ -24,14 +24,47 @@
consumer_ranking,
divergent_value_sets,
load_sources,
merge_candidate_groups,
render_inventory,
)

REGISTRY_RELATIVE = "loopx/semantics/vocabulary_v0.json"


def print_merge_candidates(inventory: dict) -> None:
"""Print the merge candidates the registry does not already explain.

Advisory, like the ranking above it: an equal value set is a question for a
reviewer, never an auto-merge. Groups that are one registered vocabulary's
own Python and TypeScript owner symbols are dropped because the registry
has already ruled them one concept; hiding them retires nothing and
classifies nothing, it only leaves the unruled groups readable. A group
whose modules span both runtimes is a registry gap: the same value set
lives in two runtimes with no vocabulary binding them.
"""
registry = json.loads((ROOT / REGISTRY_RELATIVE).read_text(encoding="utf-8"))
groups = merge_candidate_groups(inventory, registry)
raw = len(merge_candidate_groups(inventory))
print(
f"merge candidates (advisory, never auto-merged): {len(groups)} to review, "
f"{raw - len(groups)} explained by a registered vocabulary's own owner symbols, "
f"{raw} raw groups"
)
for group in sorted(groups, key=lambda item: (not item["cross_runtime"], item["names"])):
scope = "cross-runtime" if group["cross_runtime"] else "python-only"
print(f" [{scope}] {', '.join(group['names'])}")
print(f" values: {', '.join(group['values'])}")
print(f" modules: {', '.join(group['modules'])}")


def main() -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
destination = parser.add_mutually_exclusive_group()
destination.add_argument("--output", type=Path, help="write an optional report to this path instead of stdout")
destination.add_argument("--report", action="store_true", help="print the advisory consumer ranking")
destination.add_argument(
"--report", action="store_true",
help="print the advisory consumer ranking and the merge candidates the registry does not explain",
)
parser.add_argument("--check", action="store_true", help="compare an explicit --output report without writing")
parser.add_argument("--top", type=int, default=25, help="rows to print with --report")
args = parser.parse_args()
Expand All @@ -52,6 +85,8 @@ def main() -> int:
print("value_sets name definition_modules")
for row in divergent:
print(f"{row['value_sets']:>10} {row['name']} {', '.join(row['definition_modules'])}")
print()
print_merge_candidates(inventory)
return 0
if args.output is None:
print(content, end="")
Expand Down
118 changes: 118 additions & 0 deletions tests/architecture/test_semantic_inventory.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,9 @@
SourceFile,
build_inventory,
load_sources,
merge_candidate_groups,
python_facts,
registered_owner_symbol_sets,
render_inventory,
)

Expand Down Expand Up @@ -180,6 +182,122 @@ def test_module_local_convention_names_stay_out_of_semantic_budgets(collision_re
assert summary["multi_value_twins"] == 1


@pytest.fixture
def merge_candidate_repo(tmp_path: Path) -> Path:
"""Three name pairs carry one value set each; only one pair is registered.

``Kind``/``KINDS`` are the two owners of a single registered vocabulary, so
their equal value set is a naming convention across runtimes rather than
duplication. ``ALPHA_STAGES``/``MIRROR_STAGES`` is unregistered and lives
only in Python; ``OTHER_SIDES``/``ZED_SIDES`` is unregistered and spans both
runtimes, which is the registry-gap shape. ``MIRROR_STAGES`` is a registered
owner with no TypeScript counterpart, so its vocabulary explains no pair.
"""
_write(
tmp_path,
"loopx/a.py",
'from enum import Enum\n'
'class Kind(str, Enum):\n ONE = "one"\n TWO = "two"\n'
'ALPHA_STAGES = ("draft", "final")\n',
)
_write(tmp_path, "loopx/b.ts", 'export const KINDS = ["one", "two"] as const;\n')
_write(
tmp_path,
"loopx/c.py",
'MIRROR_STAGES = ("draft", "final")\nOTHER_SIDES = ("left", "right")\n',
)
_write(tmp_path, "loopx/d.ts", 'export const ZED_SIDES = ["left", "right"] as const;\n')
_write(
tmp_path,
"loopx/semantics/vocabulary_v0.json",
json.dumps(
{
"vocabularies": {
"kind": {
"owners": {
"python": "loopx/a.py::Kind",
"typescript": "loopx/b.ts::KINDS",
}
},
"mirror_stages": {
"owners": {"python": "loopx/c.py::MIRROR_STAGES", "typescript": None}
},
}
}
),
)
subprocess.run(["git", "init", "-q", str(tmp_path)], check=True)
subprocess.run(
["git", "-C", str(tmp_path), "add", "loopx/a.py", "loopx/b.ts", "loopx/c.py", "loopx/d.ts"],
check=True,
)
return tmp_path


def _registry(repo: Path) -> dict:
return json.loads((repo / "loopx/semantics/vocabulary_v0.json").read_text(encoding="utf-8"))


def test_one_vocabulary_owning_two_spellings_explains_no_merge_candidate(
merge_candidate_repo: Path,
) -> None:
"""A registered owner pair is a naming convention, not a merge candidate.

Without the registry the advisory list mixed each cross-runtime vocabulary's
own two owner symbols in with the groups nobody has ruled on, so the list
read as duplication it was not. Filtering only hides the settled pairs; it
retires nothing and classifies none of the groups that stay.
"""
inventory = build_inventory(merge_candidate_repo)
unfiltered = {tuple(group["names"]) for group in merge_candidate_groups(inventory)}
assert ("KINDS", "Kind") in unfiltered, "the unfiltered audit still sees every value-set collision"
filtered = merge_candidate_groups(inventory, _registry(merge_candidate_repo))
assert {tuple(group["names"]) for group in filtered} == {
("ALPHA_STAGES", "MIRROR_STAGES"),
("OTHER_SIDES", "ZED_SIDES"),
}, "an unregistered pair with the same value set is still reported"
assert registered_owner_symbol_sets(_registry(merge_candidate_repo)) == {
frozenset({"Kind", "KINDS"})
}, "a vocabulary with one owner symbol explains nothing"


def test_merge_candidates_mark_the_pairs_that_span_both_runtimes(
merge_candidate_repo: Path,
) -> None:
"""Cross-runtime is the registry gap; Python-only is a human question."""
groups = merge_candidate_groups(
build_inventory(merge_candidate_repo), _registry(merge_candidate_repo)
)
assert {tuple(group["names"]): group["cross_runtime"] for group in groups} == {
("ALPHA_STAGES", "MIRROR_STAGES"): False,
("OTHER_SIDES", "ZED_SIDES"): True,
}


def test_report_prints_the_unexplained_merge_candidates_in_a_stable_order(
merge_candidate_repo: Path, monkeypatch, capsys
) -> None:
"""``--report`` is where a reviewer sees the list, so it must not churn."""
from scripts import generate_semantic_inventory as generator

monkeypatch.setattr(generator, "ROOT", merge_candidate_repo)
monkeypatch.setattr(sys, "argv", ["generate_semantic_inventory", "--report"])
assert generator.main() == 0
first = capsys.readouterr().out
assert generator.main() == 0
assert capsys.readouterr().out == first, "the same tree must print the same advisory list"
listing = first.split("merge candidates", 1)[1]
assert "2 to review, 1 explained by a registered vocabulary's own owner symbols, 3 raw groups" in listing
assert "Kind" not in listing, "the explained owner pair is not printed"
assert " [cross-runtime] OTHER_SIDES, ZED_SIDES" in listing
assert " values: left, right" in listing
assert " modules: loopx/c.py, loopx/d.ts" in listing
assert " [python-only] ALPHA_STAGES, MIRROR_STAGES" in listing
assert listing.index("[cross-runtime]") < listing.index("[python-only]"), (
"registry gaps sort ahead of the Python-only groups regardless of name order"
)


def test_render_is_deterministic_valid_json(repo: Path) -> None:
inventory = build_inventory(repo)
rendered = render_inventory(inventory)
Expand Down