diff --git a/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.md b/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.md index b1cbbb9925..8982c8a18c 100644 --- a/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.md +++ b/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.md @@ -643,10 +643,19 @@ lists Python enums, closed sets, `Literal` aliases, TypeScript `as const` arrays, and duplicate definitions split into cross-runtime twins, same-runtime forks, conflicting values, and multi-value twins and forks, one entry per line. Every multi-value collision carries each defining module and its value set, so -the divergence itself is reviewable rather than only its count. Consumer counts are printed by `--report`; merge-candidate groups are available -through `merge_candidate_groups`; all inventory output is uncommitted. +the divergence itself is reviewable rather than only its count. Consumer counts +and merge-candidate groups are both printed by `--report`; +`merge_candidate_groups` returns the groups; all inventory output is +uncommitted. Merge candidates are advisory -because an equal value set is not proof of one concept. Single-module string +because an equal value set is not proof of one concept. The printed list drops +the groups whose names are exactly one registered vocabulary's own owner +symbols: `EffectiveAction` and `EFFECTIVE_ACTIONS` are two runtimes spelling one +registered concept, not two concepts to merge. Calling `merge_candidate_groups` +without the registry keeps the unfiltered list. A dropped pair is already +ruled on, so dropping it retires nothing and classifies nothing; each remaining +group is printed with its names, values, modules, and whether its modules span +both runtimes, which is the shape of a registry gap. Single-module string constants are counted, not listed. Values are additive. Removing a value, a field, an owner, or a relation is a diff --git a/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.zh-CN.md b/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.zh-CN.md index 5d09b40000..defcb781bd 100644 --- a/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.zh-CN.md +++ b/docs/architecture/rfcs/semantic-vocabulary-convergence-v0.zh-CN.md @@ -523,9 +523,14 @@ external_input | compatibility_only | unknown 报告不入库;它每行一条地列出 Python 枚举、闭集、`Literal` 别名、 TypeScript `as const` 数组,以及拆为跨运行时孪生、同运行时分叉、冲突值、多值 孪生与多值分叉四类的重复定义。每个多值冲突都带上全部定义模块及其值集,因此 -可评审的是分叉本身而不只是计数。消费者计数由 `--report` 打印,合并候选组通过 `merge_candidate_groups` 获取, -所有清单输出均不提交;合并候选是建议性的,因为值集 -相同并不能证明是同一个概念。单模块的字符串常量只计数,不列出。 +可评审的是分叉本身而不只是计数。消费者计数与合并候选组都由 `--report` 打印, +`merge_candidate_groups` 返回这些组,所有清单输出均不提交;合并候选是建议性的,因为值集 +相同并不能证明是同一个概念。打印的列表会剔除那些名字恰好等于某个已注册词表自身 +owner 符号集合的组:`EffectiveAction` 与 `EFFECTIVE_ACTIONS` 是同一个已注册概念在两个 +运行时的两种拼法,不是两个待合并的概念。不传注册表调用 `merge_candidate_groups` +仍可得到未过滤列表。被剔除的对已有定论,因此剔除既不退休任何东西也不做任何分类; +留下的每一组都带上名字、值、模块,以及模块是否横跨两个运行时——后者正是注册表 +缺口的形状。单模块的字符串常量只计数,不列出。 值是只增的。删除一个值、字段、owner 或关系属于 schema 缩减,遵循 `AGENTS.md` 规则:枚举受影响表面、调研生产者与读者、在同一 diff 中调低下限、记录维护者 diff --git a/loopx/semantics/inventory.py b/loopx/semantics/inventory.py index cea9c30bf5..b83a6cac08 100644 --- a/loopx/semantics/inventory.py +++ b/loopx/semantics/inventory.py @@ -412,7 +412,30 @@ def one_line(value: Any) -> str: return "\n".join(lines) + "\n" -def merge_candidate_groups(inventory: dict[str, Any]) -> list[dict[str, Any]]: +def registered_owner_symbol_sets(registry: dict[str, Any]) -> set[frozenset[str]]: + """Symbol sets that one registered vocabulary already binds into one concept. + + A cross-runtime vocabulary names its owner twice because the runtimes spell + it differently: ``EffectiveAction`` in Python and ``EFFECTIVE_ACTIONS`` in + TypeScript are the registered owners of ``effective_action``. The registry + entry is the decision that the two spellings are one vocabulary, so their + equal value sets carry no review question. Only sets with more than one + distinct symbol are returned; a vocabulary with a single owner, or with the + same symbol on both sides, explains nothing. + """ + sets: set[frozenset[str]] = set() + for vocabulary in registry["vocabularies"].values(): + symbols = frozenset( + owner.split("::")[-1] for owner in (vocabulary.get("owners") or {}).values() if owner + ) + if len(symbols) > 1: + sets.add(symbols) + return sets + + +def merge_candidate_groups( + inventory: dict[str, Any], registry: dict[str, Any] | None = None +) -> list[dict[str, Any]]: """Advisory: distinct names carrying an identical multi-value set. Value-set equality is a candidate signal, not proof of one concept: @@ -420,7 +443,21 @@ def merge_candidate_groups(inventory: dict[str, Any]) -> list[dict[str, Any]]: while meaning different things. Each group is for review, never auto-merged, and is printed on demand rather than committed so it cannot be mistaken for a ratified decision. + + ``registry`` drops the groups whose names are exactly the owner symbols of + one registered vocabulary. Those are a naming convention, not duplication: + the registry has already ruled the two spellings one concept, and leaving + them in the list buries the groups that nobody has ruled on. Passing + ``None`` keeps the unfiltered list, which is what an audit of the raw + value-set collisions wants. Filtering removes noise from a review list; it + retires nothing and settles no group that stays. + + ``cross_runtime`` marks a group whose modules span both runtimes. Such a + group is a registry gap by construction, because a value set living in + Python and TypeScript with no vocabulary binding them is exactly what the + registry exists to record. """ + explained = registered_owner_symbol_sets(registry) if registry is not None else set() by_values: dict[tuple[str, ...], list[dict[str, Any]]] = defaultdict(list) for section in ("python_enums", "python_closed_sets", "python_literal_aliases", "typescript_const_arrays"): for entry in inventory[section]: @@ -429,13 +466,16 @@ def merge_candidate_groups(inventory: dict[str, Any]) -> list[dict[str, Any]]: groups: list[dict[str, Any]] = [] for values, entries in by_values.items(): names = sorted({entry["name"] for entry in entries}) - if len(names) < 2: + if len(names) < 2 or frozenset(names) in explained: continue + modules = sorted({entry["module"] for entry in entries}) groups.append( { "names": names, "values": list(values), - "modules": sorted({entry["module"] for entry in entries}), + "modules": modules, + "cross_runtime": any(module.endswith(".py") for module in modules) + and any(module.endswith(".ts") for module in modules), } ) return sorted(groups, key=lambda group: (-len(group["names"]), group["names"])) diff --git a/scripts/generate_semantic_inventory.py b/scripts/generate_semantic_inventory.py index b71a5f28af..0ba8593f28 100755 --- a/scripts/generate_semantic_inventory.py +++ b/scripts/generate_semantic_inventory.py @@ -5,7 +5,7 @@ uv run python scripts/generate_semantic_inventory.py # JSON to stdout, no writes uv run python scripts/generate_semantic_inventory.py --output .local/inventory.json uv run python scripts/generate_semantic_inventory.py --output .local/inventory.json --check - uv run python scripts/generate_semantic_inventory.py --report # advisory consumer ranking + uv run python scripts/generate_semantic_inventory.py --report # advisory consumer ranking + merge candidates """ from __future__ import annotations @@ -24,14 +24,47 @@ consumer_ranking, divergent_value_sets, load_sources, + merge_candidate_groups, render_inventory, ) +REGISTRY_RELATIVE = "loopx/semantics/vocabulary_v0.json" + + +def print_merge_candidates(inventory: dict) -> None: + """Print the merge candidates the registry does not already explain. + + Advisory, like the ranking above it: an equal value set is a question for a + reviewer, never an auto-merge. Groups that are one registered vocabulary's + own Python and TypeScript owner symbols are dropped because the registry + has already ruled them one concept; hiding them retires nothing and + classifies nothing, it only leaves the unruled groups readable. A group + whose modules span both runtimes is a registry gap: the same value set + lives in two runtimes with no vocabulary binding them. + """ + registry = json.loads((ROOT / REGISTRY_RELATIVE).read_text(encoding="utf-8")) + groups = merge_candidate_groups(inventory, registry) + raw = len(merge_candidate_groups(inventory)) + print( + f"merge candidates (advisory, never auto-merged): {len(groups)} to review, " + f"{raw - len(groups)} explained by a registered vocabulary's own owner symbols, " + f"{raw} raw groups" + ) + for group in sorted(groups, key=lambda item: (not item["cross_runtime"], item["names"])): + scope = "cross-runtime" if group["cross_runtime"] else "python-only" + print(f" [{scope}] {', '.join(group['names'])}") + print(f" values: {', '.join(group['values'])}") + print(f" modules: {', '.join(group['modules'])}") + + def main() -> int: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) destination = parser.add_mutually_exclusive_group() destination.add_argument("--output", type=Path, help="write an optional report to this path instead of stdout") - destination.add_argument("--report", action="store_true", help="print the advisory consumer ranking") + destination.add_argument( + "--report", action="store_true", + help="print the advisory consumer ranking and the merge candidates the registry does not explain", + ) parser.add_argument("--check", action="store_true", help="compare an explicit --output report without writing") parser.add_argument("--top", type=int, default=25, help="rows to print with --report") args = parser.parse_args() @@ -52,6 +85,8 @@ def main() -> int: print("value_sets name definition_modules") for row in divergent: print(f"{row['value_sets']:>10} {row['name']} {', '.join(row['definition_modules'])}") + print() + print_merge_candidates(inventory) return 0 if args.output is None: print(content, end="") diff --git a/tests/architecture/test_semantic_inventory.py b/tests/architecture/test_semantic_inventory.py index 2f2c478775..96e5151543 100644 --- a/tests/architecture/test_semantic_inventory.py +++ b/tests/architecture/test_semantic_inventory.py @@ -21,7 +21,9 @@ SourceFile, build_inventory, load_sources, + merge_candidate_groups, python_facts, + registered_owner_symbol_sets, render_inventory, ) @@ -180,6 +182,122 @@ def test_module_local_convention_names_stay_out_of_semantic_budgets(collision_re assert summary["multi_value_twins"] == 1 +@pytest.fixture +def merge_candidate_repo(tmp_path: Path) -> Path: + """Three name pairs carry one value set each; only one pair is registered. + + ``Kind``/``KINDS`` are the two owners of a single registered vocabulary, so + their equal value set is a naming convention across runtimes rather than + duplication. ``ALPHA_STAGES``/``MIRROR_STAGES`` is unregistered and lives + only in Python; ``OTHER_SIDES``/``ZED_SIDES`` is unregistered and spans both + runtimes, which is the registry-gap shape. ``MIRROR_STAGES`` is a registered + owner with no TypeScript counterpart, so its vocabulary explains no pair. + """ + _write( + tmp_path, + "loopx/a.py", + 'from enum import Enum\n' + 'class Kind(str, Enum):\n ONE = "one"\n TWO = "two"\n' + 'ALPHA_STAGES = ("draft", "final")\n', + ) + _write(tmp_path, "loopx/b.ts", 'export const KINDS = ["one", "two"] as const;\n') + _write( + tmp_path, + "loopx/c.py", + 'MIRROR_STAGES = ("draft", "final")\nOTHER_SIDES = ("left", "right")\n', + ) + _write(tmp_path, "loopx/d.ts", 'export const ZED_SIDES = ["left", "right"] as const;\n') + _write( + tmp_path, + "loopx/semantics/vocabulary_v0.json", + json.dumps( + { + "vocabularies": { + "kind": { + "owners": { + "python": "loopx/a.py::Kind", + "typescript": "loopx/b.ts::KINDS", + } + }, + "mirror_stages": { + "owners": {"python": "loopx/c.py::MIRROR_STAGES", "typescript": None} + }, + } + } + ), + ) + subprocess.run(["git", "init", "-q", str(tmp_path)], check=True) + subprocess.run( + ["git", "-C", str(tmp_path), "add", "loopx/a.py", "loopx/b.ts", "loopx/c.py", "loopx/d.ts"], + check=True, + ) + return tmp_path + + +def _registry(repo: Path) -> dict: + return json.loads((repo / "loopx/semantics/vocabulary_v0.json").read_text(encoding="utf-8")) + + +def test_one_vocabulary_owning_two_spellings_explains_no_merge_candidate( + merge_candidate_repo: Path, +) -> None: + """A registered owner pair is a naming convention, not a merge candidate. + + Without the registry the advisory list mixed each cross-runtime vocabulary's + own two owner symbols in with the groups nobody has ruled on, so the list + read as duplication it was not. Filtering only hides the settled pairs; it + retires nothing and classifies none of the groups that stay. + """ + inventory = build_inventory(merge_candidate_repo) + unfiltered = {tuple(group["names"]) for group in merge_candidate_groups(inventory)} + assert ("KINDS", "Kind") in unfiltered, "the unfiltered audit still sees every value-set collision" + filtered = merge_candidate_groups(inventory, _registry(merge_candidate_repo)) + assert {tuple(group["names"]) for group in filtered} == { + ("ALPHA_STAGES", "MIRROR_STAGES"), + ("OTHER_SIDES", "ZED_SIDES"), + }, "an unregistered pair with the same value set is still reported" + assert registered_owner_symbol_sets(_registry(merge_candidate_repo)) == { + frozenset({"Kind", "KINDS"}) + }, "a vocabulary with one owner symbol explains nothing" + + +def test_merge_candidates_mark_the_pairs_that_span_both_runtimes( + merge_candidate_repo: Path, +) -> None: + """Cross-runtime is the registry gap; Python-only is a human question.""" + groups = merge_candidate_groups( + build_inventory(merge_candidate_repo), _registry(merge_candidate_repo) + ) + assert {tuple(group["names"]): group["cross_runtime"] for group in groups} == { + ("ALPHA_STAGES", "MIRROR_STAGES"): False, + ("OTHER_SIDES", "ZED_SIDES"): True, + } + + +def test_report_prints_the_unexplained_merge_candidates_in_a_stable_order( + merge_candidate_repo: Path, monkeypatch, capsys +) -> None: + """``--report`` is where a reviewer sees the list, so it must not churn.""" + from scripts import generate_semantic_inventory as generator + + monkeypatch.setattr(generator, "ROOT", merge_candidate_repo) + monkeypatch.setattr(sys, "argv", ["generate_semantic_inventory", "--report"]) + assert generator.main() == 0 + first = capsys.readouterr().out + assert generator.main() == 0 + assert capsys.readouterr().out == first, "the same tree must print the same advisory list" + listing = first.split("merge candidates", 1)[1] + assert "2 to review, 1 explained by a registered vocabulary's own owner symbols, 3 raw groups" in listing + assert "Kind" not in listing, "the explained owner pair is not printed" + assert " [cross-runtime] OTHER_SIDES, ZED_SIDES" in listing + assert " values: left, right" in listing + assert " modules: loopx/c.py, loopx/d.ts" in listing + assert " [python-only] ALPHA_STAGES, MIRROR_STAGES" in listing + assert listing.index("[cross-runtime]") < listing.index("[python-only]"), ( + "registry gaps sort ahead of the Python-only groups regardless of name order" + ) + + def test_render_is_deterministic_valid_json(repo: Path) -> None: inventory = build_inventory(repo) rendered = render_inventory(inventory)