diff --git a/docs/architecture/product-boundary.md b/docs/architecture/product-boundary.md index 18146945..3ea015cb 100644 --- a/docs/architecture/product-boundary.md +++ b/docs/architecture/product-boundary.md @@ -88,3 +88,74 @@ and intentionally obfuscated runtime code are outside its threat model. The gate covers normal source imports and the explicitly tested acquisition, aliasing, assignment, and reflection forms; release review and package tests remain responsible for hostile-code scenarios. + +## Function-level guards + +The ownership ledger classifies whole files, so it cannot see application logic +inside a core file. A layer leak guard makes that logic visible in review, and a +size record makes core growth visible at each release. + +### Layer leak guard + +`scripts/layer_leak_guard.py` AST-scans every product module under `src/hwpx` +(not `data/`, not `_moved_modules.py`) for two signals: + +- `hangul-regex`: the pattern given to `re.compile`, `match`, `search`, + `fullmatch`, `sub`, `findall` or `finditer` contains Hangul, as a literal or + as a module-level string constant passed by name; +- `plan-schema-key`: a subscript or `.get()` with the key `"sections"` or + `"blocks"`, the shape of the automation layer's document plan. + +Every existing hit is listed in `tests/data/layer_leak_allowlist.json` by file, +qualified name (a function, a method, or the module-level constant a regex is +assigned to), signal and exact count, with a classification and a reason: + +- `format-vocabulary`: Hancom format vocabulary any HWPX user needs, such as + built-in style names or field-type tokens. It stays. +- `known-leak`: genre or policy logic left over from the layer audit. It is + scheduled to move to `python-hwpx-automation` in 7.0. + +`undetected` lists known leaks that neither signal sees (a caption pattern +without Hangul, Roman-numeral headings). They are not counted, but each named +function or constant must still exist, so the entry leaves the list when the +code moves. + +A new hit fails. Before allowlisting it, apply the feature-placement test of +the layer-boundary guardrails (section 4), in order: + +1. Is it reusable by any HWPX user without a particular genre, institution or + policy? Core. +2. Is it a deterministic workflow or policy built from core primitives? + `python-hwpx-automation`. +3. Does it judge user intent, genre or ambiguity, or choose tools? The plugin. + +Touching XML does not make a feature core. Only answer 1 belongs in the +allowlist, as `format-vocabulary` with a reason. A count that drops below its +entry also fails: lower or remove the entry in the same change. + + python scripts/layer_leak_guard.py --list # every live hit + python scripts/layer_leak_guard.py # check (tests/test_layer_leak_guard.py) + +### Size history and import breadth + +`docs/size-history.json` records, per release, the physical lines of every +`.py` file under `src/hwpx`, the same per top-level subpackage (`"."` holds the +top-level modules), how many `hwpx` modules a bare `import hwpx` loads in a +fresh `python -I` interpreter with only the measured tree's `src` on the path, +and that import's median time over three runs on the recording machine. It is +written during release prep (see `docs/release-runbook.md`), not per pull +request: an exact line lock would conflict between every pair of parallel +branches. `tests/test_size_ratchet.py` only checks that the file is well formed +and that its newest entry is not newer than the `pyproject.toml` version. + +The module count is also an upper-bound ratchet in +`tests/data/import_breadth.json`. A change that makes `import hwpx` load more +modules fails; import new modules lazily where they are used, or raise the +bound in the same change and say why. A lower count passes with a warning; +tighten the bound with `--lower-bound` when convenient. Import time is never +gated. + + python scripts/size_ratchet.py # working tree vs last release, per package + python scripts/size_ratchet.py --record 6.8.0 # release prep: append the working tree + python scripts/size_ratchet.py --record 6.0.0 --ref v6.0.0 # measure a tag via git archive + python scripts/size_ratchet.py --lower-bound # tighten the module bound diff --git a/docs/release-runbook.md b/docs/release-runbook.md index 943702ad..82836c77 100644 --- a/docs/release-runbook.md +++ b/docs/release-runbook.md @@ -92,7 +92,15 @@ for the preserved-failed-tag record of trains that skipped one). 8. `python -m hwpx.capabilities --verify`. 9. Full test suite: `pytest -q --cov=hwpx --cov-report=term-missing --cov-fail-under=80`. -10. Local build + install smoke: `python -m build`, `twine check dist/*`, +10. **Record the size history.** `python scripts/size_ratchet.py` prints the + working tree against the last recorded release, per subpackage, plus the + `import hwpx` module count and time. Then `python scripts/size_ratchet.py + --record ` appends this release to `docs/size-history.json`; + commit it with the release. When core grew noticeably (a subpackage by a + large share, or more modules loaded by `import hwpx`), say what grew and why + in the release notes. If the module count dropped, tighten + `tests/data/import_breadth.json` with `--lower-bound`. +11. Local build + install smoke: `python -m build`, `twine check dist/*`, then install the built wheel into a throwaway venv (`uv venv` / `uv pip install`) and exercise a real round trip (author something, save, reopen) plus an import of anything the train specifically changed — this is the diff --git a/docs/size-history.json b/docs/size-history.json new file mode 100644 index 00000000..76bd08f4 --- /dev/null +++ b/docs/size-history.json @@ -0,0 +1,89 @@ +{ + "schemaVersion": "python-hwpx.size-history/v1", + "releases": [ + { + "version": "5.0.1", + "commit": "c73bf04", + "totalLines": 36244, + "packages": { + ".": 6286, + "_document": 2685, + "equation": 771, + "form_fit": 1459, + "ingest": 383, + "layout": 541, + "opc": 1523, + "oxml": 11668, + "quality": 1173, + "tools": 9755 + }, + "importedModules": 70, + "importMs": 58 + }, + { + "version": "6.0.0", + "commit": "ee0e903", + "totalLines": 45311, + "packages": { + ".": 6333, + "_document": 7793, + "equation": 1248, + "form_fit": 1459, + "ingest": 383, + "layout": 555, + "objects": 717, + "opc": 1536, + "oxml": 13390, + "plan": 900, + "quality": 1222, + "tools": 9775 + }, + "importedModules": 94, + "importMs": 66 + }, + { + "version": "6.6.0", + "commit": "9ffeccc", + "totalLines": 72464, + "packages": { + ".": 7221, + "_document": 10312, + "equation": 1699, + "form_fit": 1740, + "hwp5": 11154, + "ingest": 394, + "layout": 704, + "objects": 976, + "opc": 1962, + "oxml": 21853, + "plan": 900, + "quality": 1224, + "tools": 12325 + }, + "importedModules": 114, + "importMs": 79 + }, + { + "version": "6.7.0", + "commit": "9e0d02b", + "totalLines": 78053, + "packages": { + ".": 7431, + "_document": 10861, + "equation": 1827, + "form_fit": 2393, + "hwp5": 11154, + "ingest": 394, + "layout": 3150, + "objects": 990, + "opc": 2033, + "oxml": 22852, + "plan": 900, + "quality": 1227, + "tools": 12841 + }, + "importedModules": 117, + "importMs": 77 + } + ] +} diff --git a/scripts/layer_leak_guard.py b/scripts/layer_leak_guard.py new file mode 100644 index 00000000..77bd51fa --- /dev/null +++ b/scripts/layer_leak_guard.py @@ -0,0 +1,316 @@ +#!/usr/bin/env python3 +# SPDX-License-Identifier: Apache-2.0 +"""Find genre or policy logic hiding inside core-owned modules. + +The ownership ledger (``docs/architecture/module-ownership.json``) classifies +whole files, so a core file can still carry application logic: a regex that +matches Korean official-document headings, an eval-plan caption pattern, or a +parser for the automation layer's document-plan schema. A path rule cannot +see any of these. This guard looks inside the functions. + +It AST-scans every product module under ``src/hwpx`` (not ``data/`` and not +``_moved_modules.py``) for two signals: + +- ``hangul-regex``: the pattern argument of ``re.compile``/``match``/ + ``search``/``fullmatch``/``sub``/``findall``/``finditer`` contains Hangul, + either as a literal or as a module-level string constant passed by name. +- ``plan-schema-key``: a subscript or ``.get()`` with the string key + ``"sections"`` or ``"blocks"``, the shape of the automation layer's + document plan. + +Existing hits are listed, with a reason, in +``tests/data/layer_leak_allowlist.json``, keyed by file, qualified function +name and signal, with an exact count. Hancom format vocabulary (field-type +tokens, numbering-format labels) is allowed there on purpose; the leftovers +of the layer audit are marked as known leaks scheduled to move in 7.0. + +A hit that is not in the allowlist fails. So does an allowlist count that no +longer matches, so a fixed leak has to leave the list in the same change. + + python scripts/layer_leak_guard.py # check against the allowlist + python scripts/layer_leak_guard.py --list # print every live hit +""" + +from __future__ import annotations + +import argparse +import ast +import json +import pathlib +import sys +from collections import Counter +from dataclasses import dataclass + +ROOT = pathlib.Path(__file__).resolve().parent.parent +SRC = ROOT / "src" / "hwpx" +ALLOWLIST = ROOT / "tests" / "data" / "layer_leak_allowlist.json" + +#: Paths under ``src/hwpx`` that are not product modules. +EXCLUDED = ("data/", "_moved_modules.py") + +REGEX_FUNCTIONS = frozenset( + {"compile", "match", "search", "fullmatch", "sub", "findall", "finditer"} +) +PLAN_SCHEMA_KEYS = frozenset({"sections", "blocks"}) + +GUIDANCE = ( + "Genre, institution or policy logic belongs to python-hwpx-automation or the " + "plugin, not core: apply the feature-placement test in the layer-boundary " + "guardrails (section 4, summarized in docs/architecture/product-boundary.md, " + "'Function-level guards'). If the hit is Hancom format vocabulary that any " + "HWPX user needs, add it to tests/data/layer_leak_allowlist.json with a reason " + "in the same change." +) + + +def has_hangul(text: str) -> bool: + """Hangul syllables, jamo, or compatibility jamo.""" + + return any( + 0xAC00 <= ord(char) <= 0xD7A3 + or 0x1100 <= ord(char) <= 0x11FF + or 0x3130 <= ord(char) <= 0x318F + for char in text + ) + + +@dataclass(frozen=True) +class Hit: + file: str + qualname: str + kind: str + detail: str + line: int + + @property + def key(self) -> tuple[str, str, str]: + return (self.file, self.qualname, self.kind) + + +def _string_value(node: ast.expr) -> str | None: + """The text of a string constant, an implicit concatenation or ``+`` of + constants, or an f-string's literal parts.""" + + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return node.value + if isinstance(node, ast.JoinedStr): + return "".join( + part.value + for part in node.values + if isinstance(part, ast.Constant) and isinstance(part.value, str) + ) + if isinstance(node, ast.BinOp) and isinstance(node.op, ast.Add): + left, right = _string_value(node.left), _string_value(node.right) + if left is not None and right is not None: + return left + right + return None + + +class _Scanner(ast.NodeVisitor): + def __init__(self, relative: str, tree: ast.Module) -> None: + self.relative = relative + self.stack: list[str] = [] + self.hits: list[Hit] = [] + self.re_names = {"re"} + self.direct_functions: set[str] = set() + self.constants: dict[str, str] = {} + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + if alias.name == "re": + self.re_names.add(alias.asname or "re") + elif isinstance(node, ast.ImportFrom) and node.module == "re": + for alias in node.names: + if alias.name in REGEX_FUNCTIONS: + self.direct_functions.add(alias.asname or alias.name) + for statement in tree.body: + targets: list[ast.expr] = [] + value: ast.expr | None = None + if isinstance(statement, ast.Assign): + targets, value = statement.targets, statement.value + elif isinstance(statement, ast.AnnAssign) and statement.value is not None: + targets, value = [statement.target], statement.value + text = _string_value(value) if value is not None else None + if text is None: + continue + for target in targets: + if isinstance(target, ast.Name): + self.constants[target.id] = text + + @property + def qualname(self) -> str: + return ".".join(self.stack) or "" + + def _add(self, kind: str, detail: str, node: ast.AST) -> None: + self.hits.append( + Hit(self.relative, self.qualname, kind, detail, getattr(node, "lineno", 0)) + ) + + def _scoped(self, node: ast.FunctionDef | ast.AsyncFunctionDef | ast.ClassDef) -> None: + self.stack.append(node.name) + self.generic_visit(node) + self.stack.pop() + + visit_FunctionDef = _scoped + visit_AsyncFunctionDef = _scoped + visit_ClassDef = _scoped + + def _assigned(self, node: ast.Assign | ast.AnnAssign) -> None: + """A module-level ``NAME = re.compile(...)`` is reported under NAME, + its qualified name, rather than one shared ```` bucket.""" + + targets = node.targets if isinstance(node, ast.Assign) else [node.target] + if not self.stack and len(targets) == 1 and isinstance(targets[0], ast.Name): + self.stack.append(targets[0].id) + self.generic_visit(node) + self.stack.pop() + else: + self.generic_visit(node) + + visit_Assign = _assigned + visit_AnnAssign = _assigned + + def _is_regex_call(self, func: ast.expr) -> bool: + if isinstance(func, ast.Attribute) and func.attr in REGEX_FUNCTIONS: + return isinstance(func.value, ast.Name) and func.value.id in self.re_names + return isinstance(func, ast.Name) and func.id in self.direct_functions + + def visit_Call(self, node: ast.Call) -> None: + if self._is_regex_call(node.func): + pattern = node.args[0] if node.args else next( + (kw.value for kw in node.keywords if kw.arg == "pattern"), None + ) + if pattern is not None: + text = _string_value(pattern) + if text is None and isinstance(pattern, ast.Name): + text = self.constants.get(pattern.id) + if text is not None and has_hangul(text): + self._add("hangul-regex", text, node) + if ( + isinstance(node.func, ast.Attribute) + and node.func.attr == "get" + and node.args + and isinstance(node.args[0], ast.Constant) + and node.args[0].value in PLAN_SCHEMA_KEYS + ): + self._add("plan-schema-key", f".get({node.args[0].value!r})", node) + self.generic_visit(node) + + def visit_Subscript(self, node: ast.Subscript) -> None: + if isinstance(node.slice, ast.Constant) and node.slice.value in PLAN_SCHEMA_KEYS: + self._add("plan-schema-key", f"[{node.slice.value!r}]", node) + self.generic_visit(node) + + +def product_modules(src: pathlib.Path = SRC) -> list[pathlib.Path]: + modules = [] + for path in sorted(src.rglob("*.py")): + relative = path.relative_to(src).as_posix() + if "__pycache__" in relative or relative.startswith(EXCLUDED): + continue + modules.append(path) + return modules + + +def scan_source(relative: str, source: str) -> list[Hit]: + tree = ast.parse(source, filename=relative) + scanner = _Scanner(relative, tree) + scanner.visit(tree) + return scanner.hits + + +def scan(src: pathlib.Path = SRC) -> list[Hit]: + hits: list[Hit] = [] + for path in product_modules(src): + relative = path.relative_to(src).as_posix() + hits.extend(scan_source(relative, path.read_text(encoding="utf-8"))) + return hits + + +def load_allowlist(path: pathlib.Path = ALLOWLIST) -> dict[tuple[str, str, str], dict]: + data = json.loads(path.read_text(encoding="utf-8")) + return {(e["file"], e["qualname"], e["kind"]): e for e in data["entries"]} + + +def problems(hits: list[Hit], allowlist: dict[tuple[str, str, str], dict]) -> list[str]: + """Every difference between the live hits and the allowlist.""" + + counts = Counter(hit.key for hit in hits) + found: list[str] = [] + for key, count in sorted(counts.items()): + entry = allowlist.get(key) + allowed = entry["count"] if entry else 0 + if count > allowed: + examples = [h for h in hits if h.key == key] + lines = ", ".join(f"line {h.line}: {h.detail[:60]!r}" for h in examples[:3]) + found.append( + f"new {key[2]} in src/hwpx/{key[0]} {key[1]} " + f"({count} found, {allowed} allowed; {lines})" + ) + for key, entry in sorted(allowlist.items()): + if counts.get(key, 0) < entry["count"]: + found.append( + f"stale allowlist entry src/hwpx/{key[0]} {key[1]} {key[2]}: " + f"{entry['count']} allowed, {counts.get(key, 0)} found — lower or " + "remove it in tests/data/layer_leak_allowlist.json" + ) + return found + + +def defined_names(source: str) -> set[str]: + """Module-level names and dotted qualnames of every def/class in *source*.""" + + names: set[str] = set() + + def walk(body: list[ast.stmt], prefix: str) -> None: + for node in body: + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): + names.add(prefix + node.name) + walk(node.body, f"{prefix}{node.name}.") + elif not prefix and isinstance(node, (ast.Assign, ast.AnnAssign)): + targets = node.targets if isinstance(node, ast.Assign) else [node.target] + names.update(t.id for t in targets if isinstance(t, ast.Name)) + + walk(ast.parse(source).body, "") + return names + + +def undetected_problems(path: pathlib.Path = ALLOWLIST, src: pathlib.Path = SRC) -> list[str]: + """Known leaks the scan cannot see are listed by name; once one moves out + of core its entry has to go too, so the list cannot rot.""" + + data = json.loads(path.read_text(encoding="utf-8")) + found = [] + for entry in data.get("undetected", []): + module = src / entry["file"] + if not module.exists() or entry["qualname"] not in defined_names( + module.read_text(encoding="utf-8") + ): + found.append( + f"undetected known leak src/hwpx/{entry['file']} {entry['qualname']} " + "no longer exists — remove it from tests/data/layer_leak_allowlist.json" + ) + return found + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--list", action="store_true", help="print every live hit") + args = parser.parse_args(argv) + + hits = scan() + if args.list: + for hit in hits: + print(f"{hit.file}:{hit.line} {hit.qualname} {hit.kind} {hit.detail[:80]!r}") + return 0 + found = problems(hits, load_allowlist()) + undetected_problems() + if found: + print("layer leak guard failed:\n " + "\n ".join(found), file=sys.stderr) + print("\n" + GUIDANCE, file=sys.stderr) + return 1 + print(f"layer leak guard ok — {len(hits)} allowlisted hit(s)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/size_ratchet.py b/scripts/size_ratchet.py new file mode 100644 index 00000000..5105d9b8 --- /dev/null +++ b/scripts/size_ratchet.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python3 +# SPDX-License-Identifier: Apache-2.0 +"""Make core growth visible: Python line counts and ``import hwpx`` breadth. + +Core went from about 36k lines (5.0.1) to 72k (6.6.0) in two months and +nothing in review showed it. Two records keep it in view without making +every pull request fight over one file: + +- ``docs/size-history.json``: one entry per release (``--record``) with + the physical lines of every ``.py`` file under ``src/hwpx``, the same per + top-level subpackage (``"."`` collects top-level modules such as + ``document.py``), how many ``hwpx`` / ``hwpx.*`` modules a bare + ``import hwpx`` loads, and that import's time. Lines are a release-notes + fact, not a pull-request gate. +- ``tests/data/import_breadth.json``: an upper bound on that module count. + ``tests/test_size_ratchet.py`` fails when ``import hwpx`` loads more + modules than the bound, and only hints when it loads fewer. + +The module count is measured in a fresh ``python -I`` interpreter with only +the measured tree's ``src`` on the path, so neither the environment nor an +installed copy of hwpx decides what gets imported. + + python scripts/size_ratchet.py # working tree vs last release + python scripts/size_ratchet.py --record 6.8.0 # append the working tree + python scripts/size_ratchet.py --record 6.0.0 --ref v6.0.0 # measure a git ref + python scripts/size_ratchet.py --lower-bound # tighten the module bound +""" + +from __future__ import annotations + +import argparse +import json +import pathlib +import statistics +import subprocess +import sys +import tarfile +import tempfile +from collections import Counter +from typing import Any + +ROOT = pathlib.Path(__file__).resolve().parent.parent +SRC = ROOT / "src" +HISTORY = ROOT / "docs" / "size-history.json" +BREADTH = ROOT / "tests" / "data" / "import_breadth.json" +HISTORY_SCHEMA = "python-hwpx.size-history/v1" +BREADTH_SCHEMA = "python-hwpx.import-breadth/v1" +ENTRY_KEYS = frozenset( + {"version", "commit", "totalLines", "packages", "importedModules", "importMs"} +) + +_IMPORT_PROBE = """ +import sys, time +sys.path.insert(0, {src!r}) +start = time.perf_counter() +import hwpx +elapsed = time.perf_counter() - start +loaded = sorted(name for name in sys.modules if name == "hwpx" or name.startswith("hwpx.")) +assert all( + getattr(sys.modules[name], "__file__", None) is None + or sys.modules[name].__file__.startswith({src!r}) + for name in loaded +), "hwpx was imported from outside the measured tree" +print(len(loaded)) +print(elapsed) +""" + + +def line_counts(package: pathlib.Path) -> tuple[int, dict[str, int]]: + packages: Counter[str] = Counter() + for path in sorted(package.rglob("*.py")): + relative = path.relative_to(package) + if "__pycache__" in relative.parts: + continue + with path.open("rb") as handle: + lines = sum(1 for _ in handle) + packages[relative.parts[0] if len(relative.parts) > 1 else "."] += lines + return sum(packages.values()), dict(sorted(packages.items())) + + +def import_probe(src: pathlib.Path = SRC, runs: int = 1) -> tuple[int, float]: + """(hwpx modules loaded, median import seconds over *runs* fresh interpreters).""" + + counts: set[int] = set() + times: list[float] = [] + for _ in range(runs): + result = subprocess.run( + [sys.executable, "-I", "-c", _IMPORT_PROBE.format(src=str(src))], + capture_output=True, + text=True, + check=True, + ) + count, elapsed = result.stdout.split() + counts.add(int(count)) + times.append(float(elapsed)) + if len(counts) != 1: # pragma: no cover - would mean the probe is not deterministic + raise RuntimeError(f"import hwpx loaded a different module count per run: {counts}") + return counts.pop(), statistics.median(times) + + +def measure(src: pathlib.Path = SRC, runs: int = 3) -> dict[str, Any]: + total, packages = line_counts(src / "hwpx") + modules, elapsed = import_probe(src, runs) + return { + "totalLines": total, + "packages": packages, + "importedModules": modules, + "importMs": round(elapsed * 1000), + } + + +def measure_ref(ref: str, runs: int = 3) -> tuple[dict[str, Any], str]: + """Measure *ref* from ``git archive`` into a temp dir — no worktree, no checkout.""" + + commit = subprocess.run( + ["git", "rev-parse", "--short", f"{ref}^{{commit}}"], + cwd=ROOT, capture_output=True, text=True, check=True, + ).stdout.strip() + with tempfile.TemporaryDirectory() as scratch: + archive = pathlib.Path(scratch) / "src.tar" + subprocess.run( + ["git", "archive", "--format=tar", "-o", str(archive), commit, "src/hwpx"], + cwd=ROOT, check=True, + ) + with tarfile.open(archive) as tar: + try: + tar.extractall(scratch, filter="data") + except TypeError: # Python without extraction filters; our own archive + tar.extractall(scratch) + return measure(pathlib.Path(scratch) / "src", runs), commit + + +def head_commit() -> str: + return subprocess.run( + ["git", "rev-parse", "--short", "HEAD"], cwd=ROOT, capture_output=True, text=True, check=True + ).stdout.strip() + + +def load_history(path: pathlib.Path = HISTORY) -> dict[str, Any]: + if not path.exists(): + return {"schemaVersion": HISTORY_SCHEMA, "releases": []} + return json.loads(path.read_text(encoding="utf-8")) + + +def write_json(path: pathlib.Path, data: dict[str, Any]) -> None: + path.write_text(json.dumps(data, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") + + +def version_key(version: str) -> tuple[int, ...]: + return tuple(int(part) for part in version.split(".")) + + +def history_problems(history: dict[str, Any], project_version: str) -> list[str]: + """Shape of ``docs/size-history.json``; empty means well-formed.""" + + found: list[str] = [] + if history.get("schemaVersion") != HISTORY_SCHEMA: + found.append(f"schemaVersion is not {HISTORY_SCHEMA!r}") + releases = history.get("releases") or [] + if not releases: + found.append("no releases recorded") + versions: list[tuple[int, ...]] = [] + for index, entry in enumerate(releases): + if set(entry) != ENTRY_KEYS: + found.append(f"releases[{index}] keys {sorted(entry)} != {sorted(ENTRY_KEYS)}") + continue + try: + versions.append(version_key(entry["version"])) + except ValueError: + found.append(f"releases[{index}] version {entry['version']!r} is not N.N.N") + continue + if sum(entry["packages"].values()) != entry["totalLines"]: + found.append(f"releases[{index}] packages do not add up to totalLines") + if versions != sorted(versions) or len(set(versions)) != len(versions): + found.append("releases are not in strictly increasing version order") + if versions and versions[-1] > version_key(project_version): + found.append( + f"newest recorded release {releases[-1]['version']} is newer than " + f"pyproject version {project_version}" + ) + return found + + +def breadth_message(live: int, bound: int) -> tuple[bool, str]: + """(ok, message) for the upper-bound module ratchet.""" + + if live > bound: + return False, ( + f"import hwpx now loads {live} hwpx modules, above the bound of {bound} in " + "tests/data/import_breadth.json. Keep new modules out of the import path " + "(import them lazily where they are used), or, if the growth is intended, " + "raise the bound in the same change and say why in the pull request." + ) + if live < bound: + return True, ( + f"import hwpx loads {live} hwpx modules, below the bound of {bound}; " + "run `python scripts/size_ratchet.py --lower-bound` to tighten it." + ) + return True, f"import hwpx loads {live} hwpx modules (bound {bound})" + + +def _delta(after: int, before: int | None) -> str: + if before is None: + return "new" + change = after - before + percent = f" ({change / before:+.1%})" if before else "" + return f"{change:+d}{percent}" + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--record", metavar="VERSION", help="append a release entry") + parser.add_argument("--ref", help="with --record: measure this git ref instead of the working tree") + parser.add_argument("--lower-bound", action="store_true", help="set the module bound to the live count") + args = parser.parse_args(argv) + + if args.ref and not args.record: + parser.error("--ref needs --record") + + if args.record: + if args.ref: + live, commit = measure_ref(args.ref) + else: + live, commit = measure(), head_commit() + history = load_history() + releases = [r for r in history["releases"] if r["version"] != args.record] + releases.append({"version": args.record, "commit": commit, **live}) + releases.sort(key=lambda r: version_key(r["version"])) + history["releases"] = releases + write_json(HISTORY, history) + print( + f"recorded {args.record} ({commit}): {live['totalLines']} lines, " + f"{live['importedModules']} modules, {live['importMs']} ms" + ) + return 0 + + live = measure() + bound = json.loads(BREADTH.read_text(encoding="utf-8")) + if args.lower_bound: + bound["maxImportedModules"] = live["importedModules"] + write_json(BREADTH, bound) + print(f"module bound set to {live['importedModules']}") + return 0 + + releases = load_history()["releases"] + if releases: + last = releases[-1] + print(f"working tree vs {last['version']} ({last['commit']}):") + print(f" total {live['totalLines']:>7} {_delta(live['totalLines'], last['totalLines'])}") + for name, lines in live["packages"].items(): + print(f" {name:<10} {lines:>7} {_delta(lines, last['packages'].get(name))}") + for name in sorted(set(last["packages"]) - set(live["packages"])): + print(f" {name:<10} {0:>7} removed ({-last['packages'][name]:+d})") + print( + f" modules {live['importedModules']:>7} " + f"{_delta(live['importedModules'], last['importedModules'])}" + ) + print(f"import hwpx: {live['importMs']} ms (median of 3, not gated)") + ok, message = breadth_message(live["importedModules"], bound["maxImportedModules"]) + print(message) + return 0 if ok else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/data/import_breadth.json b/tests/data/import_breadth.json new file mode 100644 index 00000000..5311abba --- /dev/null +++ b/tests/data/import_breadth.json @@ -0,0 +1,4 @@ +{ + "schemaVersion": "python-hwpx.import-breadth/v1", + "maxImportedModules": 117 +} diff --git a/tests/data/layer_leak_allowlist.json b/tests/data/layer_leak_allowlist.json new file mode 100644 index 00000000..948a1c6e --- /dev/null +++ b/tests/data/layer_leak_allowlist.json @@ -0,0 +1,81 @@ +{ + "schemaVersion": "python-hwpx.layer-leak-allowlist/v1", + "entries": [ + { + "file": "tools/doc_diff.py", + "qualname": "_ATTACHMENT_LINE_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "known-leak", + "reason": "Korean official-document attachment line ('붙임 1 ... 1부'), used by the reference-consistency check. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/doc_diff.py", + "qualname": "_ATTACHMENT_REFERENCE_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "known-leak", + "reason": "Korean official-document attachment reference ('붙임 1 참조'), used by the reference-consistency check. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/doc_diff.py", + "qualname": "_TABLE_NUMBER_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "known-leak", + "reason": "Body-text table numbering ('표 1'), a writing convention checked by the reference-consistency check, not a format element. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/doc_diff.py", + "qualname": "_FIGURE_NUMBER_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "known-leak", + "reason": "Body-text figure numbering ('그림 1'), a writing convention checked by the reference-consistency check, not a format element. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/doc_diff.py", + "qualname": "_paragraphs_from_document_plan", + "kind": "plan-schema-key", + "count": 2, + "classification": "known-leak", + "reason": "Core parses the automation layer's document-plan schema (sections/blocks) so the diff can take a plan as input. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/markdown_export.py", + "qualname": "ARABIC_HEAD", + "kind": "hangul-regex", + "count": 1, + "classification": "known-leak", + "reason": "Report-style text-pattern heading ('1. 제목') guessed from plain text rather than read from the outline level; ROMAN_HEAD next to it is the same guess but has no Hangul, so it is listed under undetected. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/mail_merge.py", + "qualname": "_SAFE_FILENAME_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "format-vocabulary", + "reason": "Output file-name sanitizer that keeps Hangul letters alongside ASCII; script-neutral character class, no genre or policy." + }, + { + "file": "tools/toc_author.py", + "qualname": "_OUTLINE_NAME_RE", + "kind": "hangul-regex", + "count": 1, + "classification": "format-vocabulary", + "reason": "Hancom's built-in outline style names ('개요 1' / 'Outline 1') that a table of contents is generated from — format vocabulary." + } + ], + "undetected": [ + { + "file": "table_patch.py", + "qualname": "strip_trailing_table_captions", + "reason": "Eval-plan form clean-up: removes a '(N) word' caption baked after a table. Its pattern (_TC_CAPTION) has no Hangul, so neither signal sees it. Known leak — scheduled to move in 7.0." + }, + { + "file": "tools/markdown_export.py", + "qualname": "ROMAN_HEAD", + "reason": "Report-style text-pattern heading ('Ⅰ. 제목'); Roman numerals are not Hangul, so the scan does not see it. Moves with ARABIC_HEAD. Known leak — scheduled to move in 7.0." + } + ] +} diff --git a/tests/test_layer_leak_guard.py b/tests/test_layer_leak_guard.py new file mode 100644 index 00000000..0996624c --- /dev/null +++ b/tests/test_layer_leak_guard.py @@ -0,0 +1,181 @@ +# SPDX-License-Identifier: Apache-2.0 +"""Genre and policy logic may not hide inside core-owned functions. + +The ownership ledger classifies whole files; ``scripts/layer_leak_guard.py`` +looks inside them for Hangul regexes and the automation layer's +document-plan keys. Every existing hit is listed with a reason in +``tests/data/layer_leak_allowlist.json``. +""" + +from __future__ import annotations + +import json +import subprocess +import sys +import textwrap +from pathlib import Path + +import pytest + +import layer_leak_guard as guard # scripts/ is on the pytest pythonpath + +ROOT = Path(__file__).resolve().parent.parent +ALLOWLIST = ROOT / "tests" / "data" / "layer_leak_allowlist.json" + + +def _scan(source: str, relative: str = "tools/new_genre.py") -> list[guard.Hit]: + return guard.scan_source(relative, textwrap.dedent(source)) + + +def test_the_committed_tree_matches_the_allowlist() -> None: + found = guard.problems(guard.scan(), guard.load_allowlist()) + guard.undetected_problems() + + assert found == [], "\n".join(found) + "\n\n" + guard.GUIDANCE + + +def test_cli_passes_on_the_committed_tree() -> None: + result = subprocess.run( + [sys.executable, str(ROOT / "scripts" / "layer_leak_guard.py")], + capture_output=True, + text=True, + cwd=ROOT, + ) + + assert result.returncode == 0, result.stdout + result.stderr + + +def test_a_new_hangul_regex_fails_and_points_to_the_guardrails() -> None: + new = _scan( + """ + import re + + def classify_notice(text): + return re.search(r"^\\s*공고\\s*제\\d+호", text) + """ + ) + + found = guard.problems(guard.scan() + new, guard.load_allowlist()) + + assert len(found) == 1 + assert "new hangul-regex in src/hwpx/tools/new_genre.py classify_notice" in found[0] + assert "section 4" in guard.GUIDANCE + assert "docs/architecture/product-boundary.md" in guard.GUIDANCE + + +def test_a_second_hit_in_an_allowlisted_function_fails() -> None: + extra = _scan( + """ + def _paragraphs_from_document_plan(plan): + return plan["blocks"] + """, + relative="tools/doc_diff.py", + ) + + found = guard.problems(guard.scan() + extra, guard.load_allowlist()) + + assert len(found) == 1 + assert "3 found, 2 allowed" in found[0] + + +def test_a_fixed_leak_must_leave_the_allowlist() -> None: + hits = [h for h in guard.scan() if h.qualname != "ARABIC_HEAD"] + + found = guard.problems(hits, guard.load_allowlist()) + + assert found == [ + "stale allowlist entry src/hwpx/tools/markdown_export.py ARABIC_HEAD hangul-regex: " + "1 allowed, 0 found — lower or remove it in tests/data/layer_leak_allowlist.json" + ] + + +@pytest.mark.parametrize( + "source,qualname", + [ + # module-level constant passed by name + ('import re\nPAT = "^붙임"\ndef f(t):\n return re.match(PAT, t)\n', "f"), + # aliased module, keyword argument, method of a class + ( + "import re as _re\nclass C:\n def m(self, t):\n" + " return _re.fullmatch(pattern='개요', string=t)\n", + "C.m", + ), + # from-import, implicit concatenation + ('from re import sub\ndef g(t):\n return sub("[가-힣]" "+", "", t)\n', "g"), + # module-level compile is reported under its name + ('import re\nHEAD = re.compile(r"^제\\d+장")\n', "HEAD"), + ], +) +def test_hangul_regex_forms(source: str, qualname: str) -> None: + hits = _scan(source) + + assert [(h.kind, h.qualname) for h in hits] == [("hangul-regex", qualname)] + + +def test_plan_schema_keys() -> None: + hits = _scan( + """ + def adapt(plan): + for section in plan.get("sections", []): + yield section["blocks"] + plan["title"] + plan.get("paragraphs") + """ + ) + + assert [(h.kind, h.detail) for h in hits] == [ + ("plan-schema-key", ".get('sections')"), + ("plan-schema-key", "['blocks']"), + ] + + +def test_ascii_regexes_and_non_regex_hangul_are_not_hits() -> None: + hits = _scan( + """ + import re + LABEL = "표" + def f(t): + re.compile(r"section\\d+\\.xml$") + return LABEL in t + """ + ) + + assert hits == [] + + +def test_excluded_paths_are_not_scanned() -> None: + scanned = {path.relative_to(guard.SRC).as_posix() for path in guard.product_modules()} + + assert "_moved_modules.py" not in scanned + assert not any(path.startswith("data/") for path in scanned) + assert "table_patch.py" in scanned + + +def test_allowlist_entries_are_classified_and_explained() -> None: + data = json.loads(ALLOWLIST.read_text(encoding="utf-8")) + + for entry in data["entries"]: + assert set(entry) == {"file", "qualname", "kind", "count", "classification", "reason"} + assert entry["kind"] in {"hangul-regex", "plan-schema-key"} + assert entry["classification"] in {"known-leak", "format-vocabulary"} + assert entry["count"] >= 1 and entry["reason"] + if entry["classification"] == "known-leak": + assert "scheduled to move in 7.0" in entry["reason"] + for entry in data["undetected"]: + assert set(entry) == {"file", "qualname", "reason"} + assert "scheduled to move in 7.0" in entry["reason"] + + +def test_undetected_known_leaks_must_still_exist(tmp_path: Path) -> None: + data = json.loads(ALLOWLIST.read_text(encoding="utf-8")) + data["undetected"].append( + {"file": "table_patch.py", "qualname": "moved_away", "reason": "x"} + ) + allowlist = tmp_path / "allowlist.json" + allowlist.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8") + + found = guard.undetected_problems(allowlist) + + assert found == [ + "undetected known leak src/hwpx/table_patch.py moved_away no longer exists — " + "remove it from tests/data/layer_leak_allowlist.json" + ] diff --git a/tests/test_size_ratchet.py b/tests/test_size_ratchet.py new file mode 100644 index 00000000..d5fe266e --- /dev/null +++ b/tests/test_size_ratchet.py @@ -0,0 +1,102 @@ +# SPDX-License-Identifier: Apache-2.0 +"""Core size history is well-formed, and ``import hwpx`` stays under its bound. + +Line counts are recorded per release in ``docs/size-history.json`` during +release prep (``python scripts/size_ratchet.py --record ``); they are +not a pull-request gate, because parallel branches would conflict on any exact +lock. The number of hwpx modules ``import hwpx`` loads is an upper-bound +ratchet in ``tests/data/import_breadth.json``. +""" + +from __future__ import annotations + +import json +import re +import subprocess +import sys +import warnings +from pathlib import Path + +import size_ratchet # scripts/ is on the pytest pythonpath + +ROOT = Path(__file__).resolve().parent.parent +HISTORY = ROOT / "docs" / "size-history.json" +BREADTH = ROOT / "tests" / "data" / "import_breadth.json" + + +def _project_version() -> str: + text = (ROOT / "pyproject.toml").read_text(encoding="utf-8") + match = re.search(r'^version = "([^"]+)"', text, re.MULTILINE) + assert match is not None + return match.group(1) + + +def test_size_history_is_well_formed() -> None: + history = json.loads(HISTORY.read_text(encoding="utf-8")) + + found = size_ratchet.history_problems(history, _project_version()) + + assert found == [], "\n".join(found) + + +def test_history_checks_can_fail() -> None: + history = json.loads(HISTORY.read_text(encoding="utf-8")) + newest = history["releases"][-1] + ahead = json.loads(json.dumps(history)) + ahead["releases"].append({**newest, "version": "99.0.0"}) + shuffled = json.loads(json.dumps(history)) + shuffled["releases"].reverse() + broken = json.loads(json.dumps(history)) + broken["releases"][0]["totalLines"] += 1 + del broken["releases"][1]["importMs"] + + assert size_ratchet.history_problems(ahead, _project_version()) == [ + f"newest recorded release 99.0.0 is newer than pyproject version {_project_version()}" + ] + assert size_ratchet.history_problems(shuffled, _project_version()) == [ + "releases are not in strictly increasing version order" + ] + assert len(size_ratchet.history_problems(broken, _project_version())) == 2 + + +def test_import_breadth_stays_under_the_bound() -> None: + bound = json.loads(BREADTH.read_text(encoding="utf-8")) + assert bound["schemaVersion"] == size_ratchet.BREADTH_SCHEMA + live, _elapsed = size_ratchet.import_probe() + + ok, message = size_ratchet.breadth_message(live, bound["maxImportedModules"]) + + assert ok, message + if live < bound["maxImportedModules"]: + warnings.warn(message, UserWarning, stacklevel=1) + + +def test_breadth_ratchet_fails_only_upward() -> None: + above, above_message = size_ratchet.breadth_message(118, 117) + below, below_message = size_ratchet.breadth_message(100, 117) + equal, _ = size_ratchet.breadth_message(117, 117) + + assert not above and "above the bound of 117" in above_message + assert below and "--lower-bound" in below_message + assert equal + + +def test_import_probe_is_deterministic() -> None: + first, _ = size_ratchet.import_probe() + second, _ = size_ratchet.import_probe() + + assert first == second > 1 + + +def test_cli_prints_the_delta_against_the_last_release() -> None: + result = subprocess.run( + [sys.executable, str(ROOT / "scripts" / "size_ratchet.py")], + capture_output=True, + text=True, + cwd=ROOT, + ) + + assert result.returncode == 0, result.stdout + result.stderr + newest = json.loads(HISTORY.read_text(encoding="utf-8"))["releases"][-1]["version"] + assert f"working tree vs {newest}" in result.stdout + assert "not gated" in result.stdout