C6-Reticulum-ASM/tools/check_registry.py
2026-05-03 00:06:29 -06:00

325 lines
10 KiB
Python
Executable file
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""FUNCTIONS.md ↔ src/ cross-checker.
Per FUNCTIONS.md, the CI build fails if:
1. A .S file in src/ defines a global symbol not registered.
2. A registered function with non-`planned` status has no source file.
3. A function references a `depends-on` that does not exist in the registry.
Also reports (warnings, not failures):
* Source files whose status in FUNCTIONS.md disagrees with the @status field
in their spec block (per ADR-0007).
Functions whose name contains `*` (e.g., `x25519_field_*`) are wildcard
placeholders and are excluded from the per-function cross-check; the registry
text is still parsed for module bookkeeping.
Exits 0 on full agreement, 1 on any failure. Warnings do not affect exit code.
"""
from __future__ import annotations
import argparse
import dataclasses
import json
import re
import sys
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT / "tools"))
import parse_spec # noqa: E402
STATUS_SYMBOLS = {
"": "planned",
"": "in-progress",
"": "tested",
"": "verified",
"": "superseded",
}
VALID_STATUS_WORDS = set(STATUS_SYMBOLS.values())
MODULE_HEADER_RE = re.compile(r"^##\s+Module:\s+`([^`]+)`\s*$")
TABLE_ROW_RE = re.compile(r"^\|(.+)\|\s*$")
@dataclasses.dataclass
class FuncEntry:
name: str
module: str
status: str
owner: str
depends_on: list[str]
adrs: list[str]
def _strip_md(cell: str) -> str:
"""Strip backticks and surrounding whitespace from a cell. Asterisks are
preserved because they are meaningful (wildcard placeholders like
`x25519_field_*`).
"""
return cell.strip().strip("`").strip()
def _parse_status_cell(cell: str) -> str:
"""Extract a status word from a cell that may contain a symbol, a word, or both."""
cell = cell.strip()
if not cell:
return ""
first = cell[0]
if first in STATUS_SYMBOLS:
rest = cell[1:].strip()
# Trust the symbol; if the trailing word disagrees, return symbol meaning.
return STATUS_SYMBOLS[first]
# No symbol: rely on the word.
word = cell.split()[0].lower()
return word if word in VALID_STATUS_WORDS else cell
def _parse_list_cell(cell: str) -> list[str]:
"""Parse a comma- or whitespace-separated cell into a list, dropping em-dashes."""
cell = cell.strip()
if cell in {"", "", "-", ""}:
return []
parts: list[str] = []
for raw in re.split(r"[,\s]+", cell):
raw = _strip_md(raw)
if raw and raw not in {"", "-", ""}:
parts.append(raw)
return parts
def _cell(cells: list[str], idx: int | None, default: str = "") -> str:
if idx is None or idx < 0 or idx >= len(cells):
return default
return cells[idx]
def _header_map(cells: list[str]) -> dict[str, int]:
return {_strip_md(cell).lower(): idx for idx, cell in enumerate(cells)}
def parse_registry(path: Path) -> list[FuncEntry]:
"""Parse FUNCTIONS.md into a flat list of FuncEntry."""
text = path.read_text(encoding="utf-8")
entries: list[FuncEntry] = []
current_module: str | None = None
in_table = False
saw_header = False
columns: dict[str, int] = {}
for line in text.splitlines():
m = MODULE_HEADER_RE.match(line)
if m:
current_module = m.group(1)
in_table = False
saw_header = False
columns = {}
continue
row = TABLE_ROW_RE.match(line)
if not row:
in_table = False
saw_header = False
columns = {}
continue
cells = [c.strip() for c in row.group(1).split("|")]
# Recognise table header / separator rows: skip until past them.
if not saw_header and cells and cells[0].lower() == "function":
columns = _header_map(cells)
saw_header = True
in_table = False
continue
if saw_header and not in_table:
# Either separator (---|---|...) or a data row. The separator has
# only dashes and colons; everything else is a data row.
if all(set(c) <= set("-:") for c in cells if c):
in_table = True
continue
# No separator (rare): assume immediate data rows.
in_table = True
if not in_table or not current_module:
continue
if len(cells) < 4:
continue
name_idx = columns.get("function", 0)
status_idx = columns.get("status", 1)
owner_idx = columns.get("owner")
depends_idx = columns.get("depends-on", 2 if owner_idx is None else 3)
adrs_idx = columns.get("adrs", 3 if owner_idx is None else 4)
name_raw = _cell(cells, name_idx)
# Skip placeholder rows like "(functions added when milestone N is activated)".
if name_raw.startswith("("):
continue
name = _strip_md(name_raw)
if not name:
continue
status = _parse_status_cell(_cell(cells, status_idx))
owner = _strip_md(_cell(cells, owner_idx))
depends = _parse_list_cell(_cell(cells, depends_idx))
adrs = _parse_list_cell(_cell(cells, adrs_idx))
entries.append(
FuncEntry(
name=name,
module=current_module,
status=status,
owner=owner,
depends_on=depends,
adrs=adrs,
)
)
return entries
def _collect_src_globals(repo_root: Path) -> dict[str, Path]:
"""Map global *function* symbol → source file path, scanning every
src/**/*.S.
A symbol is a function only if it is BOTH declared `.global <sym>` and
typed `.type <sym>, @function`. Data symbols in `src/state/<module>.S`
(per ADR-0002) declare globals but not @function, so they are excluded
from the function registry cross-check.
"""
src = repo_root / "src"
if not src.is_dir():
return {}
globals_: dict[str, Path] = {}
for path in sorted(src.rglob("*.S")):
if "include" in path.parts:
continue
text = path.read_text(encoding="utf-8")
declared_global = {
m.group(1)
for m in re.finditer(r"^\s*\.global?\s+([A-Za-z_]\w*)", text, re.MULTILINE)
}
function_typed = {
m.group(1)
for m in re.finditer(
r"^\s*\.type\s+([A-Za-z_]\w*)\s*,\s*@function",
text,
re.MULTILINE,
)
}
for name in declared_global & function_typed:
globals_[name] = path
return globals_
def check(repo_root: Path = REPO_ROOT) -> tuple[list[str], list[str]]:
"""Return (errors, warnings)."""
errors: list[str] = []
warnings: list[str] = []
registry_path = repo_root / "FUNCTIONS.md"
if not registry_path.is_file():
return [f"FUNCTIONS.md not found at {registry_path}"], []
entries = parse_registry(registry_path)
by_name: dict[str, FuncEntry] = {}
for e in entries:
if e.name in by_name:
errors.append(f"FUNCTIONS.md: duplicate entry for {e.name}")
continue
by_name[e.name] = e
# Check 3: depends-on entries must exist in the registry. Wildcards
# (e.g. `sha256_*`) are satisfied if any entry's name starts with the
# prefix, which includes a wildcard entry of the same prefix.
for e in by_name.values():
for dep in e.depends_on:
if "*" in dep:
prefix = dep.rstrip("*")
if not any(n.startswith(prefix) for n in by_name):
errors.append(
f"FUNCTIONS.md: {e.name} depends on wildcard {dep!r} "
f"but no matching registered function found"
)
continue
if dep not in by_name:
errors.append(
f"FUNCTIONS.md: {e.name} depends on {dep!r} which is not registered"
)
# Check 1 & 2: source-file ↔ registry agreement.
src_globals = _collect_src_globals(repo_root)
for sym, path in src_globals.items():
if sym not in by_name:
errors.append(
f"src: {path.relative_to(repo_root)} declares global {sym!r} "
f"not registered in FUNCTIONS.md"
)
for name, entry in by_name.items():
if "*" in name:
continue # wildcard placeholder; source files appear when expanded
if entry.status == "planned":
continue # no source file expected yet
if entry.status == "superseded":
continue # superseded entries kept for history; source may be gone
if name not in src_globals:
errors.append(
f"FUNCTIONS.md: {name!r} status is {entry.status!r} "
f"but no src/**/*.S declares it as global"
)
# Warning: spec-block @status disagrees with registry status.
for name, entry in by_name.items():
if "*" in name:
continue
path = src_globals.get(name)
if path is None:
continue
try:
spec = parse_spec.parse_spec(path)
except parse_spec.SpecError as exc:
warnings.append(f"{path.relative_to(repo_root)}: cannot parse spec: {exc}")
continue
spec_status = spec.fields.get("status", "")
if spec_status and spec_status != entry.status:
warnings.append(
f"{path.relative_to(repo_root)}: spec @status {spec_status!r} "
f"disagrees with FUNCTIONS.md {entry.status!r}"
)
return errors, warnings
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--json", action="store_true")
parser.add_argument("--repo-root", type=Path, default=REPO_ROOT)
args = parser.parse_args(argv)
errors, warnings = check(args.repo_root.resolve())
if args.json:
json.dump(
{"errors": errors, "warnings": warnings},
sys.stdout,
indent=2,
)
sys.stdout.write("\n")
else:
for w in warnings:
print(f"warning: {w}", file=sys.stderr)
for e in errors:
print(f"error: {e}", file=sys.stderr)
if not errors and not warnings:
print("registry: ok")
return 1 if errors else 0
if __name__ == "__main__":
sys.exit(main())