Files changed: - AGENTS.md - CHANGES.md - README.md - VERSION - docs/why-gates-are-code.md - instructions/gates.md - instructions/kb-profiles.md - kb/CONTRACT.md - kb/CONVENTIONS.md - kb/CONVENTIONS.md.template - raw/CONTRACT.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/cli.py - tools/chemenu/cli_contract.py - tools/chemenu/commands/export_cmd.py - tools/chemenu/commands/page_ops.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/commands/search.py - tools/chemenu/guideline_export.py - tools/chemenu/kb_scan.py - tools/chemenu/repo_capture.py - tools/chemenu/search/filters.py - tools/chemenu/tests/test_cli.py - tools/chemenu/tests/test_export_guidelines.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
297 lines
12 KiB
Python
297 lines
12 KiB
Python
"""The guideline export, with no CLI attached: select pages by frontmatter,
|
|
render them mechanically into one `GUIDELINES.md`, and recognise such a file
|
|
again when it comes back.
|
|
|
|
`wikitool export guidelines` is the terminal adapter over this module; the git
|
|
side - writing the file into a captured repository's branch - lives beside
|
|
`fetch` in `repo_capture.py`. Nothing here imports `typer` or anything under
|
|
`chemenu.commands`, and nothing here imports `repo_capture`, which imports
|
|
`EXPORT_MARKER` from here: that is the direction the dependency runs.
|
|
|
|
Three properties carry the design, and each has a test:
|
|
|
|
- **Deterministic.** The output depends only on the selected pages and the
|
|
commit that last touched one of them - no timestamp, no `HEAD`. A run on an
|
|
unchanged corpus, or after an instance commit that touched no guideline, is
|
|
byte-identical, and so writes nothing into any target repository.
|
|
- **One filter semantics.** Selection is `search`'s own predicate machinery
|
|
(`search.filters`), with no text argument: a guideline is a declared choice in
|
|
frontmatter, not a full-text hit.
|
|
- **Code is never rewritten.** Every transformation of the page text runs on a
|
|
copy with code masked out (`markdown_code.strip_code_spans`, which keeps
|
|
offsets), so inline code and fenced blocks reach the output byte for byte.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import subprocess
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from chemenu import blocks, config, corpus_cache, provenance, toolpaths
|
|
from chemenu.errors import ValidationError
|
|
from chemenu.kb_scan import LINK_RE, normalize_link_target
|
|
from chemenu.markdown_code import strip_code_spans
|
|
from chemenu.page import Page
|
|
from chemenu.search import filters
|
|
from chemenu.search.service import load_pages_by_path, unreadable_pages
|
|
from chemenu.search.types import Predicate
|
|
|
|
# The first bytes of every file this stack generates for another repository.
|
|
# Defined once, here: `repo_capture` excludes a file carrying it from a
|
|
# capture, `raw accept` refuses one, and the export writes it.
|
|
EXPORT_MARKER = b"<!-- wikitool:export"
|
|
GUIDELINES_KIND = "guidelines"
|
|
GUIDELINES_PREFIX = f"<!-- wikitool:export kind={GUIDELINES_KIND}"
|
|
TARGET_PATH = "GUIDELINES.md"
|
|
LOCAL_INSTANCE = "local"
|
|
|
|
_BOM = b"\xef\xbb\xbf"
|
|
_HEADER_FIELD = re.compile(r"\b(instance|commit)=(\S+)")
|
|
_H1 = re.compile(r"#(?:[ \t]|$)")
|
|
_SCHEME_USERINFO = re.compile(r"^([A-Za-z][A-Za-z0-9+.-]*://)[^/]*@")
|
|
_SCP_USERINFO = re.compile(r"^[^/@:]+@(?=[^/:]+:)")
|
|
|
|
|
|
# --- recognising an export ------------------------------------------------------
|
|
|
|
|
|
def _first_line(data: bytes) -> bytes:
|
|
data = data[len(_BOM):] if data.startswith(_BOM) else data
|
|
return data.split(b"\n", 1)[0]
|
|
|
|
|
|
def is_export(data: bytes) -> bool:
|
|
"""Whether `data` is a file this stack exported: its first line, after an
|
|
optional BOM, starts with `EXPORT_MARKER`. Reads no more than the first
|
|
line, so a caller may pass only a file's first bytes."""
|
|
return _first_line(data).startswith(EXPORT_MARKER)
|
|
|
|
|
|
def guidelines_header(data: bytes) -> Optional[dict[str, str]]:
|
|
"""The fields of a guideline export's first line - `instance`, `commit`,
|
|
whichever it carries - or None when `data` is not a guideline export at
|
|
all. A header with no fields (`{}`) is the bare opt-in stub."""
|
|
line = _first_line(data).decode("utf-8", "replace")
|
|
if not line.startswith(GUIDELINES_PREFIX):
|
|
return None
|
|
return dict(_HEADER_FIELD.findall(line[len(GUIDELINES_PREFIX):]))
|
|
|
|
|
|
def file_is_export(path: Path) -> bool:
|
|
"""`is_export` for a file on disk, reading only its first line's worth."""
|
|
with open(path, "rb") as handle:
|
|
return is_export(handle.read(len(_BOM) + len(EXPORT_MARKER)))
|
|
|
|
|
|
# --- the instance it comes from -----------------------------------------------
|
|
|
|
|
|
def _git(args: list[str], root: Path) -> Optional[str]:
|
|
"""`git <args>` in the instance checkout: stdout on success, None on any
|
|
failure. Only local reads - nothing here reaches a remote."""
|
|
try:
|
|
result = subprocess.run(
|
|
[toolpaths.git(), *args], cwd=root, capture_output=True, text=True,
|
|
encoding="utf-8", timeout=30, check=False,
|
|
)
|
|
except (OSError, subprocess.SubprocessError):
|
|
return None
|
|
return result.stdout if result.returncode == 0 else None
|
|
|
|
|
|
def strip_userinfo(url: str) -> str:
|
|
"""`url` without a user or token in front of the host -
|
|
`https://user:token@host/x` becomes `https://host/x`, `git@host:x` becomes
|
|
`host:x`. The header lands in other repositories, some of them public."""
|
|
stripped = _SCHEME_USERINFO.sub(r"\1", url, count=1)
|
|
if stripped != url:
|
|
return stripped
|
|
return _SCP_USERINFO.sub("", url, count=1)
|
|
|
|
|
|
def instance_id(root: Optional[Path] = None) -> str:
|
|
"""The `instance=` value: `origin`'s fetch URL without userinfo, or
|
|
`local` for a checkout without an `origin`."""
|
|
url = (_git(["remote", "get-url", "origin"], root or config.ROOT) or "").strip()
|
|
if not url:
|
|
return LOCAL_INSTANCE
|
|
return "".join(strip_userinfo(url).split()) or LOCAL_INSTANCE
|
|
|
|
|
|
def git_identity(root: Optional[Path] = None) -> Optional[tuple[str, str]]:
|
|
"""`(user.name, user.email)` as the instance checkout resolves them - its
|
|
local configuration before the global one - or None if either is unset.
|
|
The commit in a target repository is made in a bare cache repository, which
|
|
would see only the global configuration; this is the identity the
|
|
instance's own commits carry."""
|
|
root = root or config.ROOT
|
|
name = (_git(["config", "user.name"], root) or "").strip()
|
|
email = (_git(["config", "user.email"], root) or "").strip()
|
|
return (name, email) if name and email else None
|
|
|
|
|
|
# --- selection ----------------------------------------------------------------
|
|
|
|
|
|
def select(predicates: tuple[Predicate, ...], kb_dir: Optional[Path] = None,
|
|
root: Optional[Path] = None) -> list[Page]:
|
|
"""The pages the predicates select, sorted by `(title.casefold(), title)`.
|
|
|
|
Raises `ValidationError` instead of returning something incomplete: with
|
|
no predicate (a forgotten filter would export the whole wiki), on an
|
|
unknown field (as `search` does), on any page under `kb/` whose
|
|
frontmatter does not parse (it could be a guideline that silently drops
|
|
out), and on zero hits (an empty file would delete every repository's
|
|
guidelines)."""
|
|
if not predicates:
|
|
raise ValidationError(
|
|
"No predicate - the export takes the guidelines by frontmatter, and without a filter "
|
|
"it would take the whole wiki. kb/CONVENTIONS.md names this instance's filter."
|
|
)
|
|
kb_dir = kb_dir or config.KB_DIR
|
|
pages = load_pages_by_path(kb_dir, root or config.ROOT)
|
|
unreadable = unreadable_pages(pages)
|
|
if unreadable:
|
|
listed = "\n".join(f" - {u['path']} ({u['reason']})" for u in unreadable)
|
|
raise ValidationError(
|
|
"These pages have frontmatter that does not parse, so the export cannot tell whether "
|
|
f"they are guidelines:\n{listed}\n Fix them first; nothing was exported."
|
|
)
|
|
filters.validate_fields(predicates, pages)
|
|
selected = filters.apply_predicates(pages, predicates, kb_dir)
|
|
if not selected:
|
|
rendered = " ".join(p.render() for p in predicates)
|
|
raise ValidationError(
|
|
f"No page matches {rendered} - an empty export would delete every repository's "
|
|
"guidelines, so nothing was exported."
|
|
)
|
|
return sorted(selected.values(), key=lambda p: (p.title.casefold(), p.title))
|
|
|
|
|
|
def check_clean(root: Optional[Path] = None) -> None:
|
|
"""Refuse a checkout that is not a git repository, or whose `kb/` or
|
|
`types/` has uncommitted changes, untracked files included - otherwise
|
|
`commit=` would name a state the export was not made from. `types/` counts
|
|
because `kind` and `subtype` are resolved through the type-specs."""
|
|
root = root or config.ROOT
|
|
if corpus_cache.head_commit(root) is None:
|
|
raise ValidationError(
|
|
f"{root} is not a git checkout with a commit - the export names the commit its pages "
|
|
"come from."
|
|
)
|
|
dirty = [
|
|
Path(path).relative_to(root).as_posix() if Path(path).is_relative_to(root) else str(path)
|
|
for path in (config.KB_DIR, config.TYPES_DIR)
|
|
if corpus_cache.is_dirty(root, path)
|
|
]
|
|
if dirty:
|
|
raise ValidationError(
|
|
f"{' and '.join(dirty)} have uncommitted changes (untracked files count). Publish them "
|
|
"first - the export names the commit its pages come from."
|
|
)
|
|
|
|
|
|
def last_commit(pages: list[Page], root: Optional[Path] = None) -> str:
|
|
"""The newest commit touching one of `pages`' files - not `HEAD`, so an
|
|
instance commit that touched no guideline leaves the output unchanged."""
|
|
root = root or config.ROOT
|
|
paths = [Path(p.path).resolve().relative_to(Path(root).resolve()).as_posix() for p in pages]
|
|
sha = (_git(["log", "-1", "--format=%H", "--", *paths], root) or "").strip()
|
|
if not sha:
|
|
raise ValidationError("None of the selected pages is in a commit yet - publish them first.")
|
|
return sha
|
|
|
|
|
|
# --- rendering ----------------------------------------------------------------
|
|
|
|
|
|
def _inline(text: str) -> str:
|
|
"""Wikilinks to their display text; `[^cite-id]` and legacy `^[[...]]`
|
|
citations removed - all of it outside code only. Matches are found on the
|
|
masked copy and applied to the original, which share offsets."""
|
|
masked = strip_code_spans(text)
|
|
edits: list[tuple[int, int, str]] = []
|
|
for m in provenance.LEGACY_CITE_RE.finditer(masked):
|
|
edits.append((m.start(), m.end(), ""))
|
|
for m in provenance.CITE_REF_RE.finditer(masked):
|
|
edits.append((m.start(), m.end(), ""))
|
|
for m in LINK_RE.finditer(masked):
|
|
rest = text[m.start(2):m.end(2)]
|
|
display = (
|
|
rest.split("|", 1)[1] if "|" in rest else normalize_link_target(text[m.start(1):m.end(1)])
|
|
)
|
|
edits.append((m.start(), m.end(), display))
|
|
# A legacy citation contains a wikilink; the one starting first wins.
|
|
edits.sort(key=lambda e: (e[0], -e[1]))
|
|
out: list[str] = []
|
|
pos = 0
|
|
for start, end, replacement in edits:
|
|
if start < pos:
|
|
continue
|
|
out.append(text[pos:start])
|
|
out.append(replacement)
|
|
pos = end
|
|
out.append(text[pos:])
|
|
return "".join(out)
|
|
|
|
|
|
def _trim_blank(lines: list[str]) -> list[str]:
|
|
start, end = 0, len(lines)
|
|
while start < end and not lines[start].strip():
|
|
start += 1
|
|
while end > start and not lines[end - 1].strip():
|
|
end -= 1
|
|
return lines[start:end]
|
|
|
|
|
|
def render_page(page: Page) -> str:
|
|
"""One page: `# <title>`, then its text without the generated links and
|
|
footnote blocks, citations removed and wikilinks resolved to text. A
|
|
leading H1 of the page's own is replaced by the title heading; every other
|
|
heading keeps its level."""
|
|
body = page.body.replace("\r\n", "\n")
|
|
body = blocks.strip(body, blocks.LINKS)
|
|
body, _definitions = provenance.split_cite_block(body)
|
|
lines = _trim_blank(_inline(body).split("\n"))
|
|
if lines and _H1.match(lines[0]):
|
|
lines = _trim_blank(lines[1:])
|
|
return "\n".join([f"# {page.title}", "", *lines]) if lines else f"# {page.title}"
|
|
|
|
|
|
def header(instance: str, commit: str) -> str:
|
|
return (
|
|
f"{GUIDELINES_PREFIX} instance={instance} commit={commit} "
|
|
"- generated, do not edit by hand -->"
|
|
)
|
|
|
|
|
|
def render(pages: list[Page], instance: str, commit: str) -> str:
|
|
"""The whole file: the header line, then each page, one blank line
|
|
between. LF line endings, exactly one trailing newline."""
|
|
parts = [header(instance, commit), *(render_page(p) for p in pages)]
|
|
return "\n\n".join(parts) + "\n"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Export:
|
|
text: str
|
|
pages: tuple[Page, ...]
|
|
instance: str
|
|
commit: str
|
|
|
|
@property
|
|
def data(self) -> bytes:
|
|
return self.text.encode("utf-8")
|
|
|
|
|
|
def build(predicates: tuple[Predicate, ...], root: Optional[Path] = None) -> Export:
|
|
"""Select, check and render in one go. Raises `ValidationError`."""
|
|
root = root or config.ROOT
|
|
pages = select(predicates, root=root)
|
|
check_clean(root)
|
|
commit = last_commit(pages, root)
|
|
instance = instance_id(root)
|
|
return Export(render(pages, instance, commit), tuple(pages), instance, commit)
|