Files changed: - AGENTS.md - CHANGES.md - README.md - VERSION - docs/why-gates-are-code.md - instructions/gates.md - instructions/kb-profiles.md - kb/CONTRACT.md - kb/CONVENTIONS.md - kb/CONVENTIONS.md.template - raw/CONTRACT.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/cli.py - tools/chemenu/cli_contract.py - tools/chemenu/commands/export_cmd.py - tools/chemenu/commands/page_ops.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/commands/search.py - tools/chemenu/guideline_export.py - tools/chemenu/kb_scan.py - tools/chemenu/repo_capture.py - tools/chemenu/search/filters.py - tools/chemenu/tests/test_cli.py - tools/chemenu/tests/test_export_guidelines.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
264 lines
11 KiB
Python
264 lines
11 KiB
Python
"""Scan kb/ into Page objects and build the wikilink graph."""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from collections import Counter
|
|
from pathlib import Path
|
|
from typing import Iterator
|
|
|
|
from chemenu.frontmatter_io import read_page
|
|
from chemenu.markdown_code import strip_code_spans
|
|
from chemenu.page import Page
|
|
|
|
WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)")
|
|
|
|
# The whole link: `[[Target]]`, `[[Target|alias]]`, `[[Target#anchor]]` -
|
|
# including the `[[Target]]` inside a `[^cite-id]: [[Target]]` Footnotes
|
|
# definition, which is exactly what lets `page_ops.retarget_body()` repoint a
|
|
# citation's link target on a rename. Group 1 is the target title; group 2 keeps
|
|
# any alias/anchor suffix untouched. Group 1 may span a line break; it is
|
|
# compared through `normalize_link_target()`, the same reading `lint` gives it.
|
|
LINK_RE = re.compile(r"\[\[([^\[\]|#]+)((?:[|#][^\[\]]*)?)\]\]")
|
|
|
|
# A line break inside `[[...]]`, with the indentation around it. The target
|
|
# class above admits a newline, so a link someone wrapped at a fixed column -
|
|
# `[[Foo Bar\n Target]]` - captures the break as part of the title, matches no
|
|
# page, and read as a missing one to `lint`, as no reference at all to `rm`'s
|
|
# inbound check, and as nothing to repoint to `rename`.
|
|
_LINE_BREAK_RUN = re.compile(r"[ \t]*(?:\r?\n[ \t]*)+")
|
|
|
|
|
|
def normalize_link_target(raw: str) -> str:
|
|
"""The title a captured wikilink target names: every whitespace run that
|
|
contains a line break folded to one space, then stripped.
|
|
|
|
Every reader of a body wikilink goes through this, so `lint`, the link
|
|
graph, `rename` and `rm` agree on what a wrapped link points at. That the
|
|
link is wrapped at all is still a finding - `wrapped_wikilinks` - because
|
|
folding it here would otherwise make it invisible.
|
|
"""
|
|
return _LINE_BREAK_RUN.sub(" ", raw).strip()
|
|
|
|
|
|
# Root-level files under kb/ that are not pages: the generated catalog map, log
|
|
# and provenance index, plus the two documents that constrain the tree rather
|
|
# than living in it - the stack's contract and this instance's own conventions.
|
|
_KB_META_FILES = {"index.md", "log.md", "provenance.md", "CONTRACT.md", "CONVENTIONS.md"}
|
|
|
|
# The per-collection authoring contract. Unlike the meta files above it is never
|
|
# at the kb root - it sits one level down, in every collection - so it has to be
|
|
# excluded by name at any depth rather than by parent directory.
|
|
_COLLECTION_CONTRACT = "COLLECTION.md"
|
|
|
|
# The generated per-collection/per-area catalog shard. Excluded by name at any
|
|
# depth for the same reason as the contract, and for one more: it lists every
|
|
# page in its subtree as a wikilink, so treating it as a page would make every
|
|
# page look linked-to and silence the orphan check entirely.
|
|
GENERATED_INDEX = "INDEX.md"
|
|
|
|
|
|
def is_page_path(relative: str) -> bool:
|
|
"""Whether a `kb/`-relative path names a page rather than routing material.
|
|
|
|
Stated over a plain path, not a filesystem entry, so callers that read a
|
|
*past* revision out of git can apply the identical rule - `migrate verify`
|
|
does. Two different answers to "is this a page" would report every
|
|
COLLECTION.md and INDEX.md as a page that has since disappeared.
|
|
"""
|
|
parts = relative.split("/")
|
|
if parts[-1] in (_COLLECTION_CONTRACT, GENERATED_INDEX):
|
|
return False
|
|
if len(parts) == 1 and parts[0] in _KB_META_FILES:
|
|
return False
|
|
return parts[-1].endswith(".md")
|
|
|
|
|
|
def iter_kb_pages(kb_dir: Path) -> Iterator[Path]:
|
|
"""Yield every page under kb_dir.
|
|
|
|
Three kinds of file are skipped: the kb-root meta files (generated catalog,
|
|
log, provenance, and the kb contract), every COLLECTION.md, and every
|
|
generated INDEX.md. None carry page frontmatter. A README.md *inside* a
|
|
collection is an ordinary page - only kb-root files are routing material.
|
|
"""
|
|
for path in sorted(kb_dir.rglob("*.md")):
|
|
if is_page_path(path.relative_to(kb_dir).as_posix()):
|
|
yield path
|
|
|
|
|
|
def load_kb_pages(kb_dir: Path) -> dict[str, Page]:
|
|
"""Load every markdown page under kb_dir, keyed by title (filename stem).
|
|
|
|
If two files share a stem (a naming collision), the later one (by sorted
|
|
path order) wins here; `wikitool lint` explicitly detects and reports such
|
|
collisions so they don't go unnoticed.
|
|
"""
|
|
pages: dict[str, Page] = {}
|
|
for path in iter_kb_pages(kb_dir):
|
|
frontmatter, body = read_page(path)
|
|
pages[path.stem] = Page(path=path, frontmatter=frontmatter, body=body)
|
|
return pages
|
|
|
|
|
|
def find_duplicate_title_paths(kb_dir: Path, root: Path) -> list[dict]:
|
|
"""Return stem collisions as {"stem": str, "paths": [str, ...]}.
|
|
|
|
Paths are repo-root-relative and sorted for stable output.
|
|
"""
|
|
by_stem: dict[str, list[str]] = {}
|
|
for path in iter_kb_pages(kb_dir):
|
|
try:
|
|
rel = path.relative_to(root).as_posix()
|
|
except ValueError:
|
|
rel = path.relative_to(kb_dir.parent).as_posix()
|
|
by_stem.setdefault(path.stem, []).append(rel)
|
|
return [
|
|
{"stem": stem, "paths": sorted(paths)}
|
|
for stem, paths in sorted(by_stem.items())
|
|
if len(paths) > 1
|
|
]
|
|
|
|
|
|
def extract_wikilinks(body: str) -> set[str]:
|
|
"""Which pages this body links to, as a set.
|
|
|
|
The right shape for `lint` and the link graph, whose question is "does
|
|
this reference resolve" - asked once per distinct target. It is the wrong
|
|
shape for asking whether a rewrite *dropped* a link: use
|
|
`count_wikilinks` for that.
|
|
|
|
Code is masked out first (see markdown_code.strip_code_spans): a
|
|
`[[Wikilink]]` shown inside a fence or backticks is an example of the
|
|
notation, and counting it made a page that documents the wiki look like it
|
|
linked to something that need not exist.
|
|
"""
|
|
return {
|
|
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
|
}
|
|
|
|
|
|
def wrapped_wikilinks(body: str) -> list[str]:
|
|
"""The normalized targets of every wikilink in this body written across a
|
|
line break, in order of appearance, code masked out as everywhere else.
|
|
|
|
A renderer does not reliably read such a link as one, and the title rule
|
|
(`kb/CONTRACT.md` § Titles are identifiers) has no room for it - so it is
|
|
reported on its own, whether or not the folded title names a page.
|
|
"""
|
|
return [
|
|
normalize_link_target(m.group(1))
|
|
for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
|
if "\n" in m.group(1)
|
|
]
|
|
|
|
|
|
# `[[Target#Section]]`, `[[Target#Section#Sub|alias]]`: group 1 is the target,
|
|
# exactly as `WIKILINK_RE` reads it; group 2 is everything from the first `#` up
|
|
# to an alias or the closing brackets. A bare self-anchor `[[#Section]]` names
|
|
# no target and is not matched - `WIKILINK_RE` reads no link there either.
|
|
ANCHORED_WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)#([^\]|]*)")
|
|
|
|
# An ATX heading at any level. Optional closing hashes are not part of its text.
|
|
_HEADING_LINE_RE = re.compile(r"^#{1,6}[ \t]+(.+?)(?:[ \t]+#+)?[ \t]*$")
|
|
|
|
|
|
def normalize_anchor(text: str) -> str:
|
|
"""The form a section anchor and a heading are compared in: inline-code
|
|
backticks dropped, whitespace runs folded, case folded.
|
|
|
|
Deliberately loose. An anchor is written by hand from a heading a reader
|
|
saw rendered, so `[[Kunde X#anna müller]]` means the `### Anna Müller`
|
|
section; what `broken_anchors` is for is a section that is *gone*, not one
|
|
spelled with different capitals.
|
|
"""
|
|
return " ".join(text.replace("`", "").split()).casefold()
|
|
|
|
|
|
def heading_anchors(body: str) -> set[str]:
|
|
"""Every heading in this body, at any level, in `normalize_anchor` form.
|
|
|
|
Code-aware like `toc.iter_headings`: detection runs on the masked body, so
|
|
a `# comment` line inside a fence is not a heading, and the text is read
|
|
back from the unmasked line at the same position.
|
|
"""
|
|
original = body.split("\n")
|
|
masked = strip_code_spans(body).split("\n")
|
|
anchors: set[str] = set()
|
|
for masked_line, original_line in zip(masked, original):
|
|
if not masked_line.startswith("#") or not _HEADING_LINE_RE.match(masked_line):
|
|
continue
|
|
match = _HEADING_LINE_RE.match(original_line)
|
|
if match:
|
|
anchors.add(normalize_anchor(match.group(1)))
|
|
return anchors
|
|
|
|
|
|
def anchored_wikilinks(body: str) -> list[tuple[str, str]]:
|
|
"""`(target, anchor)` for every wikilink in this body that names a
|
|
section, in order of appearance, code masked out as everywhere else.
|
|
|
|
The target is normalized like every other reader's; the anchor is returned
|
|
as written (`A#B` for a nested one), so a report can quote it - compare it
|
|
through `normalize_anchor`, segment by segment.
|
|
"""
|
|
return [
|
|
(normalize_link_target(m.group(1)), m.group(2).strip())
|
|
for m in ANCHORED_WIKILINK_RE.finditer(strip_code_spans(body))
|
|
if m.group(2).strip()
|
|
]
|
|
|
|
|
|
def count_wikilinks(body: str) -> Counter[str]:
|
|
"""How often this body links to each page.
|
|
|
|
The counting sibling of `extract_wikilinks`, and the reason it exists: a
|
|
page citing `[[X]]` twice that comes back citing it once has the same link
|
|
*set* and a different link *multiset*. Three of the four defects found in
|
|
the 248-page German translation were exactly that shape, and a set-based
|
|
comparison reported all three as clean.
|
|
"""
|
|
return Counter(
|
|
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
|
)
|
|
|
|
|
|
def find_nested_pages(kb_dir: Path, pages: dict[str, Page]) -> list[tuple[str, Page, int]]:
|
|
"""(title, page, depth) for every page sitting more than one directory
|
|
below its collection.
|
|
|
|
`kb/<collection>/<page>.md` and `kb/<collection>/<area>/<page>.md` are the
|
|
only two depths `kb/CONTRACT.md` § Collections describes. A third level is
|
|
not merely unconventional - it is invisible to the catalog:
|
|
`index_build.group_pages` reads exactly `parts[0]`/`parts[1]` and folds
|
|
anything past them into the area's table silently (Gitea #57), so a page
|
|
down here renders as if it sat directly in the area, under no name of its
|
|
own. `depth` is how many directories separate the page from its
|
|
collection root (1 = directly in an area, the deepest that is not this
|
|
finding).
|
|
"""
|
|
found = []
|
|
for title, page in sorted(pages.items()):
|
|
try:
|
|
parts = page.path.relative_to(kb_dir).parts
|
|
except ValueError:
|
|
continue
|
|
depth = len(parts) - 2 # collection + filename are always present
|
|
if depth > 1:
|
|
found.append((title, page, depth))
|
|
return found
|
|
|
|
|
|
def build_link_graph(pages: dict[str, Page]) -> dict[str, set[str]]:
|
|
"""Map each page title to the set of titles it links to."""
|
|
return {title: extract_wikilinks(page.body) for title, page in pages.items()}
|
|
|
|
|
|
def inbound_links(graph: dict[str, set[str]]) -> dict[str, set[str]]:
|
|
"""Map each page title to the set of titles that link to it."""
|
|
inbound: dict[str, set[str]] = {title: set() for title in graph}
|
|
for source, targets in graph.items():
|
|
for target in targets:
|
|
if target in inbound:
|
|
inbound[target].add(source)
|
|
return inbound
|