fix: a wikilink wrapped across a line break is its own lint finding; rename and rm see it (#115)
CI / verify (push) Successful in 5m15s
CI / pwsh (push) Successful in 2m1s
Release / release (push) Successful in 35s

Files changed:
- CHANGES.md
- VERSION
- instructions/wiki-lint/SKILL.md
- kb/CONTRACT.md
- tools/CONTRACT.md
- tools/chemenu/commands/lint.py
- tools/chemenu/commands/page_ops.py
- tools/chemenu/kb_scan.py
- tools/chemenu/lint_core.py
- tools/chemenu/tests/test_kb_scan.py
- tools/chemenu/tests/test_lint.py
- tools/chemenu/tests/test_page_ops.py

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
This commit is contained in:
torbenandClaude Opus 5.5 committed 2026-10-03 11:19:10 +02:00
1 parent b3022b8ffb
commit c261b8f4ca
12 files changed
+204 -13

No files matched your search

+40 -2
View File
@@ -12,6 +12,25 @@ from chemenu.page import Page
WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)")
# A line break inside `[[...]]`, with the indentation around it. The target
# class above admits a newline, so a link someone wrapped at a fixed column -
# `[[Foo Bar\n Target]]` - captures the break as part of the title, matches no
# page, and read as a missing one to `lint`, as no reference at all to `rm`'s
# inbound check, and as nothing to repoint to `rename`.
_LINE_BREAK_RUN = re.compile(r"[ \t]*(?:\r?\n[ \t]*)+")
def normalize_link_target(raw: str) -> str:
"""The title a captured wikilink target names: every whitespace run that
contains a line break folded to one space, then stripped.
Every reader of a body wikilink goes through this, so `lint`, the link
graph, `rename` and `rm` agree on what a wrapped link points at. That the
link is wrapped at all is still a finding - `wrapped_wikilinks` - because
folding it here would otherwise make it invisible.
"""
return _LINE_BREAK_RUN.sub(" ", raw).strip()
# Root-level files under kb/ that are not pages: the generated catalog map, log
# and provenance index, plus the two documents that constrain the tree rather
@@ -105,7 +124,24 @@ def extract_wikilinks(body: str) -> set[str]:
notation, and counting it made a page that documents the wiki look like it
linked to something that need not exist.
"""
return {m.group(1).strip() for m in WIKILINK_RE.finditer(strip_code_spans(body))}
return {
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
}
def wrapped_wikilinks(body: str) -> list[str]:
"""The normalized targets of every wikilink in this body written across a
line break, in order of appearance, code masked out as everywhere else.
A renderer does not reliably read such a link as one, and the title rule
(`kb/CONTRACT.md` § Titles are identifiers) has no room for it - so it is
reported on its own, whether or not the folded title names a page.
"""
return [
normalize_link_target(m.group(1))
for m in WIKILINK_RE.finditer(strip_code_spans(body))
if "\n" in m.group(1)
]
def count_wikilinks(body: str) -> Counter[str]:
@@ -117,7 +153,9 @@ def count_wikilinks(body: str) -> Counter[str]:
the 248-page German translation were exactly that shape, and a set-based
comparison reported all three as clean.
"""
return Counter(m.group(1).strip() for m in WIKILINK_RE.finditer(strip_code_spans(body)))
return Counter(
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
)
def find_nested_pages(kb_dir: Path, pages: dict[str, Page]) -> list[tuple[str, Page, int]]: