fix: a wikilink wrapped across a line break is its own lint finding; rename and rm see it (#115)
Files changed: - CHANGES.md - VERSION - instructions/wiki-lint/SKILL.md - kb/CONTRACT.md - tools/CONTRACT.md - tools/chemenu/commands/lint.py - tools/chemenu/commands/page_ops.py - tools/chemenu/kb_scan.py - tools/chemenu/lint_core.py - tools/chemenu/tests/test_kb_scan.py - tools/chemenu/tests/test_lint.py - tools/chemenu/tests/test_page_ops.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
This commit is contained in:
1 parent
b3022b8ffb
commit
c261b8f4ca
12 files changed
+204
-13
No files matched your search
@@ -12,6 +12,25 @@ from chemenu.page import Page
|
||||
|
||||
WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)")
|
||||
|
||||
# A line break inside `[[...]]`, with the indentation around it. The target
|
||||
# class above admits a newline, so a link someone wrapped at a fixed column -
|
||||
# `[[Foo Bar\n Target]]` - captures the break as part of the title, matches no
|
||||
# page, and read as a missing one to `lint`, as no reference at all to `rm`'s
|
||||
# inbound check, and as nothing to repoint to `rename`.
|
||||
_LINE_BREAK_RUN = re.compile(r"[ \t]*(?:\r?\n[ \t]*)+")
|
||||
|
||||
|
||||
def normalize_link_target(raw: str) -> str:
|
||||
"""The title a captured wikilink target names: every whitespace run that
|
||||
contains a line break folded to one space, then stripped.
|
||||
|
||||
Every reader of a body wikilink goes through this, so `lint`, the link
|
||||
graph, `rename` and `rm` agree on what a wrapped link points at. That the
|
||||
link is wrapped at all is still a finding - `wrapped_wikilinks` - because
|
||||
folding it here would otherwise make it invisible.
|
||||
"""
|
||||
return _LINE_BREAK_RUN.sub(" ", raw).strip()
|
||||
|
||||
|
||||
# Root-level files under kb/ that are not pages: the generated catalog map, log
|
||||
# and provenance index, plus the two documents that constrain the tree rather
|
||||
@@ -105,7 +124,24 @@ def extract_wikilinks(body: str) -> set[str]:
|
||||
notation, and counting it made a page that documents the wiki look like it
|
||||
linked to something that need not exist.
|
||||
"""
|
||||
return {m.group(1).strip() for m in WIKILINK_RE.finditer(strip_code_spans(body))}
|
||||
return {
|
||||
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
||||
}
|
||||
|
||||
|
||||
def wrapped_wikilinks(body: str) -> list[str]:
|
||||
"""The normalized targets of every wikilink in this body written across a
|
||||
line break, in order of appearance, code masked out as everywhere else.
|
||||
|
||||
A renderer does not reliably read such a link as one, and the title rule
|
||||
(`kb/CONTRACT.md` § Titles are identifiers) has no room for it - so it is
|
||||
reported on its own, whether or not the folded title names a page.
|
||||
"""
|
||||
return [
|
||||
normalize_link_target(m.group(1))
|
||||
for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
||||
if "\n" in m.group(1)
|
||||
]
|
||||
|
||||
|
||||
def count_wikilinks(body: str) -> Counter[str]:
|
||||
@@ -117,7 +153,9 @@ def count_wikilinks(body: str) -> Counter[str]:
|
||||
the 248-page German translation were exactly that shape, and a set-based
|
||||
comparison reported all three as clean.
|
||||
"""
|
||||
return Counter(m.group(1).strip() for m in WIKILINK_RE.finditer(strip_code_spans(body)))
|
||||
return Counter(
|
||||
normalize_link_target(m.group(1)) for m in WIKILINK_RE.finditer(strip_code_spans(body))
|
||||
)
|
||||
|
||||
|
||||
def find_nested_pages(kb_dir: Path, pages: dict[str, Page]) -> list[tuple[str, Page, int]]:
|
||||
|
||||
Reference in new issue
Block a user