Chemenu 2.1.0 - deterministischer Wissenskompiler
Chemenu kompiliert Rohnotizen zu einem verlinkten, quellengebundenen Wiki: raw/ -> types/ + tools/ -> kb/ -> reports/. Was mechanisch ist, macht tools/wikitool; was Urteil braucht, macht ein Agent unter Contracts, deren Grenzen in Code durchgesetzt sind statt im Prompt. Dieser Commit ist der Startpunkt der oeffentlichen Historie. Die vorherige Entwicklung fand in einer privaten Instanz statt und ist nicht Teil dieses Repositorys; ihre Erzaehlung steht vollstaendig in CHANGES.md, das mit 44 Eintraegen von 0.1.0 bis 2.1.0 erhalten geblieben ist. Der mitgelieferte Korpus ist ein Testbett und eine Demo: 170 Seiten ueber den Stack selbst - Gates, Lint, Versionierung, Suche, das Wiki-Muster. Er dokumentiert das Werkzeug mit den eigenen Mitteln des Werkzeugs. Lizenz: AGPL-3.0 fuer den Stack (tools/, types/), CC-BY-4.0 fuer die Inhalte. Die Grenze zwischen beiden ist der Dateiplan, den dist export berechnet - siehe NOTICE.
This commit is contained in:
commit
18ae28f918
368 files changed
+50628
No files matched your search
@@ -0,0 +1,392 @@
|
||||
"""Raw-file <-> wiki provenance tracking.
|
||||
|
||||
This module answers two directions of the same question:
|
||||
- given a raw file, which source page(s) claim to cover it, and which wiki
|
||||
pages cite that source (via frontmatter `sources:` or an inline
|
||||
`[^cite-id]` footnote marker)?
|
||||
- given a wiki page, which source pages does it cite, and which raw files
|
||||
back those sources?
|
||||
|
||||
`raw_files:` is the modern, list-valued frontmatter field on source pages
|
||||
(added by this module's tooling). For backward compatibility we also read the
|
||||
legacy scalar `source:` field when its value looks like a real, existing
|
||||
repo-relative path (not a URL, not a directory) - this lets old pages keep
|
||||
working until they are migrated.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from chemenu import config, sections
|
||||
from chemenu.markdown_code import strip_code_spans
|
||||
from chemenu.page import Page
|
||||
|
||||
# Real GFM footnotes: an inline `[^cite-id]` marker, resolved through a
|
||||
# tool-owned `[^cite-id]: [[Source - X]]` (or `[[Source - X|file.md]]`)
|
||||
# definition line - see cite_id() below for how the id is derived, and
|
||||
# split_cite_block()/render_cite_block() for the definitions block itself.
|
||||
#
|
||||
# CITE_REF_RE deliberately does *not* try to exclude a definition line's own
|
||||
# `[^id]` by pattern (e.g. "not followed by `:`") - prose legitimately
|
||||
# contains a reference immediately before a colon ("Examples[^id]:"), which
|
||||
# such a guard would misparse as a definition and silently drop. The real
|
||||
# distinction is structural, not textual: a definition only ever exists
|
||||
# inside the trailing Footnotes block (see CITE_BLOCK_HEADING), so every
|
||||
# caller scans split_cite_block()'s `head` half only, never the block or the
|
||||
# raw, unsplit body - and scans it through iter_cite_refs() rather than with
|
||||
# this pattern directly, so that code is masked out first.
|
||||
_CITE_ID_PATTERN = r"[A-Za-z0-9][A-Za-z0-9-]*"
|
||||
CITE_DEF_RE = re.compile(
|
||||
rf"^\[\^({_CITE_ID_PATTERN})\]:[ \t]*\[\[([^\]|#]+)(?:\|([^\]]+))?\]\][ \t]*$",
|
||||
re.MULTILINE,
|
||||
)
|
||||
CITE_REF_RE = re.compile(rf"\[\^({_CITE_ID_PATTERN})\]")
|
||||
|
||||
# Where the Footnotes block stops: the next ATX heading of any level. Without
|
||||
# this the block ran to the end of the file and took any following section with
|
||||
# it - see split_cite_block().
|
||||
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
|
||||
|
||||
# The pre-migration marker: `^[[Source - X]]` or `^[[Source - X|file.md]]`,
|
||||
# read by a Pandoc-style parser as an inline footnote wrapping a broken
|
||||
# shortcut link. Kept only so `lint` can flag any that were missed by the
|
||||
# migration - see legacy_citation_markers() in commands/lint.py.
|
||||
LEGACY_CITE_RE = re.compile(r"\^\[\[([^\]|#]+)(?:\|([^\]]+))?\]\]")
|
||||
|
||||
# The tool-owned block holding every `[^cite-id]: [[...]]` definition for a
|
||||
# page, always the last section in the body. GFM and Obsidian both render
|
||||
# footnote definitions regardless of the heading text; this heading is purely
|
||||
# for human readability when the raw markdown is read directly.
|
||||
#
|
||||
# Written under the canonical name, but split_cite_block() matches the aliases
|
||||
# too - a page whose block still says "## Footnotes" keeps working until it is
|
||||
# translated. See chemenu/sections.py.
|
||||
CITE_BLOCK_HEADING = f"## {sections.FOOTNOTES}"
|
||||
|
||||
_SOURCE_TITLE_PREFIX = "Source - "
|
||||
|
||||
|
||||
|
||||
def iter_cite_refs(text: str):
|
||||
"""Every *real* `[^cite-id]` reference in `text`, code masked out.
|
||||
|
||||
The one entry point for reference scanning, and the reason
|
||||
`CITE_REF_RE.finditer()` should not be called directly on a page body: a
|
||||
page that writes the notation inside backticks or a fenced block is
|
||||
describing it, not citing anything, and `undefined_footnote_refs` is a hard
|
||||
error. See markdown_code.strip_code_spans() for what that masking covers.
|
||||
|
||||
Offsets survive the masking, so a caller may still use `m.start()` against
|
||||
the text it passed in.
|
||||
"""
|
||||
return CITE_REF_RE.finditer(strip_code_spans(text))
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
"""Transliterate to ASCII, then reduce to `[a-z0-9]` runs joined by `-`."""
|
||||
normalized = unicodedata.normalize("NFKD", text)
|
||||
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
|
||||
return re.sub(r"[^A-Za-z0-9]+", "-", ascii_text).strip("-").lower()
|
||||
|
||||
|
||||
def cite_id(title: str, qualifier: Optional[str] = None) -> str:
|
||||
"""Deterministic footnote id for a (source title, optional file
|
||||
qualifier) pair: strip the `Source - ` prefix, transliterate to ASCII,
|
||||
slugify, and join title/qualifier with `--`.
|
||||
|
||||
Pure - always returns the same id for the same inputs, with no knowledge
|
||||
of what ids already exist on a page. Two distinct pairs can collide (an
|
||||
NFKD transliteration is lossy), so callers resolving a real page use
|
||||
unique_cite_id() to add a `-2`/`-3` suffix on collision.
|
||||
"""
|
||||
base_title = title[len(_SOURCE_TITLE_PREFIX):] if title.startswith(_SOURCE_TITLE_PREFIX) else title
|
||||
slug = "s-" + _slugify(base_title)
|
||||
if qualifier:
|
||||
slug += "--" + _slugify(qualifier)
|
||||
return slug if slug != "s-" else "s"
|
||||
|
||||
|
||||
def unique_cite_id(existing_ids: set[str], title: str, qualifier: Optional[str] = None) -> str:
|
||||
"""cite_id(), suffixed with `-2`, `-3`, ... until it is not in
|
||||
`existing_ids`. Callers that want to *reuse* an id already pointing at
|
||||
the same (title, qualifier) pair must check for that themselves before
|
||||
calling this - it only ever returns a free id."""
|
||||
base = cite_id(title, qualifier)
|
||||
if base not in existing_ids:
|
||||
return base
|
||||
suffix = 2
|
||||
while f"{base}-{suffix}" in existing_ids:
|
||||
suffix += 1
|
||||
return f"{base}-{suffix}"
|
||||
|
||||
|
||||
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
||||
"""Split the Footnotes block off `body`.
|
||||
|
||||
Returns (body_without_block, definitions), where definitions maps
|
||||
cite_id -> (source_title, qualifier_or_None) in file order. If there is
|
||||
no Footnotes block, definitions is {} and body is returned with trailing
|
||||
blank lines trimmed (so re-rendering after emptying the block is stable).
|
||||
|
||||
**The block is not "everything to the end of the file".** It used to be,
|
||||
and every caller here reassembles a page as `head + rendered block` - so a
|
||||
section that happened to sit after the block was silently deleted on the
|
||||
next `cite add`, `cite sync` or `rename`. That is not hypothetical: `xref
|
||||
add` appends its Relationships and See Also sections at the end of the
|
||||
file, so whether a page kept its cross-references came down to which of the
|
||||
two commands ran last. Eight pages were carrying content in that position
|
||||
when this was found.
|
||||
|
||||
So the block ends where the next heading begins, and everything after it -
|
||||
plus anything inside it that is not a citation definition - is folded back
|
||||
on to `head`. Nothing is discarded, and because the rendered block is
|
||||
always emitted last, a page that had drifted into the broken layout is
|
||||
normalised the first time any of these commands touches it.
|
||||
"""
|
||||
# Where the block *starts* is decided on the unmasked body, deliberately.
|
||||
# Masking first would mean one unclosed fence anywhere in the prose blanks
|
||||
# the real `## Footnotes` heading too, and the page then reads as having no
|
||||
# definitions at all - every citation on it undefined, from a single typo.
|
||||
# A fenced example of the heading itself is the rarer accident and the
|
||||
# cheaper one: it costs one page its block, not every citation on it.
|
||||
match = sections.heading_re(sections.FOOTNOTES).search(body)
|
||||
if not match:
|
||||
return body.rstrip("\n"), {}
|
||||
head, rest = body[: match.start()], body[match.end():]
|
||||
|
||||
next_section = _NEXT_HEADING_RE.search(rest)
|
||||
block, trailing = (rest[: next_section.start()], rest[next_section.start():]) if next_section else (rest, "")
|
||||
|
||||
# Inside the block, code is masked: a fenced example of a definition line is
|
||||
# an illustration, not a definition. strip_code_spans() preserves offsets
|
||||
# and line structure, so the masked block can be read line-for-line against
|
||||
# the real one.
|
||||
masked_block = strip_code_spans(block)
|
||||
definitions = {
|
||||
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
|
||||
for m in CITE_DEF_RE.finditer(masked_block)
|
||||
}
|
||||
# Lines inside the block that are not definitions are content too - prose
|
||||
# someone left there, a stray bullet. Rescued rather than rejected: this
|
||||
# runs under `lint` and `corpus_diff` as well, where raising would refuse
|
||||
# to read a page instead of reporting it.
|
||||
stray = "\n".join(
|
||||
line
|
||||
for line, masked in zip(block.splitlines(), masked_block.splitlines())
|
||||
if line.strip() and not CITE_DEF_RE.match(masked)
|
||||
)
|
||||
|
||||
rescued = "\n\n".join(part.strip("\n") for part in (stray, trailing) if part.strip())
|
||||
head = head.rstrip("\n")
|
||||
if rescued:
|
||||
head = f"{head}\n\n{rescued}" if head else rescued
|
||||
return head, definitions
|
||||
|
||||
|
||||
def cite_block_heading(body: str) -> str:
|
||||
"""The Footnotes heading `body` actually carries, canonical if it has none.
|
||||
|
||||
Rewriting a page must not silently retitle its block: a page still using an
|
||||
alias is untranslated, not broken, and `cite sync` has to stay a no-op on
|
||||
it. Translating the heading is the migration's job, not the tool's."""
|
||||
match = sections.heading_re(sections.FOOTNOTES).search(body)
|
||||
return match.group(0).strip() if match else CITE_BLOCK_HEADING
|
||||
|
||||
|
||||
def render_cite_block(
|
||||
definitions: dict[str, tuple[str, Optional[str]]], heading: str = CITE_BLOCK_HEADING
|
||||
) -> str:
|
||||
"""Render the Footnotes block for `definitions` (cite_id -> (title,
|
||||
qualifier)), preserving dict order. Empty dict renders "" - a page with
|
||||
no citations carries no block at all."""
|
||||
if not definitions:
|
||||
return ""
|
||||
lines = [heading, ""]
|
||||
for cid, (title, qualifier) in definitions.items():
|
||||
target = f"{title}|{qualifier}" if qualifier else title
|
||||
lines.append(f"[^{cid}]: [[{target}]]")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def render_page_body(
|
||||
head: str,
|
||||
definitions: dict[str, tuple[str, Optional[str]]],
|
||||
heading: str = CITE_BLOCK_HEADING,
|
||||
) -> str:
|
||||
"""Reassemble a page body from its non-Footnotes content and citation
|
||||
definitions - the inverse of split_cite_block(). Pass the original body's
|
||||
`cite_block_heading()` to preserve an alias the page still uses."""
|
||||
head = head.rstrip("\n")
|
||||
block = render_cite_block(definitions, heading)
|
||||
if not block:
|
||||
return head + "\n"
|
||||
return head + "\n\n" + block
|
||||
|
||||
|
||||
def extract_inline_cites(body: str) -> set[tuple[str, Optional[str]]]:
|
||||
"""Return the set of (source_title, file_qualifier_or_None) cited
|
||||
inline: every `[^cite-id]` reference resolved through this body's
|
||||
`[^cite-id]: [[...]]` definitions. A reference with no matching
|
||||
definition resolves to nothing here - see lint's undefined_footnote_refs
|
||||
for that failure mode."""
|
||||
head, definitions = split_cite_block(body)
|
||||
return {definitions[m.group(1)] for m in iter_cite_refs(head) if m.group(1) in definitions}
|
||||
|
||||
|
||||
def _looks_like_repo_path(value: str) -> bool:
|
||||
if not isinstance(value, str) or not value:
|
||||
return False
|
||||
if "://" in value:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def source_raw_files(page: Page) -> list[str]:
|
||||
"""The list of raw-relative paths a source page's frontmatter claims to cover.
|
||||
|
||||
Prefers the modern `raw_files:` list; falls back to the legacy scalar
|
||||
`source:` field if it looks like a repo-relative path (not a URL).
|
||||
"""
|
||||
raw_files = page.frontmatter.get("raw_files")
|
||||
if raw_files:
|
||||
return list(raw_files)
|
||||
legacy = page.frontmatter.get("source")
|
||||
if _looks_like_repo_path(legacy):
|
||||
return [legacy]
|
||||
return []
|
||||
|
||||
|
||||
def source_pages_by_raw_file(pages: dict[str, Page]) -> dict[str, list[str]]:
|
||||
"""Invert source_raw_files() across every source page: raw path -> [source titles]."""
|
||||
result: dict[str, list[str]] = {}
|
||||
for title, page in pages.items():
|
||||
if page.kind != "source":
|
||||
continue
|
||||
for raw_path in source_raw_files(page):
|
||||
result.setdefault(raw_path, []).append(title)
|
||||
return result
|
||||
|
||||
|
||||
def duplicate_raw_file_owners(pages: dict[str, Page]) -> list[dict]:
|
||||
"""Raw files claimed by more than one source page.
|
||||
|
||||
`raw_files:` is a maintenance claim, not a "mentions" relation (see
|
||||
types/source.md, "One raw file, one owner"). Any number of pages may *cite* a
|
||||
source; but with two owners it is undefined which page must be refreshed
|
||||
when the raw file changes, so both rot silently and neither is identifiably
|
||||
the stale one. `uncovered_raw_files()` cannot see this: it only asks whether
|
||||
a raw file is claimed at all, which is why one ingested manual's
|
||||
subtree sat with eight double-owned files unnoticed.
|
||||
"""
|
||||
return [
|
||||
{"raw_file": raw_path, "owners": sorted(titles)}
|
||||
for raw_path, titles in sorted(source_pages_by_raw_file(pages).items())
|
||||
if len(set(titles)) > 1
|
||||
]
|
||||
|
||||
|
||||
def citing_pages(pages: dict[str, Page], source_title: str) -> list[str]:
|
||||
"""Every page (other than the source page itself) that cites source_title,
|
||||
either via frontmatter `sources:` or an inline `^[[source_title]]` marker."""
|
||||
citing = []
|
||||
for title, page in pages.items():
|
||||
if title == source_title:
|
||||
continue
|
||||
if source_title in (page.frontmatter.get("sources") or []):
|
||||
citing.append(title)
|
||||
continue
|
||||
if any(cited == source_title for cited, _file in extract_inline_cites(page.body)):
|
||||
citing.append(title)
|
||||
return sorted(set(citing))
|
||||
|
||||
|
||||
def page_raw_files(pages: dict[str, Page], page: Page) -> list[str]:
|
||||
"""All raw files backing a (non-source) page, via its cited/related source pages."""
|
||||
raw_files: list[str] = []
|
||||
for source_title in page.frontmatter.get("sources") or []:
|
||||
source_page = pages.get(source_title)
|
||||
if source_page is not None:
|
||||
raw_files.extend(source_raw_files(source_page))
|
||||
for cited_title, _file in extract_inline_cites(page.body):
|
||||
source_page = pages.get(cited_title)
|
||||
if source_page is not None:
|
||||
raw_files.extend(source_raw_files(source_page))
|
||||
return list(dict.fromkeys(raw_files))
|
||||
|
||||
|
||||
def uncovered_raw_files(raw_dir: Path, pages: dict[str, Page]) -> list[str]:
|
||||
"""Raw files with no source page claiming to cover them."""
|
||||
covered = set(source_pages_by_raw_file(pages))
|
||||
all_raw = {str(p.relative_to(config.ROOT)) for p in config.iter_raw_files(raw_dir)}
|
||||
return sorted(all_raw - covered)
|
||||
|
||||
|
||||
def broken_raw_refs(pages: dict[str, Page]) -> list[dict]:
|
||||
"""`raw_files:`/legacy `source:` entries that point at a path which doesn't exist."""
|
||||
issues = []
|
||||
for title, page in pages.items():
|
||||
if page.kind != "source":
|
||||
continue
|
||||
for raw_path in source_raw_files(page):
|
||||
if not (config.ROOT / raw_path).exists():
|
||||
issues.append({"page": title, "raw_path": raw_path})
|
||||
return issues
|
||||
|
||||
|
||||
def legacy_citation_markers(pages: dict[str, Page]) -> list[dict]:
|
||||
"""Pages still carrying the pre-migration `^[[Source - X]]` marker
|
||||
instead of a real `[^cite-id]` footnote reference - see LEGACY_CITE_RE."""
|
||||
issues = []
|
||||
for title, page in pages.items():
|
||||
for m in LEGACY_CITE_RE.finditer(strip_code_spans(page.body)):
|
||||
issues.append({"page": title, "marker": m.group(0)})
|
||||
return issues
|
||||
|
||||
|
||||
def undefined_footnote_refs(pages: dict[str, Page]) -> list[dict]:
|
||||
"""`[^id]` references in a page's prose with no matching
|
||||
`[^id]: [[...]]` definition in its Footnotes block - a citation whose
|
||||
`cite add` never ran, or a hand-typed id."""
|
||||
issues = []
|
||||
for title, page in pages.items():
|
||||
head, definitions = split_cite_block(page.body)
|
||||
for m in iter_cite_refs(head):
|
||||
ref_id = m.group(1)
|
||||
if ref_id not in definitions:
|
||||
issues.append({"page": title, "ref": ref_id})
|
||||
return issues
|
||||
|
||||
|
||||
def orphan_footnote_defs(pages: dict[str, Page]) -> list[dict]:
|
||||
"""Footnotes definitions nothing in the page's prose references any
|
||||
more - what `wikitool cite sync` prunes."""
|
||||
issues = []
|
||||
for title, page in pages.items():
|
||||
head, definitions = split_cite_block(page.body)
|
||||
referenced = {m.group(1) for m in iter_cite_refs(head)}
|
||||
for ref_id, (source_title, _qualifier) in definitions.items():
|
||||
if ref_id not in referenced:
|
||||
issues.append({"page": title, "id": ref_id, "source": source_title})
|
||||
return issues
|
||||
|
||||
|
||||
def legacy_source_pages(pages: dict[str, Page]) -> list[dict]:
|
||||
"""Source pages still using a directory-valued or URL-only legacy `source:`
|
||||
field instead of the modern `raw_files:` list."""
|
||||
issues = []
|
||||
for title, page in pages.items():
|
||||
if page.kind != "source":
|
||||
continue
|
||||
if page.frontmatter.get("raw_files"):
|
||||
continue
|
||||
legacy = page.frontmatter.get("source")
|
||||
if not legacy:
|
||||
continue
|
||||
if "://" in str(legacy):
|
||||
issues.append({"page": title, "source": legacy, "reason": "url-only, no raw_files"})
|
||||
elif (config.ROOT / legacy).is_dir():
|
||||
issues.append({"page": title, "source": legacy, "reason": "directory, not a file"})
|
||||
return issues
|
||||
Reference in new issue
Block a user