Files changed: - .gitattributes - CHANGES.md - VERSION - raw/CONTRACT.md - tools/README.md - tools/chemenu/commands/_util.py - tools/chemenu/commands/dist_cmd.py - tools/chemenu/commands/docs_verify.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/eval_cmd.py - tools/chemenu/commands/git_publish.py - tools/chemenu/commands/index_build.py - tools/chemenu/commands/lint.py - tools/chemenu/commands/log_append.py - tools/chemenu/commands/migrate_cmd.py - tools/chemenu/commands/provenance_cmd.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/commands/run_budget.py - tools/chemenu/commands/upstream_cmd.py - tools/chemenu/commands/version_cmd.py - tools/chemenu/commands/work_cmd.py - tools/chemenu/config.py - tools/chemenu/corpus_cache.py - tools/chemenu/filelock.py - tools/chemenu/frontmatter_io.py - tools/chemenu/kb_scan.py - tools/chemenu/kb_state.py - tools/chemenu/lint_core.py - tools/chemenu/prerequisites.py - tools/chemenu/provenance.py - tools/chemenu/search/base.py - tools/chemenu/search/ripgrep.py - tools/chemenu/telemetry/writer.py - tools/chemenu/tests/test_dist_cmd.py - tools/chemenu/tests/test_portability.py - tools/chemenu/tests/test_search.py - tools/chemenu/tests/test_trace_ingest.py - tools/chemenu/type_resolver.py - tools/chemenu/upload.py - tools/chemenu/version.py - tools/run_wikitool.py - tools/trace_ingest.py
422 lines
18 KiB
Python
422 lines
18 KiB
Python
"""Raw-file <-> wiki provenance tracking.
|
|
|
|
This module answers two directions of the same question:
|
|
- given a raw file, which source page(s) claim to cover it, and which wiki
|
|
pages cite that source (via frontmatter `sources:` or an inline
|
|
`[^cite-id]` footnote marker)?
|
|
- given a wiki page, which source pages does it cite, and which raw files
|
|
back those sources?
|
|
|
|
`raw_files:` is the modern, list-valued frontmatter field on source pages
|
|
(added by this module's tooling). For backward compatibility we also read the
|
|
legacy scalar `source:` field when its value looks like a real, existing
|
|
repo-relative path (not a URL, not a directory) - this lets old pages keep
|
|
working until they are migrated.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import unicodedata
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from chemenu import blocks, config, conventions
|
|
from chemenu.markdown_code import strip_code_spans
|
|
from chemenu.page import Page
|
|
|
|
# Real GFM footnotes: an inline `[^cite-id]` marker, resolved through a
|
|
# tool-owned `[^cite-id]: [[Source - X]]` (or `[[Source - X|file.md]]`)
|
|
# definition line - see cite_id() below for how the id is derived, and
|
|
# split_cite_block()/render_cite_block() for the definitions block itself.
|
|
#
|
|
# CITE_REF_RE deliberately does *not* try to exclude a definition line's own
|
|
# `[^id]` by pattern (e.g. "not followed by `:`") - prose legitimately
|
|
# contains a reference immediately before a colon ("Examples[^id]:"), which
|
|
# such a guard would misparse as a definition and silently drop. The real
|
|
# distinction is structural, not textual: a definition only ever exists
|
|
# inside the trailing Footnotes block (see CITE_BLOCK_HEADING), so every
|
|
# caller scans split_cite_block()'s `head` half only, never the block or the
|
|
# raw, unsplit body - and scans it through iter_cite_refs() rather than with
|
|
# this pattern directly, so that code is masked out first.
|
|
_CITE_ID_PATTERN = r"[A-Za-z0-9][A-Za-z0-9-]*"
|
|
CITE_DEF_RE = re.compile(
|
|
rf"^\[\^({_CITE_ID_PATTERN})\]:[ \t]*\[\[([^\]|#]+)(?:\|([^\]]+))?\]\][ \t]*$",
|
|
re.MULTILINE,
|
|
)
|
|
CITE_REF_RE = re.compile(rf"\[\^({_CITE_ID_PATTERN})\]")
|
|
|
|
|
|
# The pre-migration marker: `^[[Source - X]]` or `^[[Source - X|file.md]]`,
|
|
# read by a Pandoc-style parser as an inline footnote wrapping a broken
|
|
# shortcut link. Kept only so `lint` can flag any that were missed by the
|
|
# migration - see legacy_citation_markers() in commands/lint.py.
|
|
LEGACY_CITE_RE = re.compile(r"\^\[\[([^\]|#]+)(?:\|([^\]]+))?\]\]")
|
|
|
|
# The tool-owned block holding every `[^cite-id]: [[...]]` definition for a
|
|
# page, always the last section in the body. GFM and Obsidian both render
|
|
# footnote definitions regardless of the heading text; this heading is purely
|
|
# for human readability when the raw markdown is read directly.
|
|
#
|
|
# The prefix a source page's title carries, stripped when minting a cite id so
|
|
# the id is not "s-source-x". It is the `source` type-spec's own
|
|
# `title_prefix:`, asked for at call time rather than written down here: the
|
|
# type-spec belongs to the instance, so hardcoding the string made a documented
|
|
# instance decision into a compiler constant - the same leak `sections.py` had.
|
|
#
|
|
# The literal survives as the fallback for a tree with no resolvable `source`
|
|
# type (a fixture, a half-built instance). It is what this stack shipped, so a
|
|
# corpus that can reach the fallback was minted under it, and ids stay stable.
|
|
_FALLBACK_SOURCE_TITLE_PREFIX = "Source - "
|
|
|
|
|
|
def source_title_prefix() -> str:
|
|
"""This instance's source-page title prefix, from the type-spec."""
|
|
from chemenu.type_resolver import resolver
|
|
|
|
try:
|
|
type_path = resolver.find_type_by_name("source")
|
|
if type_path:
|
|
return resolver.get_title_prefix(type_path)
|
|
except (ValueError, OSError):
|
|
pass
|
|
return _FALLBACK_SOURCE_TITLE_PREFIX
|
|
|
|
|
|
|
|
def iter_cite_refs(text: str):
|
|
"""Every *real* `[^cite-id]` reference in `text`, code masked out.
|
|
|
|
The one entry point for reference scanning, and the reason
|
|
`CITE_REF_RE.finditer()` should not be called directly on a page body: a
|
|
page that writes the notation inside backticks or a fenced block is
|
|
describing it, not citing anything, and `undefined_footnote_refs` is a hard
|
|
error. See markdown_code.strip_code_spans() for what that masking covers.
|
|
|
|
Offsets survive the masking, so a caller may still use `m.start()` against
|
|
the text it passed in.
|
|
"""
|
|
return CITE_REF_RE.finditer(strip_code_spans(text))
|
|
|
|
|
|
def _slugify(text: str) -> str:
|
|
"""Transliterate to ASCII, then reduce to `[a-z0-9]` runs joined by `-`."""
|
|
normalized = unicodedata.normalize("NFKD", text)
|
|
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
|
|
return re.sub(r"[^A-Za-z0-9]+", "-", ascii_text).strip("-").lower()
|
|
|
|
|
|
def cite_id(title: str, qualifier: Optional[str] = None) -> str:
|
|
"""Deterministic footnote id for a (source title, optional file
|
|
qualifier) pair: strip the `Source - ` prefix, transliterate to ASCII,
|
|
slugify, and join title/qualifier with `--`.
|
|
|
|
Pure - always returns the same id for the same inputs, with no knowledge
|
|
of what ids already exist on a page. Two distinct pairs can collide (an
|
|
NFKD transliteration is lossy), so callers resolving a real page use
|
|
unique_cite_id() to add a `-2`/`-3` suffix on collision.
|
|
"""
|
|
prefix = source_title_prefix()
|
|
base_title = title[len(prefix):] if prefix and title.startswith(prefix) else title
|
|
slug = "s-" + _slugify(base_title)
|
|
if qualifier:
|
|
slug += "--" + _slugify(qualifier)
|
|
return slug if slug != "s-" else "s"
|
|
|
|
|
|
def unique_cite_id(existing_ids: set[str], title: str, qualifier: Optional[str] = None) -> str:
|
|
"""cite_id(), suffixed with `-2`, `-3`, ... until it is not in
|
|
`existing_ids`. Callers that want to *reuse* an id already pointing at
|
|
the same (title, qualifier) pair must check for that themselves before
|
|
calling this - it only ever returns a free id."""
|
|
base = cite_id(title, qualifier)
|
|
if base not in existing_ids:
|
|
return base
|
|
suffix = 2
|
|
while f"{base}-{suffix}" in existing_ids:
|
|
suffix += 1
|
|
return f"{base}-{suffix}"
|
|
|
|
|
|
# Headings a pre-4.0.0 page carries above its citation definitions, for the
|
|
# migration window only. Before the block was delimited it was *located* by this
|
|
# text, which is why there are four of them - two languages times two eras. The
|
|
# list is read, never written, and `instructions/migrations/` removes the need
|
|
# for it once every page carries markers.
|
|
_LEGACY_FOOTNOTE_HEADINGS = ("Fußnoten", "Footnotes", "Fussnoten", "Notes")
|
|
|
|
_LEGACY_HEADING_RE = re.compile(
|
|
r"^## (?:" + "|".join(re.escape(name) for name in _LEGACY_FOOTNOTE_HEADINGS) + r")[ \t]*$",
|
|
re.MULTILINE,
|
|
)
|
|
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
|
|
|
|
|
|
def _definitions_in(block: str) -> dict[str, tuple[str, Optional[str]]]:
|
|
"""Every `[^id]: [[Target]]` definition in one region, code masked out.
|
|
|
|
A fenced example of a definition line is an illustration, not a definition.
|
|
`strip_code_spans` preserves offsets and line structure, so the masked text
|
|
reads line-for-line against the real one.
|
|
"""
|
|
masked = strip_code_spans(block)
|
|
return {
|
|
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
|
|
for m in CITE_DEF_RE.finditer(masked)
|
|
}
|
|
|
|
|
|
def _split_legacy_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
|
"""The pre-marker layout: a heading, then definitions, ending at the next
|
|
heading.
|
|
|
|
Kept only so the corpus stays readable between this machinery landing and
|
|
the migration reaching each page. Every weakness of the old approach lives
|
|
here - it guesses the region's end, and it can be fooled by a fenced example
|
|
of the heading - which is the argument the marker pair settles.
|
|
"""
|
|
match = _LEGACY_HEADING_RE.search(body)
|
|
if not match:
|
|
return body.rstrip("\n"), {}
|
|
head, rest = body[: match.start()], body[match.end():]
|
|
|
|
following = _NEXT_HEADING_RE.search(rest)
|
|
block, trailing = (
|
|
(rest[: following.start()], rest[following.start():]) if following else (rest, "")
|
|
)
|
|
|
|
definitions = _definitions_in(block)
|
|
masked_block = strip_code_spans(block)
|
|
# Lines inside the block that are not definitions are content too - prose
|
|
# someone left there, a stray bullet. Rescued rather than rejected: this runs
|
|
# under `lint` and `corpus_diff` as well, where raising would refuse to read
|
|
# a page instead of reporting it.
|
|
stray = "\n".join(
|
|
line
|
|
for line, masked in zip(block.splitlines(), masked_block.splitlines())
|
|
if line.strip() and not CITE_DEF_RE.match(masked)
|
|
)
|
|
rescued = "\n\n".join(part.strip("\n") for part in (stray, trailing) if part.strip())
|
|
head = head.rstrip("\n")
|
|
if rescued:
|
|
head = f"{head}\n\n{rescued}" if head else rescued
|
|
return head, definitions
|
|
|
|
|
|
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
|
"""Split the citation region off `body`.
|
|
|
|
Returns (body_without_region, definitions), where definitions maps
|
|
cite_id -> (source_title, qualifier_or_None) in file order.
|
|
|
|
**The region is delimited, not guessed.** It used to end "at the next
|
|
heading", and before that "at the end of the file" - and every caller here
|
|
reassembles a page as `head + rendered region`, so a section that happened to
|
|
sit after it was silently deleted on the next `cite add`, `cite sync` or
|
|
`rename`. Eight pages were carrying content in that position when it was
|
|
found. A marker pair answers where the region stops exactly, which is the
|
|
whole reason for it.
|
|
|
|
A page with no markers is read through the legacy path instead, so the
|
|
corpus stays readable until the migration reaches it.
|
|
"""
|
|
region = blocks.find(body, blocks.FOOTNOTES)
|
|
if region is None:
|
|
return _split_legacy_block(body)
|
|
return blocks.strip(body, blocks.FOOTNOTES).rstrip("\n"), _definitions_in(region)
|
|
|
|
|
|
def render_cite_block(definitions: dict[str, tuple[str, Optional[str]]]) -> str:
|
|
"""The citation region for `definitions`, markers included, in dict order.
|
|
|
|
An empty dict renders "" - a page with no citations carries no region at
|
|
all, rather than a heading with nothing under it.
|
|
"""
|
|
lines = []
|
|
for cid, (title, qualifier) in definitions.items():
|
|
target = f"{title}|{qualifier}" if qualifier else title
|
|
lines.append(f"[^{cid}]: [[{target}]]")
|
|
return blocks.render(
|
|
blocks.FOOTNOTES, conventions.heading(blocks.FOOTNOTES), lines
|
|
)
|
|
|
|
|
|
def render_page_body(
|
|
head: str, definitions: dict[str, tuple[str, Optional[str]]]
|
|
) -> str:
|
|
"""Reassemble a page body from its non-citation content and its definitions -
|
|
the inverse of `split_cite_block`.
|
|
|
|
The heading is no longer threaded through from the caller. It used to be, so
|
|
that rewriting a page would not silently retitle a block whose text the tool
|
|
was *matching on*; now the marker carries the identity and the heading is a
|
|
rendering value, so re-rendering it under this instance's own words is a
|
|
repair rather than a rename.
|
|
"""
|
|
return blocks.replace(head.rstrip("\n") + "\n", blocks.FOOTNOTES, render_cite_block(definitions))
|
|
|
|
|
|
def extract_inline_cites(body: str) -> set[tuple[str, Optional[str]]]:
|
|
"""Return the set of (source_title, file_qualifier_or_None) cited
|
|
inline: every `[^cite-id]` reference resolved through this body's
|
|
`[^cite-id]: [[...]]` definitions. A reference with no matching
|
|
definition resolves to nothing here - see lint's undefined_footnote_refs
|
|
for that failure mode."""
|
|
head, definitions = split_cite_block(body)
|
|
return {definitions[m.group(1)] for m in iter_cite_refs(head) if m.group(1) in definitions}
|
|
|
|
|
|
def _looks_like_repo_path(value: str) -> bool:
|
|
if not isinstance(value, str) or not value:
|
|
return False
|
|
if "://" in value:
|
|
return False
|
|
return True
|
|
|
|
|
|
def source_raw_files(page: Page) -> list[str]:
|
|
"""The list of raw-relative paths a source page's frontmatter claims to cover.
|
|
|
|
Prefers the modern `raw_files:` list; falls back to the legacy scalar
|
|
`source:` field if it looks like a repo-relative path (not a URL).
|
|
"""
|
|
raw_files = page.frontmatter.get("raw_files")
|
|
if raw_files:
|
|
return list(raw_files)
|
|
legacy = page.frontmatter.get("source")
|
|
if _looks_like_repo_path(legacy):
|
|
return [legacy]
|
|
return []
|
|
|
|
|
|
def source_pages_by_raw_file(pages: dict[str, Page]) -> dict[str, list[str]]:
|
|
"""Invert source_raw_files() across every source page: raw path -> [source titles]."""
|
|
result: dict[str, list[str]] = {}
|
|
for title, page in pages.items():
|
|
if page.kind != "source":
|
|
continue
|
|
for raw_path in source_raw_files(page):
|
|
result.setdefault(raw_path, []).append(title)
|
|
return result
|
|
|
|
|
|
def duplicate_raw_file_owners(pages: dict[str, Page]) -> list[dict]:
|
|
"""Raw files claimed by more than one source page.
|
|
|
|
`raw_files:` is a maintenance claim, not a "mentions" relation (see
|
|
types/source.md, "One raw file, one owner"). Any number of pages may *cite* a
|
|
source; but with two owners it is undefined which page must be refreshed
|
|
when the raw file changes, so both rot silently and neither is identifiably
|
|
the stale one. `uncovered_raw_files()` cannot see this: it only asks whether
|
|
a raw file is claimed at all, which is why one ingested manual's
|
|
subtree sat with eight double-owned files unnoticed.
|
|
"""
|
|
return [
|
|
{"raw_file": raw_path, "owners": sorted(titles)}
|
|
for raw_path, titles in sorted(source_pages_by_raw_file(pages).items())
|
|
if len(set(titles)) > 1
|
|
]
|
|
|
|
|
|
def citing_pages(pages: dict[str, Page], source_title: str) -> list[str]:
|
|
"""Every page (other than the source page itself) that cites source_title,
|
|
either via frontmatter `sources:` or an inline `^[[source_title]]` marker."""
|
|
citing = []
|
|
for title, page in pages.items():
|
|
if title == source_title:
|
|
continue
|
|
if source_title in (page.frontmatter.get("sources") or []):
|
|
citing.append(title)
|
|
continue
|
|
if any(cited == source_title for cited, _file in extract_inline_cites(page.body)):
|
|
citing.append(title)
|
|
return sorted(set(citing))
|
|
|
|
|
|
def page_raw_files(pages: dict[str, Page], page: Page) -> list[str]:
|
|
"""All raw files backing a (non-source) page, via its cited/related source pages."""
|
|
raw_files: list[str] = []
|
|
for source_title in page.frontmatter.get("sources") or []:
|
|
source_page = pages.get(source_title)
|
|
if source_page is not None:
|
|
raw_files.extend(source_raw_files(source_page))
|
|
for cited_title, _file in extract_inline_cites(page.body):
|
|
source_page = pages.get(cited_title)
|
|
if source_page is not None:
|
|
raw_files.extend(source_raw_files(source_page))
|
|
return list(dict.fromkeys(raw_files))
|
|
|
|
|
|
def uncovered_raw_files(raw_dir: Path, pages: dict[str, Page]) -> list[str]:
|
|
"""Raw files with no source page claiming to cover them."""
|
|
covered = set(source_pages_by_raw_file(pages))
|
|
all_raw = {p.relative_to(config.ROOT).as_posix() for p in config.iter_raw_files(raw_dir)}
|
|
return sorted(all_raw - covered)
|
|
|
|
|
|
def broken_raw_refs(pages: dict[str, Page]) -> list[dict]:
|
|
"""`raw_files:`/legacy `source:` entries that point at a path which doesn't exist."""
|
|
issues = []
|
|
for title, page in pages.items():
|
|
if page.kind != "source":
|
|
continue
|
|
for raw_path in source_raw_files(page):
|
|
if not (config.ROOT / raw_path).exists():
|
|
issues.append({"page": title, "raw_path": raw_path})
|
|
return issues
|
|
|
|
|
|
def legacy_citation_markers(pages: dict[str, Page]) -> list[dict]:
|
|
"""Pages still carrying the pre-migration `^[[Source - X]]` marker
|
|
instead of a real `[^cite-id]` footnote reference - see LEGACY_CITE_RE."""
|
|
issues = []
|
|
for title, page in pages.items():
|
|
for m in LEGACY_CITE_RE.finditer(strip_code_spans(page.body)):
|
|
issues.append({"page": title, "marker": m.group(0)})
|
|
return issues
|
|
|
|
|
|
def undefined_footnote_refs(pages: dict[str, Page]) -> list[dict]:
|
|
"""`[^id]` references in a page's prose with no matching
|
|
`[^id]: [[...]]` definition in its Footnotes block - a citation whose
|
|
`cite add` never ran, or a hand-typed id."""
|
|
issues = []
|
|
for title, page in pages.items():
|
|
head, definitions = split_cite_block(page.body)
|
|
for m in iter_cite_refs(head):
|
|
ref_id = m.group(1)
|
|
if ref_id not in definitions:
|
|
issues.append({"page": title, "ref": ref_id})
|
|
return issues
|
|
|
|
|
|
def orphan_footnote_defs(pages: dict[str, Page]) -> list[dict]:
|
|
"""Footnotes definitions nothing in the page's prose references any
|
|
more - what `wikitool cite sync` prunes."""
|
|
issues = []
|
|
for title, page in pages.items():
|
|
head, definitions = split_cite_block(page.body)
|
|
referenced = {m.group(1) for m in iter_cite_refs(head)}
|
|
for ref_id, (source_title, _qualifier) in definitions.items():
|
|
if ref_id not in referenced:
|
|
issues.append({"page": title, "id": ref_id, "source": source_title})
|
|
return issues
|
|
|
|
|
|
def legacy_source_pages(pages: dict[str, Page]) -> list[dict]:
|
|
"""Source pages still using a directory-valued or URL-only legacy `source:`
|
|
field instead of the modern `raw_files:` list."""
|
|
issues = []
|
|
for title, page in pages.items():
|
|
if page.kind != "source":
|
|
continue
|
|
if page.frontmatter.get("raw_files"):
|
|
continue
|
|
legacy = page.frontmatter.get("source")
|
|
if not legacy:
|
|
continue
|
|
if "://" in str(legacy):
|
|
issues.append({"page": title, "source": legacy, "reason": "url-only, no raw_files"})
|
|
elif (config.ROOT / legacy).is_dir():
|
|
issues.append({"page": title, "source": legacy, "reason": "directory, not a file"})
|
|
return issues
|