Files
chemenu/tools/chemenu/provenance.py
T
torben d0f08d1fba
CI / verify (push) Successful in 2m42s
CI / pwsh (push) Successful in 1m55s
Release / release (push) Successful in 36s
feat: Windows portability - path separators, LF line endings, UTF-8 decoding and output, msvcrt lock fallback (#152)
Files changed:
- .gitattributes
- CHANGES.md
- VERSION
- raw/CONTRACT.md
- tools/README.md
- tools/chemenu/commands/_util.py
- tools/chemenu/commands/dist_cmd.py
- tools/chemenu/commands/docs_verify.py
- tools/chemenu/commands/doctor.py
- tools/chemenu/commands/eval_cmd.py
- tools/chemenu/commands/git_publish.py
- tools/chemenu/commands/index_build.py
- tools/chemenu/commands/lint.py
- tools/chemenu/commands/log_append.py
- tools/chemenu/commands/migrate_cmd.py
- tools/chemenu/commands/provenance_cmd.py
- tools/chemenu/commands/raw_cmd.py
- tools/chemenu/commands/run_budget.py
- tools/chemenu/commands/upstream_cmd.py
- tools/chemenu/commands/version_cmd.py
- tools/chemenu/commands/work_cmd.py
- tools/chemenu/config.py
- tools/chemenu/corpus_cache.py
- tools/chemenu/filelock.py
- tools/chemenu/frontmatter_io.py
- tools/chemenu/kb_scan.py
- tools/chemenu/kb_state.py
- tools/chemenu/lint_core.py
- tools/chemenu/prerequisites.py
- tools/chemenu/provenance.py
- tools/chemenu/search/base.py
- tools/chemenu/search/ripgrep.py
- tools/chemenu/telemetry/writer.py
- tools/chemenu/tests/test_dist_cmd.py
- tools/chemenu/tests/test_portability.py
- tools/chemenu/tests/test_search.py
- tools/chemenu/tests/test_trace_ingest.py
- tools/chemenu/type_resolver.py
- tools/chemenu/upload.py
- tools/chemenu/version.py
- tools/run_wikitool.py
- tools/trace_ingest.py
2026-10-01 21:15:03 +02:00

422 lines
18 KiB
Python

"""Raw-file <-> wiki provenance tracking.
This module answers two directions of the same question:
- given a raw file, which source page(s) claim to cover it, and which wiki
pages cite that source (via frontmatter `sources:` or an inline
`[^cite-id]` footnote marker)?
- given a wiki page, which source pages does it cite, and which raw files
back those sources?
`raw_files:` is the modern, list-valued frontmatter field on source pages
(added by this module's tooling). For backward compatibility we also read the
legacy scalar `source:` field when its value looks like a real, existing
repo-relative path (not a URL, not a directory) - this lets old pages keep
working until they are migrated.
"""
from __future__ import annotations
import re
import unicodedata
from pathlib import Path
from typing import Optional
from chemenu import blocks, config, conventions
from chemenu.markdown_code import strip_code_spans
from chemenu.page import Page
# Real GFM footnotes: an inline `[^cite-id]` marker, resolved through a
# tool-owned `[^cite-id]: [[Source - X]]` (or `[[Source - X|file.md]]`)
# definition line - see cite_id() below for how the id is derived, and
# split_cite_block()/render_cite_block() for the definitions block itself.
#
# CITE_REF_RE deliberately does *not* try to exclude a definition line's own
# `[^id]` by pattern (e.g. "not followed by `:`") - prose legitimately
# contains a reference immediately before a colon ("Examples[^id]:"), which
# such a guard would misparse as a definition and silently drop. The real
# distinction is structural, not textual: a definition only ever exists
# inside the trailing Footnotes block (see CITE_BLOCK_HEADING), so every
# caller scans split_cite_block()'s `head` half only, never the block or the
# raw, unsplit body - and scans it through iter_cite_refs() rather than with
# this pattern directly, so that code is masked out first.
_CITE_ID_PATTERN = r"[A-Za-z0-9][A-Za-z0-9-]*"
CITE_DEF_RE = re.compile(
rf"^\[\^({_CITE_ID_PATTERN})\]:[ \t]*\[\[([^\]|#]+)(?:\|([^\]]+))?\]\][ \t]*$",
re.MULTILINE,
)
CITE_REF_RE = re.compile(rf"\[\^({_CITE_ID_PATTERN})\]")
# The pre-migration marker: `^[[Source - X]]` or `^[[Source - X|file.md]]`,
# read by a Pandoc-style parser as an inline footnote wrapping a broken
# shortcut link. Kept only so `lint` can flag any that were missed by the
# migration - see legacy_citation_markers() in commands/lint.py.
LEGACY_CITE_RE = re.compile(r"\^\[\[([^\]|#]+)(?:\|([^\]]+))?\]\]")
# The tool-owned block holding every `[^cite-id]: [[...]]` definition for a
# page, always the last section in the body. GFM and Obsidian both render
# footnote definitions regardless of the heading text; this heading is purely
# for human readability when the raw markdown is read directly.
#
# The prefix a source page's title carries, stripped when minting a cite id so
# the id is not "s-source-x". It is the `source` type-spec's own
# `title_prefix:`, asked for at call time rather than written down here: the
# type-spec belongs to the instance, so hardcoding the string made a documented
# instance decision into a compiler constant - the same leak `sections.py` had.
#
# The literal survives as the fallback for a tree with no resolvable `source`
# type (a fixture, a half-built instance). It is what this stack shipped, so a
# corpus that can reach the fallback was minted under it, and ids stay stable.
_FALLBACK_SOURCE_TITLE_PREFIX = "Source - "
def source_title_prefix() -> str:
"""This instance's source-page title prefix, from the type-spec."""
from chemenu.type_resolver import resolver
try:
type_path = resolver.find_type_by_name("source")
if type_path:
return resolver.get_title_prefix(type_path)
except (ValueError, OSError):
pass
return _FALLBACK_SOURCE_TITLE_PREFIX
def iter_cite_refs(text: str):
"""Every *real* `[^cite-id]` reference in `text`, code masked out.
The one entry point for reference scanning, and the reason
`CITE_REF_RE.finditer()` should not be called directly on a page body: a
page that writes the notation inside backticks or a fenced block is
describing it, not citing anything, and `undefined_footnote_refs` is a hard
error. See markdown_code.strip_code_spans() for what that masking covers.
Offsets survive the masking, so a caller may still use `m.start()` against
the text it passed in.
"""
return CITE_REF_RE.finditer(strip_code_spans(text))
def _slugify(text: str) -> str:
"""Transliterate to ASCII, then reduce to `[a-z0-9]` runs joined by `-`."""
normalized = unicodedata.normalize("NFKD", text)
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
return re.sub(r"[^A-Za-z0-9]+", "-", ascii_text).strip("-").lower()
def cite_id(title: str, qualifier: Optional[str] = None) -> str:
"""Deterministic footnote id for a (source title, optional file
qualifier) pair: strip the `Source - ` prefix, transliterate to ASCII,
slugify, and join title/qualifier with `--`.
Pure - always returns the same id for the same inputs, with no knowledge
of what ids already exist on a page. Two distinct pairs can collide (an
NFKD transliteration is lossy), so callers resolving a real page use
unique_cite_id() to add a `-2`/`-3` suffix on collision.
"""
prefix = source_title_prefix()
base_title = title[len(prefix):] if prefix and title.startswith(prefix) else title
slug = "s-" + _slugify(base_title)
if qualifier:
slug += "--" + _slugify(qualifier)
return slug if slug != "s-" else "s"
def unique_cite_id(existing_ids: set[str], title: str, qualifier: Optional[str] = None) -> str:
"""cite_id(), suffixed with `-2`, `-3`, ... until it is not in
`existing_ids`. Callers that want to *reuse* an id already pointing at
the same (title, qualifier) pair must check for that themselves before
calling this - it only ever returns a free id."""
base = cite_id(title, qualifier)
if base not in existing_ids:
return base
suffix = 2
while f"{base}-{suffix}" in existing_ids:
suffix += 1
return f"{base}-{suffix}"
# Headings a pre-4.0.0 page carries above its citation definitions, for the
# migration window only. Before the block was delimited it was *located* by this
# text, which is why there are four of them - two languages times two eras. The
# list is read, never written, and `instructions/migrations/` removes the need
# for it once every page carries markers.
_LEGACY_FOOTNOTE_HEADINGS = ("Fußnoten", "Footnotes", "Fussnoten", "Notes")
_LEGACY_HEADING_RE = re.compile(
r"^## (?:" + "|".join(re.escape(name) for name in _LEGACY_FOOTNOTE_HEADINGS) + r")[ \t]*$",
re.MULTILINE,
)
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
def _definitions_in(block: str) -> dict[str, tuple[str, Optional[str]]]:
"""Every `[^id]: [[Target]]` definition in one region, code masked out.
A fenced example of a definition line is an illustration, not a definition.
`strip_code_spans` preserves offsets and line structure, so the masked text
reads line-for-line against the real one.
"""
masked = strip_code_spans(block)
return {
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
for m in CITE_DEF_RE.finditer(masked)
}
def _split_legacy_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
"""The pre-marker layout: a heading, then definitions, ending at the next
heading.
Kept only so the corpus stays readable between this machinery landing and
the migration reaching each page. Every weakness of the old approach lives
here - it guesses the region's end, and it can be fooled by a fenced example
of the heading - which is the argument the marker pair settles.
"""
match = _LEGACY_HEADING_RE.search(body)
if not match:
return body.rstrip("\n"), {}
head, rest = body[: match.start()], body[match.end():]
following = _NEXT_HEADING_RE.search(rest)
block, trailing = (
(rest[: following.start()], rest[following.start():]) if following else (rest, "")
)
definitions = _definitions_in(block)
masked_block = strip_code_spans(block)
# Lines inside the block that are not definitions are content too - prose
# someone left there, a stray bullet. Rescued rather than rejected: this runs
# under `lint` and `corpus_diff` as well, where raising would refuse to read
# a page instead of reporting it.
stray = "\n".join(
line
for line, masked in zip(block.splitlines(), masked_block.splitlines())
if line.strip() and not CITE_DEF_RE.match(masked)
)
rescued = "\n\n".join(part.strip("\n") for part in (stray, trailing) if part.strip())
head = head.rstrip("\n")
if rescued:
head = f"{head}\n\n{rescued}" if head else rescued
return head, definitions
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
"""Split the citation region off `body`.
Returns (body_without_region, definitions), where definitions maps
cite_id -> (source_title, qualifier_or_None) in file order.
**The region is delimited, not guessed.** It used to end "at the next
heading", and before that "at the end of the file" - and every caller here
reassembles a page as `head + rendered region`, so a section that happened to
sit after it was silently deleted on the next `cite add`, `cite sync` or
`rename`. Eight pages were carrying content in that position when it was
found. A marker pair answers where the region stops exactly, which is the
whole reason for it.
A page with no markers is read through the legacy path instead, so the
corpus stays readable until the migration reaches it.
"""
region = blocks.find(body, blocks.FOOTNOTES)
if region is None:
return _split_legacy_block(body)
return blocks.strip(body, blocks.FOOTNOTES).rstrip("\n"), _definitions_in(region)
def render_cite_block(definitions: dict[str, tuple[str, Optional[str]]]) -> str:
"""The citation region for `definitions`, markers included, in dict order.
An empty dict renders "" - a page with no citations carries no region at
all, rather than a heading with nothing under it.
"""
lines = []
for cid, (title, qualifier) in definitions.items():
target = f"{title}|{qualifier}" if qualifier else title
lines.append(f"[^{cid}]: [[{target}]]")
return blocks.render(
blocks.FOOTNOTES, conventions.heading(blocks.FOOTNOTES), lines
)
def render_page_body(
head: str, definitions: dict[str, tuple[str, Optional[str]]]
) -> str:
"""Reassemble a page body from its non-citation content and its definitions -
the inverse of `split_cite_block`.
The heading is no longer threaded through from the caller. It used to be, so
that rewriting a page would not silently retitle a block whose text the tool
was *matching on*; now the marker carries the identity and the heading is a
rendering value, so re-rendering it under this instance's own words is a
repair rather than a rename.
"""
return blocks.replace(head.rstrip("\n") + "\n", blocks.FOOTNOTES, render_cite_block(definitions))
def extract_inline_cites(body: str) -> set[tuple[str, Optional[str]]]:
"""Return the set of (source_title, file_qualifier_or_None) cited
inline: every `[^cite-id]` reference resolved through this body's
`[^cite-id]: [[...]]` definitions. A reference with no matching
definition resolves to nothing here - see lint's undefined_footnote_refs
for that failure mode."""
head, definitions = split_cite_block(body)
return {definitions[m.group(1)] for m in iter_cite_refs(head) if m.group(1) in definitions}
def _looks_like_repo_path(value: str) -> bool:
if not isinstance(value, str) or not value:
return False
if "://" in value:
return False
return True
def source_raw_files(page: Page) -> list[str]:
"""The list of raw-relative paths a source page's frontmatter claims to cover.
Prefers the modern `raw_files:` list; falls back to the legacy scalar
`source:` field if it looks like a repo-relative path (not a URL).
"""
raw_files = page.frontmatter.get("raw_files")
if raw_files:
return list(raw_files)
legacy = page.frontmatter.get("source")
if _looks_like_repo_path(legacy):
return [legacy]
return []
def source_pages_by_raw_file(pages: dict[str, Page]) -> dict[str, list[str]]:
"""Invert source_raw_files() across every source page: raw path -> [source titles]."""
result: dict[str, list[str]] = {}
for title, page in pages.items():
if page.kind != "source":
continue
for raw_path in source_raw_files(page):
result.setdefault(raw_path, []).append(title)
return result
def duplicate_raw_file_owners(pages: dict[str, Page]) -> list[dict]:
"""Raw files claimed by more than one source page.
`raw_files:` is a maintenance claim, not a "mentions" relation (see
types/source.md, "One raw file, one owner"). Any number of pages may *cite* a
source; but with two owners it is undefined which page must be refreshed
when the raw file changes, so both rot silently and neither is identifiably
the stale one. `uncovered_raw_files()` cannot see this: it only asks whether
a raw file is claimed at all, which is why one ingested manual's
subtree sat with eight double-owned files unnoticed.
"""
return [
{"raw_file": raw_path, "owners": sorted(titles)}
for raw_path, titles in sorted(source_pages_by_raw_file(pages).items())
if len(set(titles)) > 1
]
def citing_pages(pages: dict[str, Page], source_title: str) -> list[str]:
"""Every page (other than the source page itself) that cites source_title,
either via frontmatter `sources:` or an inline `^[[source_title]]` marker."""
citing = []
for title, page in pages.items():
if title == source_title:
continue
if source_title in (page.frontmatter.get("sources") or []):
citing.append(title)
continue
if any(cited == source_title for cited, _file in extract_inline_cites(page.body)):
citing.append(title)
return sorted(set(citing))
def page_raw_files(pages: dict[str, Page], page: Page) -> list[str]:
"""All raw files backing a (non-source) page, via its cited/related source pages."""
raw_files: list[str] = []
for source_title in page.frontmatter.get("sources") or []:
source_page = pages.get(source_title)
if source_page is not None:
raw_files.extend(source_raw_files(source_page))
for cited_title, _file in extract_inline_cites(page.body):
source_page = pages.get(cited_title)
if source_page is not None:
raw_files.extend(source_raw_files(source_page))
return list(dict.fromkeys(raw_files))
def uncovered_raw_files(raw_dir: Path, pages: dict[str, Page]) -> list[str]:
"""Raw files with no source page claiming to cover them."""
covered = set(source_pages_by_raw_file(pages))
all_raw = {p.relative_to(config.ROOT).as_posix() for p in config.iter_raw_files(raw_dir)}
return sorted(all_raw - covered)
def broken_raw_refs(pages: dict[str, Page]) -> list[dict]:
"""`raw_files:`/legacy `source:` entries that point at a path which doesn't exist."""
issues = []
for title, page in pages.items():
if page.kind != "source":
continue
for raw_path in source_raw_files(page):
if not (config.ROOT / raw_path).exists():
issues.append({"page": title, "raw_path": raw_path})
return issues
def legacy_citation_markers(pages: dict[str, Page]) -> list[dict]:
"""Pages still carrying the pre-migration `^[[Source - X]]` marker
instead of a real `[^cite-id]` footnote reference - see LEGACY_CITE_RE."""
issues = []
for title, page in pages.items():
for m in LEGACY_CITE_RE.finditer(strip_code_spans(page.body)):
issues.append({"page": title, "marker": m.group(0)})
return issues
def undefined_footnote_refs(pages: dict[str, Page]) -> list[dict]:
"""`[^id]` references in a page's prose with no matching
`[^id]: [[...]]` definition in its Footnotes block - a citation whose
`cite add` never ran, or a hand-typed id."""
issues = []
for title, page in pages.items():
head, definitions = split_cite_block(page.body)
for m in iter_cite_refs(head):
ref_id = m.group(1)
if ref_id not in definitions:
issues.append({"page": title, "ref": ref_id})
return issues
def orphan_footnote_defs(pages: dict[str, Page]) -> list[dict]:
"""Footnotes definitions nothing in the page's prose references any
more - what `wikitool cite sync` prunes."""
issues = []
for title, page in pages.items():
head, definitions = split_cite_block(page.body)
referenced = {m.group(1) for m in iter_cite_refs(head)}
for ref_id, (source_title, _qualifier) in definitions.items():
if ref_id not in referenced:
issues.append({"page": title, "id": ref_id, "source": source_title})
return issues
def legacy_source_pages(pages: dict[str, Page]) -> list[dict]:
"""Source pages still using a directory-valued or URL-only legacy `source:`
field instead of the modern `raw_files:` list."""
issues = []
for title, page in pages.items():
if page.kind != "source":
continue
if page.frontmatter.get("raw_files"):
continue
legacy = page.frontmatter.get("source")
if not legacy:
continue
if "://" in str(legacy):
issues.append({"page": title, "source": legacy, "reason": "url-only, no raw_files"})
elif (config.ROOT / legacy).is_dir():
issues.append({"page": title, "source": legacy, "reason": "directory, not a file"})
return issues