feat: Prosa ist kein Identifier - Link-Taxonomie als Enum, generierte Regionen mit Markern (4.0.0)
Files changed: - .gitea/workflows/ci.yml - AGENTS.md - CHANGES.md - VERSION - instructions/CONTRACT.md - instructions/link-taxonomy.md - instructions/migrations/4.0.0-link-taxonomy.md - instructions/setup-instance.md - kb/CONTRACT.md - kb/CONVENTIONS.md - kb/CONVENTIONS.md.template - kb/comparisons/COLLECTION.md - kb/concepts/COLLECTION.md - kb/entities/COLLECTION.md - kb/sources/COLLECTION.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/blocks.py - tools/chemenu/cli.py - tools/chemenu/commands/cite_cmd.py - tools/chemenu/commands/dist_cmd.py - tools/chemenu/commands/docs_verify.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/links_cmd.py - tools/chemenu/commands/migrate_cmd.py - tools/chemenu/commands/new_page.py - tools/chemenu/commands/page_ops.py - tools/chemenu/commands/run_budget.py - tools/chemenu/commands/xref.py - tools/chemenu/conventions.py - tools/chemenu/corpus_diff.py - tools/chemenu/frontmatter_io.py - tools/chemenu/kb_collections.py - tools/chemenu/kb_state.py - tools/chemenu/links.py - tools/chemenu/lint_core.py - tools/chemenu/provenance.py - tools/chemenu/sections.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_blocks.py - tools/chemenu/tests/test_cite_cmd.py - tools/chemenu/tests/test_conventions.py - tools/chemenu/tests/test_dist_cmd.py - tools/chemenu/tests/test_doctor.py - tools/chemenu/tests/test_migrate_cmd.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_pipeline_l0.py - tools/chemenu/tests/test_types_cmd.py - tools/chemenu/tests/test_xref.py - types/concept.schema.yaml - types/entity.md - types/entity.schema.yaml - types/instruction.schema.yaml - types/type-spec.md - work/link-taxonomy-migration/README.md - work/link-taxonomy-migration/plan.md
This commit is contained in:
1 parent
502971d147
commit
177c7e9ce8
56 files changed
+2692
-750
No files matched your search
+104
-95
@@ -20,7 +20,7 @@ import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from chemenu import config, sections
|
||||
from chemenu import blocks, config, conventions
|
||||
from chemenu.markdown_code import strip_code_spans
|
||||
from chemenu.page import Page
|
||||
|
||||
@@ -45,10 +45,6 @@ CITE_DEF_RE = re.compile(
|
||||
)
|
||||
CITE_REF_RE = re.compile(rf"\[\^({_CITE_ID_PATTERN})\]")
|
||||
|
||||
# Where the Footnotes block stops: the next ATX heading of any level. Without
|
||||
# this the block ran to the end of the file and took any following section with
|
||||
# it - see split_cite_block().
|
||||
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
|
||||
|
||||
# The pre-migration marker: `^[[Source - X]]` or `^[[Source - X|file.md]]`,
|
||||
# read by a Pandoc-style parser as an inline footnote wrapping a broken
|
||||
@@ -61,27 +57,29 @@ LEGACY_CITE_RE = re.compile(r"\^\[\[([^\]|#]+)(?:\|([^\]]+))?\]\]")
|
||||
# footnote definitions regardless of the heading text; this heading is purely
|
||||
# for human readability when the raw markdown is read directly.
|
||||
#
|
||||
# Written under the canonical name, but split_cite_block() matches the aliases
|
||||
# too - a page whose block still says "## Footnotes" keeps working until it is
|
||||
# translated. See chemenu/sections.py.
|
||||
# The prefix a source page's title carries, stripped when minting a cite id so
|
||||
# the id is not "s-source-x". It is the `source` type-spec's own
|
||||
# `title_prefix:`, asked for at call time rather than written down here: the
|
||||
# type-spec belongs to the instance, so hardcoding the string made a documented
|
||||
# instance decision into a compiler constant - the same leak `sections.py` had.
|
||||
#
|
||||
# Resolved on access rather than bound at import (PEP 562), because the
|
||||
# canonical name is now this instance's own - `kb/CONVENTIONS.md`, via
|
||||
# chemenu.conventions - and a module constant would freeze whichever corpus the
|
||||
# process started in. The functions below take it as a default the same way, via
|
||||
# None rather than an evaluated default argument.
|
||||
def __getattr__(name: str) -> str:
|
||||
if name == "CITE_BLOCK_HEADING":
|
||||
return f"## {sections.FOOTNOTES}"
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||
# The literal survives as the fallback for a tree with no resolvable `source`
|
||||
# type (a fixture, a half-built instance). It is what this stack shipped, so a
|
||||
# corpus that can reach the fallback was minted under it, and ids stay stable.
|
||||
_FALLBACK_SOURCE_TITLE_PREFIX = "Source - "
|
||||
|
||||
|
||||
def cite_block_heading_default() -> str:
|
||||
"""The Footnotes heading this instance writes, `## ` included."""
|
||||
return f"## {sections.FOOTNOTES}"
|
||||
def source_title_prefix() -> str:
|
||||
"""This instance's source-page title prefix, from the type-spec."""
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
|
||||
_SOURCE_TITLE_PREFIX = "Source - "
|
||||
try:
|
||||
type_path = resolver.find_type_by_name("source")
|
||||
if type_path:
|
||||
return resolver.get_title_prefix(type_path)
|
||||
except (ValueError, OSError):
|
||||
pass
|
||||
return _FALLBACK_SOURCE_TITLE_PREFIX
|
||||
|
||||
|
||||
|
||||
@@ -117,7 +115,8 @@ def cite_id(title: str, qualifier: Optional[str] = None) -> str:
|
||||
NFKD transliteration is lossy), so callers resolving a real page use
|
||||
unique_cite_id() to add a `-2`/`-3` suffix on collision.
|
||||
"""
|
||||
base_title = title[len(_SOURCE_TITLE_PREFIX):] if title.startswith(_SOURCE_TITLE_PREFIX) else title
|
||||
prefix = source_title_prefix()
|
||||
base_title = title[len(prefix):] if prefix and title.startswith(prefix) else title
|
||||
slug = "s-" + _slugify(base_title)
|
||||
if qualifier:
|
||||
slug += "--" + _slugify(qualifier)
|
||||
@@ -138,62 +137,64 @@ def unique_cite_id(existing_ids: set[str], title: str, qualifier: Optional[str]
|
||||
return f"{base}-{suffix}"
|
||||
|
||||
|
||||
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
||||
"""Split the Footnotes block off `body`.
|
||||
# Headings a pre-4.0.0 page carries above its citation definitions, for the
|
||||
# migration window only. Before the block was delimited it was *located* by this
|
||||
# text, which is why there are four of them - two languages times two eras. The
|
||||
# list is read, never written, and `instructions/migrations/` removes the need
|
||||
# for it once every page carries markers.
|
||||
_LEGACY_FOOTNOTE_HEADINGS = ("Fußnoten", "Footnotes", "Fussnoten", "Notes")
|
||||
|
||||
Returns (body_without_block, definitions), where definitions maps
|
||||
cite_id -> (source_title, qualifier_or_None) in file order. If there is
|
||||
no Footnotes block, definitions is {} and body is returned with trailing
|
||||
blank lines trimmed (so re-rendering after emptying the block is stable).
|
||||
_LEGACY_HEADING_RE = re.compile(
|
||||
r"^## (?:" + "|".join(re.escape(name) for name in _LEGACY_FOOTNOTE_HEADINGS) + r")[ \t]*$",
|
||||
re.MULTILINE,
|
||||
)
|
||||
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
|
||||
|
||||
**The block is not "everything to the end of the file".** It used to be,
|
||||
and every caller here reassembles a page as `head + rendered block` - so a
|
||||
section that happened to sit after the block was silently deleted on the
|
||||
next `cite add`, `cite sync` or `rename`. That is not hypothetical: `xref
|
||||
add` appends its Relationships and See Also sections at the end of the
|
||||
file, so whether a page kept its cross-references came down to which of the
|
||||
two commands ran last. Eight pages were carrying content in that position
|
||||
when this was found.
|
||||
|
||||
So the block ends where the next heading begins, and everything after it -
|
||||
plus anything inside it that is not a citation definition - is folded back
|
||||
on to `head`. Nothing is discarded, and because the rendered block is
|
||||
always emitted last, a page that had drifted into the broken layout is
|
||||
normalised the first time any of these commands touches it.
|
||||
def _definitions_in(block: str) -> dict[str, tuple[str, Optional[str]]]:
|
||||
"""Every `[^id]: [[Target]]` definition in one region, code masked out.
|
||||
|
||||
A fenced example of a definition line is an illustration, not a definition.
|
||||
`strip_code_spans` preserves offsets and line structure, so the masked text
|
||||
reads line-for-line against the real one.
|
||||
"""
|
||||
# Where the block *starts* is decided on the unmasked body, deliberately.
|
||||
# Masking first would mean one unclosed fence anywhere in the prose blanks
|
||||
# the real `## Footnotes` heading too, and the page then reads as having no
|
||||
# definitions at all - every citation on it undefined, from a single typo.
|
||||
# A fenced example of the heading itself is the rarer accident and the
|
||||
# cheaper one: it costs one page its block, not every citation on it.
|
||||
match = sections.heading_re(sections.FOOTNOTES).search(body)
|
||||
masked = strip_code_spans(block)
|
||||
return {
|
||||
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
|
||||
for m in CITE_DEF_RE.finditer(masked)
|
||||
}
|
||||
|
||||
|
||||
def _split_legacy_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
||||
"""The pre-marker layout: a heading, then definitions, ending at the next
|
||||
heading.
|
||||
|
||||
Kept only so the corpus stays readable between this machinery landing and
|
||||
the migration reaching each page. Every weakness of the old approach lives
|
||||
here - it guesses the region's end, and it can be fooled by a fenced example
|
||||
of the heading - which is the argument the marker pair settles.
|
||||
"""
|
||||
match = _LEGACY_HEADING_RE.search(body)
|
||||
if not match:
|
||||
return body.rstrip("\n"), {}
|
||||
head, rest = body[: match.start()], body[match.end():]
|
||||
|
||||
next_section = _NEXT_HEADING_RE.search(rest)
|
||||
block, trailing = (rest[: next_section.start()], rest[next_section.start():]) if next_section else (rest, "")
|
||||
following = _NEXT_HEADING_RE.search(rest)
|
||||
block, trailing = (
|
||||
(rest[: following.start()], rest[following.start():]) if following else (rest, "")
|
||||
)
|
||||
|
||||
# Inside the block, code is masked: a fenced example of a definition line is
|
||||
# an illustration, not a definition. strip_code_spans() preserves offsets
|
||||
# and line structure, so the masked block can be read line-for-line against
|
||||
# the real one.
|
||||
definitions = _definitions_in(block)
|
||||
masked_block = strip_code_spans(block)
|
||||
definitions = {
|
||||
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
|
||||
for m in CITE_DEF_RE.finditer(masked_block)
|
||||
}
|
||||
# Lines inside the block that are not definitions are content too - prose
|
||||
# someone left there, a stray bullet. Rescued rather than rejected: this
|
||||
# runs under `lint` and `corpus_diff` as well, where raising would refuse
|
||||
# to read a page instead of reporting it.
|
||||
# someone left there, a stray bullet. Rescued rather than rejected: this runs
|
||||
# under `lint` and `corpus_diff` as well, where raising would refuse to read
|
||||
# a page instead of reporting it.
|
||||
stray = "\n".join(
|
||||
line
|
||||
for line, masked in zip(block.splitlines(), masked_block.splitlines())
|
||||
if line.strip() and not CITE_DEF_RE.match(masked)
|
||||
)
|
||||
|
||||
rescued = "\n\n".join(part.strip("\n") for part in (stray, trailing) if part.strip())
|
||||
head = head.rstrip("\n")
|
||||
if rescued:
|
||||
@@ -201,49 +202,57 @@ def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]
|
||||
return head, definitions
|
||||
|
||||
|
||||
def cite_block_heading(body: str) -> str:
|
||||
"""The Footnotes heading `body` actually carries, canonical if it has none.
|
||||
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
|
||||
"""Split the citation region off `body`.
|
||||
|
||||
Rewriting a page must not silently retitle its block: a page still using an
|
||||
alias is untranslated, not broken, and `cite sync` has to stay a no-op on
|
||||
it. Translating the heading is the migration's job, not the tool's."""
|
||||
match = sections.heading_re(sections.FOOTNOTES).search(body)
|
||||
return match.group(0).strip() if match else cite_block_heading_default()
|
||||
Returns (body_without_region, definitions), where definitions maps
|
||||
cite_id -> (source_title, qualifier_or_None) in file order.
|
||||
|
||||
**The region is delimited, not guessed.** It used to end "at the next
|
||||
heading", and before that "at the end of the file" - and every caller here
|
||||
reassembles a page as `head + rendered region`, so a section that happened to
|
||||
sit after it was silently deleted on the next `cite add`, `cite sync` or
|
||||
`rename`. Eight pages were carrying content in that position when it was
|
||||
found. A marker pair answers where the region stops exactly, which is the
|
||||
whole reason for it.
|
||||
|
||||
A page with no markers is read through the legacy path instead, so the
|
||||
corpus stays readable until the migration reaches it.
|
||||
"""
|
||||
region = blocks.find(body, blocks.FOOTNOTES)
|
||||
if region is None:
|
||||
return _split_legacy_block(body)
|
||||
return blocks.strip(body, blocks.FOOTNOTES).rstrip("\n"), _definitions_in(region)
|
||||
|
||||
|
||||
def render_cite_block(
|
||||
definitions: dict[str, tuple[str, Optional[str]]], heading: Optional[str] = None
|
||||
) -> str:
|
||||
"""Render the Footnotes block for `definitions` (cite_id -> (title,
|
||||
qualifier)), preserving dict order. Empty dict renders "" - a page with
|
||||
no citations carries no block at all.
|
||||
def render_cite_block(definitions: dict[str, tuple[str, Optional[str]]]) -> str:
|
||||
"""The citation region for `definitions`, markers included, in dict order.
|
||||
|
||||
`heading=None` means this instance's canonical Footnotes heading, resolved
|
||||
at call time. It cannot be an evaluated default: the name comes from
|
||||
`kb/CONVENTIONS.md`, so a default bound at import would answer for whichever
|
||||
corpus the process started in."""
|
||||
if not definitions:
|
||||
return ""
|
||||
lines = [heading or cite_block_heading_default(), ""]
|
||||
An empty dict renders "" - a page with no citations carries no region at
|
||||
all, rather than a heading with nothing under it.
|
||||
"""
|
||||
lines = []
|
||||
for cid, (title, qualifier) in definitions.items():
|
||||
target = f"{title}|{qualifier}" if qualifier else title
|
||||
lines.append(f"[^{cid}]: [[{target}]]")
|
||||
return "\n".join(lines) + "\n"
|
||||
return blocks.render(
|
||||
blocks.FOOTNOTES, conventions.heading(blocks.FOOTNOTES), lines
|
||||
)
|
||||
|
||||
|
||||
def render_page_body(
|
||||
head: str,
|
||||
definitions: dict[str, tuple[str, Optional[str]]],
|
||||
heading: Optional[str] = None,
|
||||
head: str, definitions: dict[str, tuple[str, Optional[str]]]
|
||||
) -> str:
|
||||
"""Reassemble a page body from its non-Footnotes content and citation
|
||||
definitions - the inverse of split_cite_block(). Pass the original body's
|
||||
`cite_block_heading()` to preserve an alias the page still uses."""
|
||||
head = head.rstrip("\n")
|
||||
block = render_cite_block(definitions, heading)
|
||||
if not block:
|
||||
return head + "\n"
|
||||
return head + "\n\n" + block
|
||||
"""Reassemble a page body from its non-citation content and its definitions -
|
||||
the inverse of `split_cite_block`.
|
||||
|
||||
The heading is no longer threaded through from the caller. It used to be, so
|
||||
that rewriting a page would not silently retitle a block whose text the tool
|
||||
was *matching on*; now the marker carries the identity and the heading is a
|
||||
rendering value, so re-rendering it under this instance's own words is a
|
||||
repair rather than a rename.
|
||||
"""
|
||||
return blocks.replace(head.rstrip("\n") + "\n", blocks.FOOTNOTES, render_cite_block(definitions))
|
||||
|
||||
|
||||
def extract_inline_cites(body: str) -> set[tuple[str, Optional[str]]]:
|
||||
|
||||
Reference in new issue
Block a user