feat: Prosa ist kein Identifier - Link-Taxonomie als Enum, generierte Regionen mit Markern (4.0.0)
CI / verify (push) Successful in 55s
Release / release (push) Successful in 38s

Files changed:
- .gitea/workflows/ci.yml
- AGENTS.md
- CHANGES.md
- VERSION
- instructions/CONTRACT.md
- instructions/link-taxonomy.md
- instructions/migrations/4.0.0-link-taxonomy.md
- instructions/setup-instance.md
- kb/CONTRACT.md
- kb/CONVENTIONS.md
- kb/CONVENTIONS.md.template
- kb/comparisons/COLLECTION.md
- kb/concepts/COLLECTION.md
- kb/entities/COLLECTION.md
- kb/sources/COLLECTION.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/blocks.py
- tools/chemenu/cli.py
- tools/chemenu/commands/cite_cmd.py
- tools/chemenu/commands/dist_cmd.py
- tools/chemenu/commands/docs_verify.py
- tools/chemenu/commands/doctor.py
- tools/chemenu/commands/links_cmd.py
- tools/chemenu/commands/migrate_cmd.py
- tools/chemenu/commands/new_page.py
- tools/chemenu/commands/page_ops.py
- tools/chemenu/commands/run_budget.py
- tools/chemenu/commands/xref.py
- tools/chemenu/conventions.py
- tools/chemenu/corpus_diff.py
- tools/chemenu/frontmatter_io.py
- tools/chemenu/kb_collections.py
- tools/chemenu/kb_state.py
- tools/chemenu/links.py
- tools/chemenu/lint_core.py
- tools/chemenu/provenance.py
- tools/chemenu/sections.py
- tools/chemenu/tests/conftest.py
- tools/chemenu/tests/test_blocks.py
- tools/chemenu/tests/test_cite_cmd.py
- tools/chemenu/tests/test_conventions.py
- tools/chemenu/tests/test_dist_cmd.py
- tools/chemenu/tests/test_doctor.py
- tools/chemenu/tests/test_migrate_cmd.py
- tools/chemenu/tests/test_new_page.py
- tools/chemenu/tests/test_pipeline_l0.py
- tools/chemenu/tests/test_types_cmd.py
- tools/chemenu/tests/test_xref.py
- types/concept.schema.yaml
- types/entity.md
- types/entity.schema.yaml
- types/instruction.schema.yaml
- types/type-spec.md
- work/link-taxonomy-migration/README.md
- work/link-taxonomy-migration/plan.md
This commit is contained in:
torben committed 2026-09-02 18:39:22 +02:00
1 parent 502971d147
commit 177c7e9ce8
56 files changed
+2692 -750

No files matched your search

+104 -95
View File
@@ -20,7 +20,7 @@ import unicodedata
from pathlib import Path
from typing import Optional
from chemenu import config, sections
from chemenu import blocks, config, conventions
from chemenu.markdown_code import strip_code_spans
from chemenu.page import Page
@@ -45,10 +45,6 @@ CITE_DEF_RE = re.compile(
)
CITE_REF_RE = re.compile(rf"\[\^({_CITE_ID_PATTERN})\]")
# Where the Footnotes block stops: the next ATX heading of any level. Without
# this the block ran to the end of the file and took any following section with
# it - see split_cite_block().
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
# The pre-migration marker: `^[[Source - X]]` or `^[[Source - X|file.md]]`,
# read by a Pandoc-style parser as an inline footnote wrapping a broken
@@ -61,27 +57,29 @@ LEGACY_CITE_RE = re.compile(r"\^\[\[([^\]|#]+)(?:\|([^\]]+))?\]\]")
# footnote definitions regardless of the heading text; this heading is purely
# for human readability when the raw markdown is read directly.
#
# Written under the canonical name, but split_cite_block() matches the aliases
# too - a page whose block still says "## Footnotes" keeps working until it is
# translated. See chemenu/sections.py.
# The prefix a source page's title carries, stripped when minting a cite id so
# the id is not "s-source-x". It is the `source` type-spec's own
# `title_prefix:`, asked for at call time rather than written down here: the
# type-spec belongs to the instance, so hardcoding the string made a documented
# instance decision into a compiler constant - the same leak `sections.py` had.
#
# Resolved on access rather than bound at import (PEP 562), because the
# canonical name is now this instance's own - `kb/CONVENTIONS.md`, via
# chemenu.conventions - and a module constant would freeze whichever corpus the
# process started in. The functions below take it as a default the same way, via
# None rather than an evaluated default argument.
def __getattr__(name: str) -> str:
if name == "CITE_BLOCK_HEADING":
return f"## {sections.FOOTNOTES}"
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
# The literal survives as the fallback for a tree with no resolvable `source`
# type (a fixture, a half-built instance). It is what this stack shipped, so a
# corpus that can reach the fallback was minted under it, and ids stay stable.
_FALLBACK_SOURCE_TITLE_PREFIX = "Source - "
def cite_block_heading_default() -> str:
"""The Footnotes heading this instance writes, `## ` included."""
return f"## {sections.FOOTNOTES}"
def source_title_prefix() -> str:
"""This instance's source-page title prefix, from the type-spec."""
from chemenu.type_resolver import resolver
_SOURCE_TITLE_PREFIX = "Source - "
try:
type_path = resolver.find_type_by_name("source")
if type_path:
return resolver.get_title_prefix(type_path)
except (ValueError, OSError):
pass
return _FALLBACK_SOURCE_TITLE_PREFIX
@@ -117,7 +115,8 @@ def cite_id(title: str, qualifier: Optional[str] = None) -> str:
NFKD transliteration is lossy), so callers resolving a real page use
unique_cite_id() to add a `-2`/`-3` suffix on collision.
"""
base_title = title[len(_SOURCE_TITLE_PREFIX):] if title.startswith(_SOURCE_TITLE_PREFIX) else title
prefix = source_title_prefix()
base_title = title[len(prefix):] if prefix and title.startswith(prefix) else title
slug = "s-" + _slugify(base_title)
if qualifier:
slug += "--" + _slugify(qualifier)
@@ -138,62 +137,64 @@ def unique_cite_id(existing_ids: set[str], title: str, qualifier: Optional[str]
return f"{base}-{suffix}"
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
"""Split the Footnotes block off `body`.
# Headings a pre-4.0.0 page carries above its citation definitions, for the
# migration window only. Before the block was delimited it was *located* by this
# text, which is why there are four of them - two languages times two eras. The
# list is read, never written, and `instructions/migrations/` removes the need
# for it once every page carries markers.
_LEGACY_FOOTNOTE_HEADINGS = ("Fußnoten", "Footnotes", "Fussnoten", "Notes")
Returns (body_without_block, definitions), where definitions maps
cite_id -> (source_title, qualifier_or_None) in file order. If there is
no Footnotes block, definitions is {} and body is returned with trailing
blank lines trimmed (so re-rendering after emptying the block is stable).
_LEGACY_HEADING_RE = re.compile(
r"^## (?:" + "|".join(re.escape(name) for name in _LEGACY_FOOTNOTE_HEADINGS) + r")[ \t]*$",
re.MULTILINE,
)
_NEXT_HEADING_RE = re.compile(r"^#{1,6} ", re.MULTILINE)
**The block is not "everything to the end of the file".** It used to be,
and every caller here reassembles a page as `head + rendered block` - so a
section that happened to sit after the block was silently deleted on the
next `cite add`, `cite sync` or `rename`. That is not hypothetical: `xref
add` appends its Relationships and See Also sections at the end of the
file, so whether a page kept its cross-references came down to which of the
two commands ran last. Eight pages were carrying content in that position
when this was found.
So the block ends where the next heading begins, and everything after it -
plus anything inside it that is not a citation definition - is folded back
on to `head`. Nothing is discarded, and because the rendered block is
always emitted last, a page that had drifted into the broken layout is
normalised the first time any of these commands touches it.
def _definitions_in(block: str) -> dict[str, tuple[str, Optional[str]]]:
"""Every `[^id]: [[Target]]` definition in one region, code masked out.
A fenced example of a definition line is an illustration, not a definition.
`strip_code_spans` preserves offsets and line structure, so the masked text
reads line-for-line against the real one.
"""
# Where the block *starts* is decided on the unmasked body, deliberately.
# Masking first would mean one unclosed fence anywhere in the prose blanks
# the real `## Footnotes` heading too, and the page then reads as having no
# definitions at all - every citation on it undefined, from a single typo.
# A fenced example of the heading itself is the rarer accident and the
# cheaper one: it costs one page its block, not every citation on it.
match = sections.heading_re(sections.FOOTNOTES).search(body)
masked = strip_code_spans(block)
return {
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
for m in CITE_DEF_RE.finditer(masked)
}
def _split_legacy_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
"""The pre-marker layout: a heading, then definitions, ending at the next
heading.
Kept only so the corpus stays readable between this machinery landing and
the migration reaching each page. Every weakness of the old approach lives
here - it guesses the region's end, and it can be fooled by a fenced example
of the heading - which is the argument the marker pair settles.
"""
match = _LEGACY_HEADING_RE.search(body)
if not match:
return body.rstrip("\n"), {}
head, rest = body[: match.start()], body[match.end():]
next_section = _NEXT_HEADING_RE.search(rest)
block, trailing = (rest[: next_section.start()], rest[next_section.start():]) if next_section else (rest, "")
following = _NEXT_HEADING_RE.search(rest)
block, trailing = (
(rest[: following.start()], rest[following.start():]) if following else (rest, "")
)
# Inside the block, code is masked: a fenced example of a definition line is
# an illustration, not a definition. strip_code_spans() preserves offsets
# and line structure, so the masked block can be read line-for-line against
# the real one.
definitions = _definitions_in(block)
masked_block = strip_code_spans(block)
definitions = {
m.group(1): (m.group(2).strip(), m.group(3).strip() if m.group(3) else None)
for m in CITE_DEF_RE.finditer(masked_block)
}
# Lines inside the block that are not definitions are content too - prose
# someone left there, a stray bullet. Rescued rather than rejected: this
# runs under `lint` and `corpus_diff` as well, where raising would refuse
# to read a page instead of reporting it.
# someone left there, a stray bullet. Rescued rather than rejected: this runs
# under `lint` and `corpus_diff` as well, where raising would refuse to read
# a page instead of reporting it.
stray = "\n".join(
line
for line, masked in zip(block.splitlines(), masked_block.splitlines())
if line.strip() and not CITE_DEF_RE.match(masked)
)
rescued = "\n\n".join(part.strip("\n") for part in (stray, trailing) if part.strip())
head = head.rstrip("\n")
if rescued:
@@ -201,49 +202,57 @@ def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]
return head, definitions
def cite_block_heading(body: str) -> str:
"""The Footnotes heading `body` actually carries, canonical if it has none.
def split_cite_block(body: str) -> tuple[str, dict[str, tuple[str, Optional[str]]]]:
"""Split the citation region off `body`.
Rewriting a page must not silently retitle its block: a page still using an
alias is untranslated, not broken, and `cite sync` has to stay a no-op on
it. Translating the heading is the migration's job, not the tool's."""
match = sections.heading_re(sections.FOOTNOTES).search(body)
return match.group(0).strip() if match else cite_block_heading_default()
Returns (body_without_region, definitions), where definitions maps
cite_id -> (source_title, qualifier_or_None) in file order.
**The region is delimited, not guessed.** It used to end "at the next
heading", and before that "at the end of the file" - and every caller here
reassembles a page as `head + rendered region`, so a section that happened to
sit after it was silently deleted on the next `cite add`, `cite sync` or
`rename`. Eight pages were carrying content in that position when it was
found. A marker pair answers where the region stops exactly, which is the
whole reason for it.
A page with no markers is read through the legacy path instead, so the
corpus stays readable until the migration reaches it.
"""
region = blocks.find(body, blocks.FOOTNOTES)
if region is None:
return _split_legacy_block(body)
return blocks.strip(body, blocks.FOOTNOTES).rstrip("\n"), _definitions_in(region)
def render_cite_block(
definitions: dict[str, tuple[str, Optional[str]]], heading: Optional[str] = None
) -> str:
"""Render the Footnotes block for `definitions` (cite_id -> (title,
qualifier)), preserving dict order. Empty dict renders "" - a page with
no citations carries no block at all.
def render_cite_block(definitions: dict[str, tuple[str, Optional[str]]]) -> str:
"""The citation region for `definitions`, markers included, in dict order.
`heading=None` means this instance's canonical Footnotes heading, resolved
at call time. It cannot be an evaluated default: the name comes from
`kb/CONVENTIONS.md`, so a default bound at import would answer for whichever
corpus the process started in."""
if not definitions:
return ""
lines = [heading or cite_block_heading_default(), ""]
An empty dict renders "" - a page with no citations carries no region at
all, rather than a heading with nothing under it.
"""
lines = []
for cid, (title, qualifier) in definitions.items():
target = f"{title}|{qualifier}" if qualifier else title
lines.append(f"[^{cid}]: [[{target}]]")
return "\n".join(lines) + "\n"
return blocks.render(
blocks.FOOTNOTES, conventions.heading(blocks.FOOTNOTES), lines
)
def render_page_body(
head: str,
definitions: dict[str, tuple[str, Optional[str]]],
heading: Optional[str] = None,
head: str, definitions: dict[str, tuple[str, Optional[str]]]
) -> str:
"""Reassemble a page body from its non-Footnotes content and citation
definitions - the inverse of split_cite_block(). Pass the original body's
`cite_block_heading()` to preserve an alias the page still uses."""
head = head.rstrip("\n")
block = render_cite_block(definitions, heading)
if not block:
return head + "\n"
return head + "\n\n" + block
"""Reassemble a page body from its non-citation content and its definitions -
the inverse of `split_cite_block`.
The heading is no longer threaded through from the caller. It used to be, so
that rewriting a page would not silently retitle a block whose text the tool
was *matching on*; now the marker carries the identity and the heading is a
rendering value, so re-rendering it under this instance's own words is a
repair rather than a rename.
"""
return blocks.replace(head.rstrip("\n") + "\n", blocks.FOOTNOTES, render_cite_block(definitions))
def extract_inline_cites(body: str) -> set[tuple[str, Optional[str]]]: