feat: Prosa ist kein Identifier - Link-Taxonomie als Enum, generierte Regionen mit Markern (4.0.0)
CI / verify (push) Successful in 55s
Release / release (push) Successful in 38s

Files changed:
- .gitea/workflows/ci.yml
- AGENTS.md
- CHANGES.md
- VERSION
- instructions/CONTRACT.md
- instructions/link-taxonomy.md
- instructions/migrations/4.0.0-link-taxonomy.md
- instructions/setup-instance.md
- kb/CONTRACT.md
- kb/CONVENTIONS.md
- kb/CONVENTIONS.md.template
- kb/comparisons/COLLECTION.md
- kb/concepts/COLLECTION.md
- kb/entities/COLLECTION.md
- kb/sources/COLLECTION.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/blocks.py
- tools/chemenu/cli.py
- tools/chemenu/commands/cite_cmd.py
- tools/chemenu/commands/dist_cmd.py
- tools/chemenu/commands/docs_verify.py
- tools/chemenu/commands/doctor.py
- tools/chemenu/commands/links_cmd.py
- tools/chemenu/commands/migrate_cmd.py
- tools/chemenu/commands/new_page.py
- tools/chemenu/commands/page_ops.py
- tools/chemenu/commands/run_budget.py
- tools/chemenu/commands/xref.py
- tools/chemenu/conventions.py
- tools/chemenu/corpus_diff.py
- tools/chemenu/frontmatter_io.py
- tools/chemenu/kb_collections.py
- tools/chemenu/kb_state.py
- tools/chemenu/links.py
- tools/chemenu/lint_core.py
- tools/chemenu/provenance.py
- tools/chemenu/sections.py
- tools/chemenu/tests/conftest.py
- tools/chemenu/tests/test_blocks.py
- tools/chemenu/tests/test_cite_cmd.py
- tools/chemenu/tests/test_conventions.py
- tools/chemenu/tests/test_dist_cmd.py
- tools/chemenu/tests/test_doctor.py
- tools/chemenu/tests/test_migrate_cmd.py
- tools/chemenu/tests/test_new_page.py
- tools/chemenu/tests/test_pipeline_l0.py
- tools/chemenu/tests/test_types_cmd.py
- tools/chemenu/tests/test_xref.py
- types/concept.schema.yaml
- types/entity.md
- types/entity.schema.yaml
- types/instruction.schema.yaml
- types/type-spec.md
- work/link-taxonomy-migration/README.md
- work/link-taxonomy-migration/plan.md
This commit is contained in:
torben committed 2026-09-02 18:39:22 +02:00
1 parent 502971d147
commit 177c7e9ce8
56 files changed
+2692 -750

No files matched your search

+2 -3
View File
@@ -28,7 +28,6 @@ from chemenu.kb_scan import load_kb_pages
from chemenu.provenance import (
CITE_REF_RE,
cite_id,
cite_block_heading,
render_page_body,
split_cite_block,
unique_cite_id,
@@ -81,7 +80,7 @@ def upsert_citation(page: Page, source_title: str, qualifier: Optional[str]) ->
if sources_changed:
sources.append(source_title)
new_body = render_page_body(head, definitions, cite_block_heading(page.body))
new_body = render_page_body(head, definitions)
changed = block_changed or sources_changed or new_body != page.body
return marker_id, new_body, changed
@@ -152,7 +151,7 @@ def sync_page(page: Page) -> tuple[str, bool, list[str], list[str]]:
ordered[cid] = definitions[cid]
seen.add(cid)
new_body = render_page_body(head, ordered, cite_block_heading(page.body))
new_body = render_page_body(head, ordered)
changed = new_body != page.body
return new_body, changed, pruned, undefined
+62 -1
View File
@@ -264,6 +264,60 @@ class Origin(NamedTuple):
update_url: Optional[str] = None
def instance_owned_type_stems() -> set[str]:
"""Type-spec stems whose instances are knowledge pages, and which therefore
belong to the instance rather than to the stack.
The line is `root:`, and it was already in the frontmatter before anyone
drew it: `root: kb` means the type describes a page the instance writes, so
its prose, its template and its language are the instance's business.
Anything else - `instruction` (`root: repo`), `lint-report` (no `base_dir`
at all), `type-spec` itself - describes a stack artifact and ships verbatim.
Read from `types/` rather than listed, so an instance adding its own page
type gets the same treatment without a code change.
"""
from chemenu.type_resolver import resolver
stems: set[str] = set()
for type_path, frontmatter in resolver.list_type_specs():
stem = Path(type_path).stem
if stem == "type-spec":
continue
if not frontmatter.get("base_dir"):
continue
if (frontmatter.get("root") or "kb") != "kb":
continue
stems.add(stem)
return stems
def _plan_types() -> dict[str, PlannedFile]:
"""`types/`, with the page type-specs re-keyed as templates.
Same split as the collection contracts, for the same reason and by the same
mechanism: the shipped content is a working default rather than something
wrong for the receiver, so the file itself crosses - under a name that has
to be adopted before it counts. A type-spec's `.schema.yaml` travels with
it, because the two are one type (see types/type-spec.md § Anatomy) and
adopting half of it would leave a spec validated by a file it does not own.
"""
plan = _copy_tree(config.TYPES_DIR, "types", frozenset())
stems = instance_owned_type_stems()
if not stems:
return plan
rekeyed: dict[str, PlannedFile] = {}
for relative, planned in plan.items():
name = relative.rsplit("/", 1)[-1]
stem = name.split(".", 1)[0]
if stem in stems:
rekeyed[f"{relative}.template"] = planned
else:
rekeyed[relative] = planned
return rekeyed
def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
"""Every (destination-relative path -> planned file) the export writes."""
plan: dict[str, PlannedFile] = {}
@@ -286,7 +340,7 @@ def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
plan[name] = _read_planned_file(source, name)
plan.update(_copy_tree(config.INSTRUCTIONS_DIR, "instructions", frozenset(INSTRUCTIONS_EXCLUDE_DIRS)))
plan.update(_copy_tree(config.TYPES_DIR, "types", frozenset()))
plan.update(_plan_types())
plan.update(_copy_tree(
config.ROOT / "tools", "tools", frozenset(TOOLS_EXCLUDE_DIRS), _is_coverage_output
))
@@ -382,6 +436,7 @@ _INSTANCE_OWNED_KB_FILES = (kb_collections.CONTRACT_NAME, conventions.CONVENTION
def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
"""Planned paths that carry one instance's own data instead of machinery."""
owned_types = instance_owned_type_stems()
leaks: list[str] = []
for relative in sorted(plan):
name = relative.rsplit("/", 1)[-1]
@@ -389,6 +444,12 @@ def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
leaks.append(f"{relative} (one instance's own personalization)")
elif relative.startswith("kb/") and name in _INSTANCE_OWNED_KB_FILES:
leaks.append(f"{relative} (this instance's authoring conventions; ship the .template)")
elif (
relative.startswith("types/")
and not relative.endswith(".template")
and name.split(".", 1)[0] in owned_types
):
leaks.append(f"{relative} (this instance's page type-spec; ship the .template)")
elif relative.startswith("instructions/dev/"):
leaks.append(f"{relative} (stack-development only)")
elif relative.startswith(_CONTENT_PREFIXES) and name not in _CONTENT_ALLOWED_NAMES:
+43
View File
@@ -260,10 +260,53 @@ def check_collection_contracts() -> list[str]:
issues += kb_collections.declaration_issues()
issues += conventions.declaration_issues()
issues += check_stack_required_types()
return issues
def check_stack_required_types() -> list[str]:
"""The minimum the stack asks of the type layer, and nothing beyond it.
The four page type-specs belong to the instance: it may translate them,
rewrite their templates, add sections. What it may not do is remove the one
type the provenance path is built on, or drop the field that path reads.
Everything else about `types/source.md` - its prose, its template, its title
prefix, its directory - is the instance's, and is deliberately not checked
here.
"""
from chemenu.type_resolver import resolver
issues: list[str] = []
for type_name in kb_collections.STACK_REQUIRED_TYPES:
try:
type_path = resolver.find_type_by_name(type_name)
except (ValueError, OSError) as exc:
issues.append(f"types/ could not be read to find the `{type_name}` type: {exc}")
continue
if not type_path:
issues.append(
f"no type-spec declares `name: {type_name}` - `sources coverage`, `[^cite-id]` "
f"resolution and `kb/provenance.md` all ask `page.kind == \"{type_name}\"`, so "
f"without it the whole raw/ -> kb/ provenance path resolves against nothing"
)
continue
try:
schema = resolver.get_schema(type_path) or {}
except (ValueError, OSError) as exc:
issues.append(f"{type_path}: its schema could not be read: {exc}")
continue
declared = set(schema.get("required") or [])
for field in kb_collections.STACK_REQUIRED_TYPE_FIELDS.get(type_name, ()):
if field not in declared:
issues.append(
f"{type_path}: its schema must require `{field}` - it is what the "
f"provenance path reads, and a `{type_name}` page without it claims no "
f"raw material at all"
)
return issues
def check_legacy_type_blocks() -> list[str]:
issues = []
guarded = [
+14 -11
View File
@@ -217,17 +217,18 @@ def check_conventions() -> Check:
"""Whether this instance has said how its own pages are written.
`kb/CONVENTIONS.md` carries the decisions `kb/CONTRACT.md` deliberately no
longer makes: the KB language and its three tool-owned section headings, the
relationship-label vocabulary, the tone examples, the confidence rubric, the
ADR prefix. The compiler reads the section names out of it, so an instance
without one is not merely undocumented - `xref add` and `cite add` fall back
to the names this stack hardcoded before the file existed, which is right
only for a corpus that was written under them.
longer makes: the KB language and the headings its two generated regions
render under, the tone examples, the confidence rubric, the naming forms.
Hence `FAIL` rather than `WARN`, and hence the same two failure modes the
personalization pair has: the distribution can ship the template but never
the filled file, so a template renamed and left unanswered looks present and
decides nothing.
`FAIL` rather than `WARN` because those decisions bind every page, and
because it has the same two failure modes the personalization pair has: the
distribution can ship the template but never the filled file, so a template
renamed and left unanswered looks present and decides nothing.
The headings themselves are only cosmetic now - the marker pair carries each
region's identity, so a default renders wrong words rather than corrupting
structure. That is why this check is about the *file*, not about rescuing a
lookup the compiler can no longer get wrong.
"""
path = conventions.conventions_file()
fix = (
@@ -246,7 +247,9 @@ def check_conventions() -> Check:
if issues:
return Check("conventions", "FAIL", "; ".join(issues), fix)
declared = conventions.language() or "unspecified"
headings = ", ".join(conventions.canonical(slot) for slot in conventions.SLOTS)
from chemenu import blocks
headings = ", ".join(conventions.heading(block) for block in blocks.BLOCKS)
return Check(
"conventions", "OK",
f"kb/{conventions.CONVENTIONS_FILENAME} present, language {declared}, "
+95
View File
@@ -0,0 +1,95 @@
"""`wikitool links` - the declared graph around one page, both directions.
The half that makes authored directional edges liveable. An edge is written once,
on the page that asserts it, so the question "what points at *this* page" has no
answer stored anywhere - it is computed from the graph, which is the only way it
is ever complete. A mirrored edge only ever recorded what someone remembered to
mirror.
Read-only, and exempt from the iteration budget for the same reason `search` is:
it answers a question rather than changing anything, and an agent that has to
ration looking things up starts guessing instead.
"""
from __future__ import annotations
import json as _json
from typing import Optional
import typer
from chemenu import config, links
from chemenu.commands._util import console, fail
from chemenu.kb_scan import load_kb_pages
from chemenu.page import Page
app = typer.Typer(help="Show the declared edges into and out of a page.")
EDGE_FIELD = "related"
def _collection_of(page: Page) -> Optional[str]:
try:
return page.path.relative_to(config.KB_DIR).parts[0]
except (ValueError, IndexError):
return None
def outbound(pages: dict[str, Page], title: str) -> list[dict]:
"""Edges this page asserts, in file order."""
page = pages[title]
return [
{"target": edge.target, "label": edge.label, "resolves": edge.target in pages}
for edge in links.edges(page.frontmatter, EDGE_FIELD)
]
def inbound(pages: dict[str, Page], title: str) -> list[dict]:
"""Edges other pages assert *about* this one.
A full scan of the corpus rather than a stored list, deliberately: the whole
argument for dropping mirrored edges is that this answer is derived and
therefore cannot go stale or be half-written.
"""
found = [
{"source": other, "label": edge.label, "collection": _collection_of(page)}
for other, page in pages.items()
for edge in links.edges(page.frontmatter, EDGE_FIELD)
if edge.target == title
]
return sorted(found, key=lambda item: (item["label"] or "", item["source"]))
@app.command("show")
def links_show(
page: str = typer.Option(..., "--page", help="Exact page title"),
json_out: bool = typer.Option(False, "--json", help="Print the edges as JSON"),
):
"""Show the edges out of and into a page.
Outbound is what the page declares in `related:`. Inbound is computed across
the corpus - nothing stores it, which is exactly why it is complete."""
pages = load_kb_pages(config.KB_DIR)
if page not in pages:
fail(f"No page titled '{page}' found under kb/.")
out, back = outbound(pages, page), inbound(pages, page)
if json_out:
typer.echo(_json.dumps({"page": page, "outbound": out, "inbound": back}, indent=2))
return
console.print(f"[bold]{page}[/bold]")
console.print(f"\n[cyan]asserts ({len(out)})[/cyan]")
if not out:
console.print(" (none)")
for edge in out:
label = edge["label"] or "[dim]unlabelled[/dim]"
missing = "" if edge["resolves"] else " [red](no such page)[/red]"
console.print(f" {label} -> [[{edge['target']}]]{missing}")
console.print(f"\n[cyan]asserted about it ({len(back)})[/cyan]")
if not back:
console.print(" (none - nothing in the corpus declares an edge to this page)")
for edge in back:
label = edge["label"] or "[dim]unlabelled[/dim]"
console.print(f" [[{edge['source']}]] {label} ->")
+91 -3
View File
@@ -64,6 +64,7 @@ def list_command(
"name": m.name,
"migrates_to": str(m.target),
"migration_kind": m.kind,
"obligation": m.obligation,
"description": m.description,
"path": m.relative_path,
}
@@ -78,7 +79,10 @@ def list_command(
success(f"No migration documents under {rel_path(kb_state.migrations_dir())}.")
return
for migration in migrations:
console.print(f"[bold]{migration.target}[/bold] {migration.name} ({migration.kind})")
console.print(
f"[bold]{migration.target}[/bold] {migration.name} "
f"({migration.kind}, {migration.obligation})"
)
if migration.description:
console.print(f" {migration.description}")
@@ -86,6 +90,50 @@ def list_command(
# --- migrate status --------------------------------------------------------
def _report_offers(
offered: list["kb_state.Migration"], divergent: Optional[list[str]]
) -> None:
"""Print the optional half of `status`, above the outstanding chain.
Deliberately never affects the exit code and never says "outstanding". An
offer is the stack proposing a better default for a file the instance owns;
an instance that keeps its own version is in a correct state, not a late
one. Mixing the two is how the message that actually matters - your content
no longer fits your machinery - stops being read.
"""
if not offered:
return
console.print(
f"[cyan]{len(offered)} optional upgrade(s) available[/cyan] - none of them block:"
)
for migration in offered:
console.print(f" {migration.target} {migration.name} ({migration.kind})")
if migration.description:
console.print(f" {migration.description}")
console.print(f" {migration.relative_path}")
if divergent is None:
console.print(
" [dim]This tree carries no release stamp, so which of your files still match "
"what you were given cannot be answered here.[/dim]"
)
return
if divergent:
console.print(
f" [dim]{len(divergent)} file(s) differ from the release you installed - those are "
"yours to reconcile by hand rather than overwrite:[/dim]"
)
for relative in divergent[:10]:
console.print(f" [dim]{relative}[/dim]")
if len(divergent) > 10:
console.print(f" [dim]... and {len(divergent) - 10} more[/dim]")
else:
console.print(
" [dim]No file differs from the release you installed, so an offer can be taken "
"by copying.[/dim]"
)
@app.command("status")
def status_command(
json_out: bool = typer.Option(False, "--json", help="Print the chain as JSON"),
@@ -113,6 +161,8 @@ def status_command(
return
pending = kb_state.chain(migrations, kb_version, stack)
offered = kb_state.offers(migrations, kb_state.applied_names(kb_state.read_kb_state()))
divergent = kb_state.divergent_files()
if json_out:
typer.echo(
@@ -124,6 +174,11 @@ def status_command(
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
for m in pending
],
"offered": [
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
for m in offered
],
"divergent_files": divergent,
},
indent=2,
)
@@ -131,6 +186,7 @@ def status_command(
return
console.print(f"stack {stack}, content {kb_version}")
_report_offers(offered, divergent)
if not pending:
if kb_version < stack:
console.print(
@@ -166,7 +222,12 @@ def done_command(
Refuses any version that is not the *next* link in the chain: skipping a
migration is how a corpus ends up in a shape no version describes, and an
interrupted multi-step upgrade has to be resumable rather than guessable."""
interrupted multi-step upgrade has to be resumable rather than guessable.
An `offered` migration is recorded but does not move the version, and no
ordering rule applies to it - it is not a link in the chain. The record is
the only thing that distinguishes an offer someone took from one they
ignored, precisely because the version stays put."""
stack, kb_version = _versions()
if kb_version is None:
fail(
@@ -182,6 +243,34 @@ def done_command(
return
migrations = kb_state.load_migrations()
state = kb_state.read_kb_state() or {}
# An offer is recorded but does not advance the version: it is not a link in
# the chain, so there is no ordering rule to check and nothing to skip. The
# ledger is what makes it stop being offered - without that record there
# would be no way to tell a taken offer from an ignored one, because
# `kb_version` deliberately does not move.
offered = {m.name: m for m in migrations if not m.is_required}
taken = next((m for m in offered.values() if str(m.target) == version), None)
if taken is not None:
if taken.name in kb_state.applied_names(state):
success(f"{taken.name} is already recorded as taken. Nothing to do.")
return
if dry_run:
success(f"Dry run: would record the optional {taken.name}. Nothing written.")
return
applied = list(state.get("applied") or [])
entry = {"migration": taken.name, "at": today_iso(), "obligation": kb_state.OFFERED}
if pages is not None:
entry["pages"] = pages
applied.append(entry)
kb_state.write_kb_state(kb_version, applied)
success(
f"Recorded the optional {taken.name}. Content stays at {kb_version} - an offer "
"changes a file you own, not the shape of your content."
)
return
expected = kb_state.next_link(migrations, kb_version, stack)
if expected is None:
fail(
@@ -197,7 +286,6 @@ def done_command(
)
return
state = kb_state.read_kb_state() or {}
applied = list(state.get("applied") or [])
entry = {"migration": expected.name, "at": today_iso()}
if pages is not None:
+2 -7
View File
@@ -24,7 +24,7 @@ import re
import typer
from chemenu import config, conventions
from chemenu import config
from chemenu.commands._util import (
check_collision,
check_raw_files_exist,
@@ -166,11 +166,7 @@ def _apply_template_variables(template: str, variables: Dict[str, Any]) -> str:
"""Apply variable substitutions to a template string.
Supports:
- `{field}` - plain substitution from `variables[field]`, including the
`{section.<slot>}` names this instance gave the three tool-owned
headings (see chemenu.conventions). Those are what took the KB language
out of `types/*.md`: a template writes `## {section.relationships}`, so
scaffolding a page in another language needs no edit under `types/`
- `{field}` - plain substitution from `variables[field]`
- `{field|filter}` - apply a named filter (bullets, join, capitalize)
to `variables[field]`'s value, so templates can render list/enum
frontmatter fields directly instead of the caller precomputing a
@@ -333,7 +329,6 @@ def new_page_command(
**frontmatter,
"name": name,
"today": today.isoformat(),
**conventions.section_variables(),
},
)
+11 -8
View File
@@ -25,7 +25,7 @@ from typing import Optional
import typer
from chemenu import config
from chemenu import config, links
from chemenu.commands._util import check_collision, fail, rel_path, success
from chemenu.frontmatter_io import write_page
from chemenu.page import Page
@@ -33,7 +33,6 @@ from chemenu.kb_scan import load_kb_pages
from chemenu.provenance import (
CITE_REF_RE,
cite_id,
cite_block_heading,
render_page_body,
split_cite_block,
unique_cite_id,
@@ -103,7 +102,7 @@ def retarget_cite_ids(body: str, old: str, new: str) -> str:
return body
new_head = CITE_REF_RE.sub(lambda m: f"[^{renames.get(m.group(1), m.group(1))}]", head)
return render_page_body(new_head, new_definitions, cite_block_heading(body))
return render_page_body(new_head, new_definitions)
def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
@@ -114,9 +113,11 @@ def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
values = page.frontmatter.get(field)
if not values:
continue
updated = [new if value == old else value for value in values]
if updated != values:
page.frontmatter[field] = updated
# Through `links` so a labelled edge keeps its label across a rename:
# the entry is `{label: target}`, and a plain equality swap would have
# compared the mapping against a title and silently left it pointing at
# the old page.
if links.retarget(page.frontmatter, field, old, new):
changed = True
return changed
@@ -157,8 +158,10 @@ def strip_frontmatter_ref(page: Page, title: str) -> bool:
values = page.frontmatter.get(field)
if not values:
continue
updated = [value for value in values if value != title]
if updated == values:
before = list(values)
links.remove(page.frontmatter, field, title)
updated = page.frontmatter.get(field) or []
if updated == before:
continue
if not updated and field not in declared:
del page.frontmatter[field]
+4
View File
@@ -82,6 +82,10 @@ SKIP_COMMAND_PATHS = {
("eval", "score"),
("eval", "sessions"),
("cite", "id"),
# Retrieval, like `search`: an agent that has to ration looking up what
# points at a page starts guessing instead - and under authored directional
# edges this is the *only* way to ask that question.
("links", "show"),
("version", "show"),
("version", "check"),
("version", "notes"),
+116 -96
View File
@@ -1,9 +1,22 @@
"""Bidirectional cross-reference management between wiki pages.
"""Cross-reference management between wiki pages.
`xref add` keeps two pages' frontmatter `related:` lists AND their body
"## Relationships" sections in sync in one operation, instead of the 3-5
separate manual edits this used to take per pair of pages. It is idempotent:
re-running it never duplicates a link.
`xref add` writes **one** edge: a label plus a target, into the asserting page's
`related:` frontmatter, and re-renders that page's generated links region from
it. It is idempotent, and re-running with a different label relabels rather than
duplicating.
It used to write four things at once - `related:` and a Relationships bullet on
both pages, plus reciprocal See Also bullets. That made every edge symmetric by
construction, which is not what a link means: an edge is an authored reader aid,
and "follow this to verify the premise" rarely reads the same from the other
end. Worse, it is incompatible with per-collection label authorisation, because
the mirrored half is written into a collection whose rules the author never
read.
The reverse direction is therefore authored separately, when it is a primary
statement of its own - and navigation does not depend on anyone bothering:
`index rebuild` renders the inbound view from the graph, completely and without
maintenance. See instructions/link-taxonomy.md.
"""
from __future__ import annotations
@@ -12,7 +25,7 @@ from pathlib import Path
import typer
from chemenu import config, sections
from chemenu import blocks, config, conventions, kb_collections, links
from chemenu.commands._util import fail, parse_list, success
from chemenu.commands.page_ops import strip_frontmatter_ref
from chemenu.frontmatter_io import write_page
@@ -71,121 +84,119 @@ def _back_reference_field(source: Page, target: Page) -> str | None:
return collection if collection in _declared_ref_fields(source) else None
def add_related(frontmatter: dict, other_title: str) -> bool:
"""Add other_title to frontmatter['related'] if not already present.
Returns True if a change was made."""
related = frontmatter.setdefault("related", [])
if other_title in related:
return False
related.append(other_title)
return True
def _section_bounds(body: str, heading: str) -> tuple[int, int] | None:
match = sections.heading_re(heading).search(body)
if not match:
def _collection_of(page: Page) -> str | None:
"""The collection a page lives in, or None if it is outside `kb/`."""
try:
return page.path.relative_to(config.KB_DIR).parts[0]
except (ValueError, IndexError):
return None
start = match.end()
next_heading = re.search(r"^## ", body[start:], re.MULTILINE)
end = start + next_heading.start() if next_heading else len(body)
return start, end
def add_bullet_to_section(body: str, heading: str, bullet: str, dedup_link: str) -> str:
"""Insert `bullet` into the `## {heading}` section of body, unless a
wikilink to dedup_link already appears there. Creates the section
(before the See Also section if present, else at the end) if missing.
def render_links_block(page: Page) -> str:
"""The page's generated links region, built from its `related:` edges.
`heading` is a canonical name from `sections`; an existing section is found
under its aliases too, so a page that has not been translated yet is still
appended to rather than given a duplicate section. A section this creates
always carries the canonical name."""
bounds = _section_bounds(body, heading)
if bounds is None:
section = f"## {heading}\n\n{bullet}\n\n"
see_also = sections.heading_re(sections.SEE_ALSO).search(body)
if heading != sections.SEE_ALSO and see_also:
return body[: see_also.start()] + section + body[see_also.start() :]
return body.rstrip("\n") + "\n\n" + section.rstrip("\n") + "\n"
start, end = bounds
section_text = body[start:end]
if f"[[{dedup_link}]]" in section_text:
return body
trimmed = section_text.rstrip("\n")
new_section = trimmed + "\n" + bullet + "\n\n"
return body[:start] + new_section + body[end:]
The body is a *rendering* of the frontmatter, not a second place the graph
is stored. That is what removed the need to parse a German bullet back into
a relationship: the label lives in the data, and this writes it out.
"""
lines = []
for edge in links.edges(page.frontmatter, "related"):
if edge.is_labelled:
lines.append(f"- **{edge.label}:** [[{edge.target}]]")
else:
lines.append(f"- [[{edge.target}]]")
return blocks.render(blocks.LINKS, conventions.heading(blocks.LINKS), lines)
def add_relationship_bullet(body: str, label: str, other_title: str) -> str:
bullet = f"- **{label}:** [[{other_title}]]"
return add_bullet_to_section(body, sections.RELATIONSHIPS, bullet, other_title)
def apply_links_block(page: Page, body: str | None = None) -> str:
"""`body` with the links region re-rendered from `page.frontmatter`."""
return blocks.replace(
page.body if body is None else body, blocks.LINKS, render_links_block(page)
)
def add_see_also_bullet(body: str, other_title: str) -> str:
return add_bullet_to_section(body, sections.SEE_ALSO, f"- [[{other_title}]]", other_title)
def _check_authorised(source: Page, target: Page, label: str) -> None:
"""Refuse a label the source collection has not authorised for that
destination.
Checked here rather than only in `lint` because this is the moment the
author is present: a refusal names the authorised set and can be answered by
picking a better label, while a lint finding a day later is answered by
whoever is holding the report.
"""
source_collection = _collection_of(source)
destination = _collection_of(target)
if source_collection is None or destination is None:
return
allowed = kb_collections.authorised_labels(source_collection, destination)
if not allowed:
fail(
f"kb/{source_collection}/COLLECTION.md authorises no labels for edges into "
f"kb/{destination}/. Add an `outbound:` entry for it, or do not link there "
f"from this collection."
)
if label not in allowed:
fail(
f"'{label}' is not authorised for kb/{source_collection}/ -> kb/{destination}/.\n"
f" Authorised: {', '.join(sorted(allowed))}\n"
f" The catalogue and what each label asserts: instructions/link-taxonomy.md\n"
f" Authorising a further label is a deliberate edit to "
f"kb/{source_collection}/COLLECTION.md, not a way around this refusal."
)
@app.command("add")
def xref_add(
a: str = typer.Option(..., "--a", help="Exact title of page A"),
b: str = typer.Option(..., "--b", help="Exact title of page B"),
rel_a: str = typer.Option("related to", "--rel-a", help="Relationship label on A pointing to B"),
rel_b: str = typer.Option("related to", "--rel-b", help="Relationship label on B pointing to A"),
see_also: bool = typer.Option(True, "--see-also/--no-see-also", help="Also add reciprocal 'See Also' bullets"),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes to both pages instead of writing"),
a: str = typer.Option(..., "--a", help="Exact title of the page that asserts the edge"),
b: str = typer.Option(..., "--b", help="Exact title of the page it points at"),
rel: str = typer.Option(
..., "--rel", help="Label from instructions/link-taxonomy.md, e.g. depends-on"
),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview the change instead of writing"),
):
"""Declare that A <rel> B. One edge, on A only.
Say the sentence before choosing the label: `[A] <rel> [B]`. If it only
reads true backwards, the edge belongs on B - run this the other way round
rather than reaching for an inverse label.
B is not modified and does not need to point back. Its inbound view is
rendered from the graph.
"""
pages = load_kb_pages(config.KB_DIR)
page_a = _find_page(pages, a)
page_b = _find_page(pages, b)
# Both refusals before either write, so a rejected pair leaves no half-link.
# Every refusal before the single write, so a rejected edge leaves nothing.
_require_related_field(page_a, a)
_require_related_field(page_b, b)
_check_authorised(page_a, page_b, rel)
related_changed_a = add_related(page_a.frontmatter, b)
related_changed_b = add_related(page_b.frontmatter, a)
body_a = add_relationship_bullet(page_a.body, rel_a, b)
body_b = add_relationship_bullet(page_b.body, rel_b, a)
if see_also:
body_a = add_see_also_bullet(body_a, b)
body_b = add_see_also_bullet(body_b, a)
changed_a = related_changed_a or body_a != page_a.body
changed_b = related_changed_b or body_b != page_b.body
changed = links.upsert(page_a.frontmatter, "related", links.Edge(b, rel))
body = apply_links_block(page_a)
changed = changed or body != page_a.body
if dry_run:
state_a = "would update" if changed_a else "already up to date"
state_b = "would update" if changed_b else "already up to date"
typer.echo(f"[dry-run] '{a}': {state_a} (related / Relationships / See Also)")
typer.echo(f"[dry-run] '{b}': {state_b} (related / Relationships / See Also)")
typer.echo(
f"[dry-run] '{a}': {'would declare' if changed else 'already declares'} "
f"{rel} -> '{b}'"
)
typer.echo("No files written (--dry-run).")
return
try:
write_page(page_a.path, page_a.frontmatter, body_a)
except OSError as exc:
fail(f"Failed to write '{a}': {exc}. '{b}' was not touched - fix the write failure and retry once.")
if not changed:
success(f"'{a}' already declares {rel} -> '{b}'; nothing changed.")
return
try:
write_page(page_b.path, page_b.frontmatter, body_b)
write_page(page_a.path, page_a.frontmatter, body)
except OSError as exc:
fail(
f"'{a}' was updated but writing '{b}' failed: {exc}. The link is now one-directional - "
f"fix the write failure, then re-run `xref add --a \"{a}\" --b \"{b}\"` (idempotent, safe to retry)."
)
success(f"Linked '{a}' <-> '{b}' ({rel_a} / {rel_b})")
fail(f"Failed to write '{a}': {exc}")
success(f"'{a}' {rel} '{b}'")
def remove_related(frontmatter: dict, other_title: str) -> bool:
"""Drop other_title from frontmatter['related'] if present. Returns True if
a change was made."""
related = frontmatter.get("related")
if not related or other_title not in related:
return False
frontmatter["related"] = [title for title in related if title != other_title]
return True
"""Drop every edge pointing at other_title. True if a change was made."""
return links.remove(frontmatter, "related", other_title)
def remove_link_bullets(body: str, other_title: str) -> str:
"""Remove the whole-line Relationships/See Also bullets `xref add` writes -
@@ -223,14 +234,19 @@ def xref_remove(
page_a = _find_page(pages, a)
page_b = pages.get(b)
body_a = remove_link_bullets(page_a.body, b)
changed_a = strip_frontmatter_ref(page_a, b) or body_a != page_a.body
# The frontmatter first, then the region re-rendered from it - the body is a
# rendering, so editing the bullet out directly would leave an empty region
# behind and, worse, put the two out of step.
changed_a = strip_frontmatter_ref(page_a, b)
body_a = apply_links_block(page_a, remove_link_bullets(page_a.body, b))
changed_a = changed_a or body_a != page_a.body
changed_b = False
body_b = ""
if page_b is not None:
body_b = remove_link_bullets(page_b.body, a)
changed_b = strip_frontmatter_ref(page_b, a) or body_b != page_b.body
changed_b = strip_frontmatter_ref(page_b, a)
body_b = apply_links_block(page_b, remove_link_bullets(page_b.body, a))
changed_b = changed_b or body_b != page_b.body
if dry_run:
typer.echo(f"[dry-run] '{a}': {'would update' if changed_a else 'no reference to remove'}")
@@ -278,7 +294,11 @@ def xref_link_source(
sources = page.frontmatter.setdefault("sources", [])
if source not in sources:
sources.append(source)
body = add_see_also_bullet(page.body, source)
# No body bullet. `sources:` *is* the record, and the See Also bullet
# this used to add was the reciprocal half of a bidirectional model
# that no longer exists - 353 of the corpus's 555 such bullets were
# provably redundant with an edge that already said the same thing.
body = page.body
# The way back. Until this existed the command wrote only the targets,
# so a source page's own `entities:`/`concepts:` stayed as `new` left