feat: Prosa ist kein Identifier - Link-Taxonomie als Enum, generierte Regionen mit Markern (4.0.0)
Files changed: - .gitea/workflows/ci.yml - AGENTS.md - CHANGES.md - VERSION - instructions/CONTRACT.md - instructions/link-taxonomy.md - instructions/migrations/4.0.0-link-taxonomy.md - instructions/setup-instance.md - kb/CONTRACT.md - kb/CONVENTIONS.md - kb/CONVENTIONS.md.template - kb/comparisons/COLLECTION.md - kb/concepts/COLLECTION.md - kb/entities/COLLECTION.md - kb/sources/COLLECTION.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/blocks.py - tools/chemenu/cli.py - tools/chemenu/commands/cite_cmd.py - tools/chemenu/commands/dist_cmd.py - tools/chemenu/commands/docs_verify.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/links_cmd.py - tools/chemenu/commands/migrate_cmd.py - tools/chemenu/commands/new_page.py - tools/chemenu/commands/page_ops.py - tools/chemenu/commands/run_budget.py - tools/chemenu/commands/xref.py - tools/chemenu/conventions.py - tools/chemenu/corpus_diff.py - tools/chemenu/frontmatter_io.py - tools/chemenu/kb_collections.py - tools/chemenu/kb_state.py - tools/chemenu/links.py - tools/chemenu/lint_core.py - tools/chemenu/provenance.py - tools/chemenu/sections.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_blocks.py - tools/chemenu/tests/test_cite_cmd.py - tools/chemenu/tests/test_conventions.py - tools/chemenu/tests/test_dist_cmd.py - tools/chemenu/tests/test_doctor.py - tools/chemenu/tests/test_migrate_cmd.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_pipeline_l0.py - tools/chemenu/tests/test_types_cmd.py - tools/chemenu/tests/test_xref.py - types/concept.schema.yaml - types/entity.md - types/entity.schema.yaml - types/instruction.schema.yaml - types/type-spec.md - work/link-taxonomy-migration/README.md - work/link-taxonomy-migration/plan.md
This commit is contained in:
1 parent
502971d147
commit
177c7e9ce8
56 files changed
+2692
-750
No files matched your search
@@ -28,7 +28,6 @@ from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.provenance import (
|
||||
CITE_REF_RE,
|
||||
cite_id,
|
||||
cite_block_heading,
|
||||
render_page_body,
|
||||
split_cite_block,
|
||||
unique_cite_id,
|
||||
@@ -81,7 +80,7 @@ def upsert_citation(page: Page, source_title: str, qualifier: Optional[str]) ->
|
||||
if sources_changed:
|
||||
sources.append(source_title)
|
||||
|
||||
new_body = render_page_body(head, definitions, cite_block_heading(page.body))
|
||||
new_body = render_page_body(head, definitions)
|
||||
changed = block_changed or sources_changed or new_body != page.body
|
||||
return marker_id, new_body, changed
|
||||
|
||||
@@ -152,7 +151,7 @@ def sync_page(page: Page) -> tuple[str, bool, list[str], list[str]]:
|
||||
ordered[cid] = definitions[cid]
|
||||
seen.add(cid)
|
||||
|
||||
new_body = render_page_body(head, ordered, cite_block_heading(page.body))
|
||||
new_body = render_page_body(head, ordered)
|
||||
changed = new_body != page.body
|
||||
return new_body, changed, pruned, undefined
|
||||
|
||||
|
||||
@@ -264,6 +264,60 @@ class Origin(NamedTuple):
|
||||
update_url: Optional[str] = None
|
||||
|
||||
|
||||
def instance_owned_type_stems() -> set[str]:
|
||||
"""Type-spec stems whose instances are knowledge pages, and which therefore
|
||||
belong to the instance rather than to the stack.
|
||||
|
||||
The line is `root:`, and it was already in the frontmatter before anyone
|
||||
drew it: `root: kb` means the type describes a page the instance writes, so
|
||||
its prose, its template and its language are the instance's business.
|
||||
Anything else - `instruction` (`root: repo`), `lint-report` (no `base_dir`
|
||||
at all), `type-spec` itself - describes a stack artifact and ships verbatim.
|
||||
|
||||
Read from `types/` rather than listed, so an instance adding its own page
|
||||
type gets the same treatment without a code change.
|
||||
"""
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
stems: set[str] = set()
|
||||
for type_path, frontmatter in resolver.list_type_specs():
|
||||
stem = Path(type_path).stem
|
||||
if stem == "type-spec":
|
||||
continue
|
||||
if not frontmatter.get("base_dir"):
|
||||
continue
|
||||
if (frontmatter.get("root") or "kb") != "kb":
|
||||
continue
|
||||
stems.add(stem)
|
||||
return stems
|
||||
|
||||
|
||||
def _plan_types() -> dict[str, PlannedFile]:
|
||||
"""`types/`, with the page type-specs re-keyed as templates.
|
||||
|
||||
Same split as the collection contracts, for the same reason and by the same
|
||||
mechanism: the shipped content is a working default rather than something
|
||||
wrong for the receiver, so the file itself crosses - under a name that has
|
||||
to be adopted before it counts. A type-spec's `.schema.yaml` travels with
|
||||
it, because the two are one type (see types/type-spec.md § Anatomy) and
|
||||
adopting half of it would leave a spec validated by a file it does not own.
|
||||
"""
|
||||
plan = _copy_tree(config.TYPES_DIR, "types", frozenset())
|
||||
stems = instance_owned_type_stems()
|
||||
if not stems:
|
||||
return plan
|
||||
|
||||
rekeyed: dict[str, PlannedFile] = {}
|
||||
for relative, planned in plan.items():
|
||||
name = relative.rsplit("/", 1)[-1]
|
||||
stem = name.split(".", 1)[0]
|
||||
if stem in stems:
|
||||
rekeyed[f"{relative}.template"] = planned
|
||||
else:
|
||||
rekeyed[relative] = planned
|
||||
return rekeyed
|
||||
|
||||
|
||||
def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
|
||||
"""Every (destination-relative path -> planned file) the export writes."""
|
||||
plan: dict[str, PlannedFile] = {}
|
||||
@@ -286,7 +340,7 @@ def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
|
||||
plan[name] = _read_planned_file(source, name)
|
||||
|
||||
plan.update(_copy_tree(config.INSTRUCTIONS_DIR, "instructions", frozenset(INSTRUCTIONS_EXCLUDE_DIRS)))
|
||||
plan.update(_copy_tree(config.TYPES_DIR, "types", frozenset()))
|
||||
plan.update(_plan_types())
|
||||
plan.update(_copy_tree(
|
||||
config.ROOT / "tools", "tools", frozenset(TOOLS_EXCLUDE_DIRS), _is_coverage_output
|
||||
))
|
||||
@@ -382,6 +436,7 @@ _INSTANCE_OWNED_KB_FILES = (kb_collections.CONTRACT_NAME, conventions.CONVENTION
|
||||
|
||||
def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
|
||||
"""Planned paths that carry one instance's own data instead of machinery."""
|
||||
owned_types = instance_owned_type_stems()
|
||||
leaks: list[str] = []
|
||||
for relative in sorted(plan):
|
||||
name = relative.rsplit("/", 1)[-1]
|
||||
@@ -389,6 +444,12 @@ def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
|
||||
leaks.append(f"{relative} (one instance's own personalization)")
|
||||
elif relative.startswith("kb/") and name in _INSTANCE_OWNED_KB_FILES:
|
||||
leaks.append(f"{relative} (this instance's authoring conventions; ship the .template)")
|
||||
elif (
|
||||
relative.startswith("types/")
|
||||
and not relative.endswith(".template")
|
||||
and name.split(".", 1)[0] in owned_types
|
||||
):
|
||||
leaks.append(f"{relative} (this instance's page type-spec; ship the .template)")
|
||||
elif relative.startswith("instructions/dev/"):
|
||||
leaks.append(f"{relative} (stack-development only)")
|
||||
elif relative.startswith(_CONTENT_PREFIXES) and name not in _CONTENT_ALLOWED_NAMES:
|
||||
|
||||
@@ -260,10 +260,53 @@ def check_collection_contracts() -> list[str]:
|
||||
|
||||
issues += kb_collections.declaration_issues()
|
||||
issues += conventions.declaration_issues()
|
||||
issues += check_stack_required_types()
|
||||
|
||||
return issues
|
||||
|
||||
|
||||
def check_stack_required_types() -> list[str]:
|
||||
"""The minimum the stack asks of the type layer, and nothing beyond it.
|
||||
|
||||
The four page type-specs belong to the instance: it may translate them,
|
||||
rewrite their templates, add sections. What it may not do is remove the one
|
||||
type the provenance path is built on, or drop the field that path reads.
|
||||
Everything else about `types/source.md` - its prose, its template, its title
|
||||
prefix, its directory - is the instance's, and is deliberately not checked
|
||||
here.
|
||||
"""
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
issues: list[str] = []
|
||||
for type_name in kb_collections.STACK_REQUIRED_TYPES:
|
||||
try:
|
||||
type_path = resolver.find_type_by_name(type_name)
|
||||
except (ValueError, OSError) as exc:
|
||||
issues.append(f"types/ could not be read to find the `{type_name}` type: {exc}")
|
||||
continue
|
||||
if not type_path:
|
||||
issues.append(
|
||||
f"no type-spec declares `name: {type_name}` - `sources coverage`, `[^cite-id]` "
|
||||
f"resolution and `kb/provenance.md` all ask `page.kind == \"{type_name}\"`, so "
|
||||
f"without it the whole raw/ -> kb/ provenance path resolves against nothing"
|
||||
)
|
||||
continue
|
||||
try:
|
||||
schema = resolver.get_schema(type_path) or {}
|
||||
except (ValueError, OSError) as exc:
|
||||
issues.append(f"{type_path}: its schema could not be read: {exc}")
|
||||
continue
|
||||
declared = set(schema.get("required") or [])
|
||||
for field in kb_collections.STACK_REQUIRED_TYPE_FIELDS.get(type_name, ()):
|
||||
if field not in declared:
|
||||
issues.append(
|
||||
f"{type_path}: its schema must require `{field}` - it is what the "
|
||||
f"provenance path reads, and a `{type_name}` page without it claims no "
|
||||
f"raw material at all"
|
||||
)
|
||||
return issues
|
||||
|
||||
|
||||
def check_legacy_type_blocks() -> list[str]:
|
||||
issues = []
|
||||
guarded = [
|
||||
|
||||
@@ -217,17 +217,18 @@ def check_conventions() -> Check:
|
||||
"""Whether this instance has said how its own pages are written.
|
||||
|
||||
`kb/CONVENTIONS.md` carries the decisions `kb/CONTRACT.md` deliberately no
|
||||
longer makes: the KB language and its three tool-owned section headings, the
|
||||
relationship-label vocabulary, the tone examples, the confidence rubric, the
|
||||
ADR prefix. The compiler reads the section names out of it, so an instance
|
||||
without one is not merely undocumented - `xref add` and `cite add` fall back
|
||||
to the names this stack hardcoded before the file existed, which is right
|
||||
only for a corpus that was written under them.
|
||||
longer makes: the KB language and the headings its two generated regions
|
||||
render under, the tone examples, the confidence rubric, the naming forms.
|
||||
|
||||
Hence `FAIL` rather than `WARN`, and hence the same two failure modes the
|
||||
personalization pair has: the distribution can ship the template but never
|
||||
the filled file, so a template renamed and left unanswered looks present and
|
||||
decides nothing.
|
||||
`FAIL` rather than `WARN` because those decisions bind every page, and
|
||||
because it has the same two failure modes the personalization pair has: the
|
||||
distribution can ship the template but never the filled file, so a template
|
||||
renamed and left unanswered looks present and decides nothing.
|
||||
|
||||
The headings themselves are only cosmetic now - the marker pair carries each
|
||||
region's identity, so a default renders wrong words rather than corrupting
|
||||
structure. That is why this check is about the *file*, not about rescuing a
|
||||
lookup the compiler can no longer get wrong.
|
||||
"""
|
||||
path = conventions.conventions_file()
|
||||
fix = (
|
||||
@@ -246,7 +247,9 @@ def check_conventions() -> Check:
|
||||
if issues:
|
||||
return Check("conventions", "FAIL", "; ".join(issues), fix)
|
||||
declared = conventions.language() or "unspecified"
|
||||
headings = ", ".join(conventions.canonical(slot) for slot in conventions.SLOTS)
|
||||
from chemenu import blocks
|
||||
|
||||
headings = ", ".join(conventions.heading(block) for block in blocks.BLOCKS)
|
||||
return Check(
|
||||
"conventions", "OK",
|
||||
f"kb/{conventions.CONVENTIONS_FILENAME} present, language {declared}, "
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
"""`wikitool links` - the declared graph around one page, both directions.
|
||||
|
||||
The half that makes authored directional edges liveable. An edge is written once,
|
||||
on the page that asserts it, so the question "what points at *this* page" has no
|
||||
answer stored anywhere - it is computed from the graph, which is the only way it
|
||||
is ever complete. A mirrored edge only ever recorded what someone remembered to
|
||||
mirror.
|
||||
|
||||
Read-only, and exempt from the iteration budget for the same reason `search` is:
|
||||
it answers a question rather than changing anything, and an agent that has to
|
||||
ration looking things up starts guessing instead.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json as _json
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, links
|
||||
from chemenu.commands._util import console, fail
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.page import Page
|
||||
|
||||
app = typer.Typer(help="Show the declared edges into and out of a page.")
|
||||
|
||||
EDGE_FIELD = "related"
|
||||
|
||||
|
||||
def _collection_of(page: Page) -> Optional[str]:
|
||||
try:
|
||||
return page.path.relative_to(config.KB_DIR).parts[0]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
|
||||
|
||||
def outbound(pages: dict[str, Page], title: str) -> list[dict]:
|
||||
"""Edges this page asserts, in file order."""
|
||||
page = pages[title]
|
||||
return [
|
||||
{"target": edge.target, "label": edge.label, "resolves": edge.target in pages}
|
||||
for edge in links.edges(page.frontmatter, EDGE_FIELD)
|
||||
]
|
||||
|
||||
|
||||
def inbound(pages: dict[str, Page], title: str) -> list[dict]:
|
||||
"""Edges other pages assert *about* this one.
|
||||
|
||||
A full scan of the corpus rather than a stored list, deliberately: the whole
|
||||
argument for dropping mirrored edges is that this answer is derived and
|
||||
therefore cannot go stale or be half-written.
|
||||
"""
|
||||
found = [
|
||||
{"source": other, "label": edge.label, "collection": _collection_of(page)}
|
||||
for other, page in pages.items()
|
||||
for edge in links.edges(page.frontmatter, EDGE_FIELD)
|
||||
if edge.target == title
|
||||
]
|
||||
return sorted(found, key=lambda item: (item["label"] or "", item["source"]))
|
||||
|
||||
|
||||
@app.command("show")
|
||||
def links_show(
|
||||
page: str = typer.Option(..., "--page", help="Exact page title"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the edges as JSON"),
|
||||
):
|
||||
"""Show the edges out of and into a page.
|
||||
|
||||
Outbound is what the page declares in `related:`. Inbound is computed across
|
||||
the corpus - nothing stores it, which is exactly why it is complete."""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
if page not in pages:
|
||||
fail(f"No page titled '{page}' found under kb/.")
|
||||
|
||||
out, back = outbound(pages, page), inbound(pages, page)
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps({"page": page, "outbound": out, "inbound": back}, indent=2))
|
||||
return
|
||||
|
||||
console.print(f"[bold]{page}[/bold]")
|
||||
console.print(f"\n[cyan]asserts ({len(out)})[/cyan]")
|
||||
if not out:
|
||||
console.print(" (none)")
|
||||
for edge in out:
|
||||
label = edge["label"] or "[dim]unlabelled[/dim]"
|
||||
missing = "" if edge["resolves"] else " [red](no such page)[/red]"
|
||||
console.print(f" {label} -> [[{edge['target']}]]{missing}")
|
||||
|
||||
console.print(f"\n[cyan]asserted about it ({len(back)})[/cyan]")
|
||||
if not back:
|
||||
console.print(" (none - nothing in the corpus declares an edge to this page)")
|
||||
for edge in back:
|
||||
label = edge["label"] or "[dim]unlabelled[/dim]"
|
||||
console.print(f" [[{edge['source']}]] {label} ->")
|
||||
@@ -64,6 +64,7 @@ def list_command(
|
||||
"name": m.name,
|
||||
"migrates_to": str(m.target),
|
||||
"migration_kind": m.kind,
|
||||
"obligation": m.obligation,
|
||||
"description": m.description,
|
||||
"path": m.relative_path,
|
||||
}
|
||||
@@ -78,7 +79,10 @@ def list_command(
|
||||
success(f"No migration documents under {rel_path(kb_state.migrations_dir())}.")
|
||||
return
|
||||
for migration in migrations:
|
||||
console.print(f"[bold]{migration.target}[/bold] {migration.name} ({migration.kind})")
|
||||
console.print(
|
||||
f"[bold]{migration.target}[/bold] {migration.name} "
|
||||
f"({migration.kind}, {migration.obligation})"
|
||||
)
|
||||
if migration.description:
|
||||
console.print(f" {migration.description}")
|
||||
|
||||
@@ -86,6 +90,50 @@ def list_command(
|
||||
# --- migrate status --------------------------------------------------------
|
||||
|
||||
|
||||
def _report_offers(
|
||||
offered: list["kb_state.Migration"], divergent: Optional[list[str]]
|
||||
) -> None:
|
||||
"""Print the optional half of `status`, above the outstanding chain.
|
||||
|
||||
Deliberately never affects the exit code and never says "outstanding". An
|
||||
offer is the stack proposing a better default for a file the instance owns;
|
||||
an instance that keeps its own version is in a correct state, not a late
|
||||
one. Mixing the two is how the message that actually matters - your content
|
||||
no longer fits your machinery - stops being read.
|
||||
"""
|
||||
if not offered:
|
||||
return
|
||||
console.print(
|
||||
f"[cyan]{len(offered)} optional upgrade(s) available[/cyan] - none of them block:"
|
||||
)
|
||||
for migration in offered:
|
||||
console.print(f" {migration.target} {migration.name} ({migration.kind})")
|
||||
if migration.description:
|
||||
console.print(f" {migration.description}")
|
||||
console.print(f" {migration.relative_path}")
|
||||
|
||||
if divergent is None:
|
||||
console.print(
|
||||
" [dim]This tree carries no release stamp, so which of your files still match "
|
||||
"what you were given cannot be answered here.[/dim]"
|
||||
)
|
||||
return
|
||||
if divergent:
|
||||
console.print(
|
||||
f" [dim]{len(divergent)} file(s) differ from the release you installed - those are "
|
||||
"yours to reconcile by hand rather than overwrite:[/dim]"
|
||||
)
|
||||
for relative in divergent[:10]:
|
||||
console.print(f" [dim]{relative}[/dim]")
|
||||
if len(divergent) > 10:
|
||||
console.print(f" [dim]... and {len(divergent) - 10} more[/dim]")
|
||||
else:
|
||||
console.print(
|
||||
" [dim]No file differs from the release you installed, so an offer can be taken "
|
||||
"by copying.[/dim]"
|
||||
)
|
||||
|
||||
|
||||
@app.command("status")
|
||||
def status_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the chain as JSON"),
|
||||
@@ -113,6 +161,8 @@ def status_command(
|
||||
return
|
||||
|
||||
pending = kb_state.chain(migrations, kb_version, stack)
|
||||
offered = kb_state.offers(migrations, kb_state.applied_names(kb_state.read_kb_state()))
|
||||
divergent = kb_state.divergent_files()
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
@@ -124,6 +174,11 @@ def status_command(
|
||||
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
|
||||
for m in pending
|
||||
],
|
||||
"offered": [
|
||||
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
|
||||
for m in offered
|
||||
],
|
||||
"divergent_files": divergent,
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
@@ -131,6 +186,7 @@ def status_command(
|
||||
return
|
||||
|
||||
console.print(f"stack {stack}, content {kb_version}")
|
||||
_report_offers(offered, divergent)
|
||||
if not pending:
|
||||
if kb_version < stack:
|
||||
console.print(
|
||||
@@ -166,7 +222,12 @@ def done_command(
|
||||
|
||||
Refuses any version that is not the *next* link in the chain: skipping a
|
||||
migration is how a corpus ends up in a shape no version describes, and an
|
||||
interrupted multi-step upgrade has to be resumable rather than guessable."""
|
||||
interrupted multi-step upgrade has to be resumable rather than guessable.
|
||||
|
||||
An `offered` migration is recorded but does not move the version, and no
|
||||
ordering rule applies to it - it is not a link in the chain. The record is
|
||||
the only thing that distinguishes an offer someone took from one they
|
||||
ignored, precisely because the version stays put."""
|
||||
stack, kb_version = _versions()
|
||||
if kb_version is None:
|
||||
fail(
|
||||
@@ -182,6 +243,34 @@ def done_command(
|
||||
return
|
||||
|
||||
migrations = kb_state.load_migrations()
|
||||
state = kb_state.read_kb_state() or {}
|
||||
|
||||
# An offer is recorded but does not advance the version: it is not a link in
|
||||
# the chain, so there is no ordering rule to check and nothing to skip. The
|
||||
# ledger is what makes it stop being offered - without that record there
|
||||
# would be no way to tell a taken offer from an ignored one, because
|
||||
# `kb_version` deliberately does not move.
|
||||
offered = {m.name: m for m in migrations if not m.is_required}
|
||||
taken = next((m for m in offered.values() if str(m.target) == version), None)
|
||||
if taken is not None:
|
||||
if taken.name in kb_state.applied_names(state):
|
||||
success(f"{taken.name} is already recorded as taken. Nothing to do.")
|
||||
return
|
||||
if dry_run:
|
||||
success(f"Dry run: would record the optional {taken.name}. Nothing written.")
|
||||
return
|
||||
applied = list(state.get("applied") or [])
|
||||
entry = {"migration": taken.name, "at": today_iso(), "obligation": kb_state.OFFERED}
|
||||
if pages is not None:
|
||||
entry["pages"] = pages
|
||||
applied.append(entry)
|
||||
kb_state.write_kb_state(kb_version, applied)
|
||||
success(
|
||||
f"Recorded the optional {taken.name}. Content stays at {kb_version} - an offer "
|
||||
"changes a file you own, not the shape of your content."
|
||||
)
|
||||
return
|
||||
|
||||
expected = kb_state.next_link(migrations, kb_version, stack)
|
||||
if expected is None:
|
||||
fail(
|
||||
@@ -197,7 +286,6 @@ def done_command(
|
||||
)
|
||||
return
|
||||
|
||||
state = kb_state.read_kb_state() or {}
|
||||
applied = list(state.get("applied") or [])
|
||||
entry = {"migration": expected.name, "at": today_iso()}
|
||||
if pages is not None:
|
||||
|
||||
@@ -24,7 +24,7 @@ import re
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, conventions
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import (
|
||||
check_collision,
|
||||
check_raw_files_exist,
|
||||
@@ -166,11 +166,7 @@ def _apply_template_variables(template: str, variables: Dict[str, Any]) -> str:
|
||||
"""Apply variable substitutions to a template string.
|
||||
|
||||
Supports:
|
||||
- `{field}` - plain substitution from `variables[field]`, including the
|
||||
`{section.<slot>}` names this instance gave the three tool-owned
|
||||
headings (see chemenu.conventions). Those are what took the KB language
|
||||
out of `types/*.md`: a template writes `## {section.relationships}`, so
|
||||
scaffolding a page in another language needs no edit under `types/`
|
||||
- `{field}` - plain substitution from `variables[field]`
|
||||
- `{field|filter}` - apply a named filter (bullets, join, capitalize)
|
||||
to `variables[field]`'s value, so templates can render list/enum
|
||||
frontmatter fields directly instead of the caller precomputing a
|
||||
@@ -333,7 +329,6 @@ def new_page_command(
|
||||
**frontmatter,
|
||||
"name": name,
|
||||
"today": today.isoformat(),
|
||||
**conventions.section_variables(),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu import config, links
|
||||
from chemenu.commands._util import check_collision, fail, rel_path, success
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.page import Page
|
||||
@@ -33,7 +33,6 @@ from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.provenance import (
|
||||
CITE_REF_RE,
|
||||
cite_id,
|
||||
cite_block_heading,
|
||||
render_page_body,
|
||||
split_cite_block,
|
||||
unique_cite_id,
|
||||
@@ -103,7 +102,7 @@ def retarget_cite_ids(body: str, old: str, new: str) -> str:
|
||||
return body
|
||||
|
||||
new_head = CITE_REF_RE.sub(lambda m: f"[^{renames.get(m.group(1), m.group(1))}]", head)
|
||||
return render_page_body(new_head, new_definitions, cite_block_heading(body))
|
||||
return render_page_body(new_head, new_definitions)
|
||||
|
||||
|
||||
def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
|
||||
@@ -114,9 +113,11 @@ def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
|
||||
values = page.frontmatter.get(field)
|
||||
if not values:
|
||||
continue
|
||||
updated = [new if value == old else value for value in values]
|
||||
if updated != values:
|
||||
page.frontmatter[field] = updated
|
||||
# Through `links` so a labelled edge keeps its label across a rename:
|
||||
# the entry is `{label: target}`, and a plain equality swap would have
|
||||
# compared the mapping against a title and silently left it pointing at
|
||||
# the old page.
|
||||
if links.retarget(page.frontmatter, field, old, new):
|
||||
changed = True
|
||||
return changed
|
||||
|
||||
@@ -157,8 +158,10 @@ def strip_frontmatter_ref(page: Page, title: str) -> bool:
|
||||
values = page.frontmatter.get(field)
|
||||
if not values:
|
||||
continue
|
||||
updated = [value for value in values if value != title]
|
||||
if updated == values:
|
||||
before = list(values)
|
||||
links.remove(page.frontmatter, field, title)
|
||||
updated = page.frontmatter.get(field) or []
|
||||
if updated == before:
|
||||
continue
|
||||
if not updated and field not in declared:
|
||||
del page.frontmatter[field]
|
||||
|
||||
@@ -82,6 +82,10 @@ SKIP_COMMAND_PATHS = {
|
||||
("eval", "score"),
|
||||
("eval", "sessions"),
|
||||
("cite", "id"),
|
||||
# Retrieval, like `search`: an agent that has to ration looking up what
|
||||
# points at a page starts guessing instead - and under authored directional
|
||||
# edges this is the *only* way to ask that question.
|
||||
("links", "show"),
|
||||
("version", "show"),
|
||||
("version", "check"),
|
||||
("version", "notes"),
|
||||
|
||||
+116
-96
@@ -1,9 +1,22 @@
|
||||
"""Bidirectional cross-reference management between wiki pages.
|
||||
"""Cross-reference management between wiki pages.
|
||||
|
||||
`xref add` keeps two pages' frontmatter `related:` lists AND their body
|
||||
"## Relationships" sections in sync in one operation, instead of the 3-5
|
||||
separate manual edits this used to take per pair of pages. It is idempotent:
|
||||
re-running it never duplicates a link.
|
||||
`xref add` writes **one** edge: a label plus a target, into the asserting page's
|
||||
`related:` frontmatter, and re-renders that page's generated links region from
|
||||
it. It is idempotent, and re-running with a different label relabels rather than
|
||||
duplicating.
|
||||
|
||||
It used to write four things at once - `related:` and a Relationships bullet on
|
||||
both pages, plus reciprocal See Also bullets. That made every edge symmetric by
|
||||
construction, which is not what a link means: an edge is an authored reader aid,
|
||||
and "follow this to verify the premise" rarely reads the same from the other
|
||||
end. Worse, it is incompatible with per-collection label authorisation, because
|
||||
the mirrored half is written into a collection whose rules the author never
|
||||
read.
|
||||
|
||||
The reverse direction is therefore authored separately, when it is a primary
|
||||
statement of its own - and navigation does not depend on anyone bothering:
|
||||
`index rebuild` renders the inbound view from the graph, completely and without
|
||||
maintenance. See instructions/link-taxonomy.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -12,7 +25,7 @@ from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, sections
|
||||
from chemenu import blocks, config, conventions, kb_collections, links
|
||||
from chemenu.commands._util import fail, parse_list, success
|
||||
from chemenu.commands.page_ops import strip_frontmatter_ref
|
||||
from chemenu.frontmatter_io import write_page
|
||||
@@ -71,121 +84,119 @@ def _back_reference_field(source: Page, target: Page) -> str | None:
|
||||
return collection if collection in _declared_ref_fields(source) else None
|
||||
|
||||
|
||||
def add_related(frontmatter: dict, other_title: str) -> bool:
|
||||
"""Add other_title to frontmatter['related'] if not already present.
|
||||
Returns True if a change was made."""
|
||||
related = frontmatter.setdefault("related", [])
|
||||
if other_title in related:
|
||||
return False
|
||||
related.append(other_title)
|
||||
return True
|
||||
|
||||
|
||||
def _section_bounds(body: str, heading: str) -> tuple[int, int] | None:
|
||||
match = sections.heading_re(heading).search(body)
|
||||
if not match:
|
||||
def _collection_of(page: Page) -> str | None:
|
||||
"""The collection a page lives in, or None if it is outside `kb/`."""
|
||||
try:
|
||||
return page.path.relative_to(config.KB_DIR).parts[0]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
start = match.end()
|
||||
next_heading = re.search(r"^## ", body[start:], re.MULTILINE)
|
||||
end = start + next_heading.start() if next_heading else len(body)
|
||||
return start, end
|
||||
|
||||
|
||||
def add_bullet_to_section(body: str, heading: str, bullet: str, dedup_link: str) -> str:
|
||||
"""Insert `bullet` into the `## {heading}` section of body, unless a
|
||||
wikilink to dedup_link already appears there. Creates the section
|
||||
(before the See Also section if present, else at the end) if missing.
|
||||
def render_links_block(page: Page) -> str:
|
||||
"""The page's generated links region, built from its `related:` edges.
|
||||
|
||||
`heading` is a canonical name from `sections`; an existing section is found
|
||||
under its aliases too, so a page that has not been translated yet is still
|
||||
appended to rather than given a duplicate section. A section this creates
|
||||
always carries the canonical name."""
|
||||
bounds = _section_bounds(body, heading)
|
||||
if bounds is None:
|
||||
section = f"## {heading}\n\n{bullet}\n\n"
|
||||
see_also = sections.heading_re(sections.SEE_ALSO).search(body)
|
||||
if heading != sections.SEE_ALSO and see_also:
|
||||
return body[: see_also.start()] + section + body[see_also.start() :]
|
||||
return body.rstrip("\n") + "\n\n" + section.rstrip("\n") + "\n"
|
||||
|
||||
start, end = bounds
|
||||
section_text = body[start:end]
|
||||
if f"[[{dedup_link}]]" in section_text:
|
||||
return body
|
||||
trimmed = section_text.rstrip("\n")
|
||||
new_section = trimmed + "\n" + bullet + "\n\n"
|
||||
return body[:start] + new_section + body[end:]
|
||||
The body is a *rendering* of the frontmatter, not a second place the graph
|
||||
is stored. That is what removed the need to parse a German bullet back into
|
||||
a relationship: the label lives in the data, and this writes it out.
|
||||
"""
|
||||
lines = []
|
||||
for edge in links.edges(page.frontmatter, "related"):
|
||||
if edge.is_labelled:
|
||||
lines.append(f"- **{edge.label}:** [[{edge.target}]]")
|
||||
else:
|
||||
lines.append(f"- [[{edge.target}]]")
|
||||
return blocks.render(blocks.LINKS, conventions.heading(blocks.LINKS), lines)
|
||||
|
||||
|
||||
def add_relationship_bullet(body: str, label: str, other_title: str) -> str:
|
||||
bullet = f"- **{label}:** [[{other_title}]]"
|
||||
return add_bullet_to_section(body, sections.RELATIONSHIPS, bullet, other_title)
|
||||
def apply_links_block(page: Page, body: str | None = None) -> str:
|
||||
"""`body` with the links region re-rendered from `page.frontmatter`."""
|
||||
return blocks.replace(
|
||||
page.body if body is None else body, blocks.LINKS, render_links_block(page)
|
||||
)
|
||||
|
||||
|
||||
def add_see_also_bullet(body: str, other_title: str) -> str:
|
||||
return add_bullet_to_section(body, sections.SEE_ALSO, f"- [[{other_title}]]", other_title)
|
||||
def _check_authorised(source: Page, target: Page, label: str) -> None:
|
||||
"""Refuse a label the source collection has not authorised for that
|
||||
destination.
|
||||
|
||||
Checked here rather than only in `lint` because this is the moment the
|
||||
author is present: a refusal names the authorised set and can be answered by
|
||||
picking a better label, while a lint finding a day later is answered by
|
||||
whoever is holding the report.
|
||||
"""
|
||||
source_collection = _collection_of(source)
|
||||
destination = _collection_of(target)
|
||||
if source_collection is None or destination is None:
|
||||
return
|
||||
allowed = kb_collections.authorised_labels(source_collection, destination)
|
||||
if not allowed:
|
||||
fail(
|
||||
f"kb/{source_collection}/COLLECTION.md authorises no labels for edges into "
|
||||
f"kb/{destination}/. Add an `outbound:` entry for it, or do not link there "
|
||||
f"from this collection."
|
||||
)
|
||||
if label not in allowed:
|
||||
fail(
|
||||
f"'{label}' is not authorised for kb/{source_collection}/ -> kb/{destination}/.\n"
|
||||
f" Authorised: {', '.join(sorted(allowed))}\n"
|
||||
f" The catalogue and what each label asserts: instructions/link-taxonomy.md\n"
|
||||
f" Authorising a further label is a deliberate edit to "
|
||||
f"kb/{source_collection}/COLLECTION.md, not a way around this refusal."
|
||||
)
|
||||
|
||||
|
||||
@app.command("add")
|
||||
def xref_add(
|
||||
a: str = typer.Option(..., "--a", help="Exact title of page A"),
|
||||
b: str = typer.Option(..., "--b", help="Exact title of page B"),
|
||||
rel_a: str = typer.Option("related to", "--rel-a", help="Relationship label on A pointing to B"),
|
||||
rel_b: str = typer.Option("related to", "--rel-b", help="Relationship label on B pointing to A"),
|
||||
see_also: bool = typer.Option(True, "--see-also/--no-see-also", help="Also add reciprocal 'See Also' bullets"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes to both pages instead of writing"),
|
||||
a: str = typer.Option(..., "--a", help="Exact title of the page that asserts the edge"),
|
||||
b: str = typer.Option(..., "--b", help="Exact title of the page it points at"),
|
||||
rel: str = typer.Option(
|
||||
..., "--rel", help="Label from instructions/link-taxonomy.md, e.g. depends-on"
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview the change instead of writing"),
|
||||
):
|
||||
"""Declare that A <rel> B. One edge, on A only.
|
||||
|
||||
Say the sentence before choosing the label: `[A] <rel> [B]`. If it only
|
||||
reads true backwards, the edge belongs on B - run this the other way round
|
||||
rather than reaching for an inverse label.
|
||||
|
||||
B is not modified and does not need to point back. Its inbound view is
|
||||
rendered from the graph.
|
||||
"""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
page_a = _find_page(pages, a)
|
||||
page_b = _find_page(pages, b)
|
||||
|
||||
# Both refusals before either write, so a rejected pair leaves no half-link.
|
||||
# Every refusal before the single write, so a rejected edge leaves nothing.
|
||||
_require_related_field(page_a, a)
|
||||
_require_related_field(page_b, b)
|
||||
_check_authorised(page_a, page_b, rel)
|
||||
|
||||
related_changed_a = add_related(page_a.frontmatter, b)
|
||||
related_changed_b = add_related(page_b.frontmatter, a)
|
||||
|
||||
body_a = add_relationship_bullet(page_a.body, rel_a, b)
|
||||
body_b = add_relationship_bullet(page_b.body, rel_b, a)
|
||||
if see_also:
|
||||
body_a = add_see_also_bullet(body_a, b)
|
||||
body_b = add_see_also_bullet(body_b, a)
|
||||
|
||||
changed_a = related_changed_a or body_a != page_a.body
|
||||
changed_b = related_changed_b or body_b != page_b.body
|
||||
changed = links.upsert(page_a.frontmatter, "related", links.Edge(b, rel))
|
||||
body = apply_links_block(page_a)
|
||||
changed = changed or body != page_a.body
|
||||
|
||||
if dry_run:
|
||||
state_a = "would update" if changed_a else "already up to date"
|
||||
state_b = "would update" if changed_b else "already up to date"
|
||||
typer.echo(f"[dry-run] '{a}': {state_a} (related / Relationships / See Also)")
|
||||
typer.echo(f"[dry-run] '{b}': {state_b} (related / Relationships / See Also)")
|
||||
typer.echo(
|
||||
f"[dry-run] '{a}': {'would declare' if changed else 'already declares'} "
|
||||
f"{rel} -> '{b}'"
|
||||
)
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
try:
|
||||
write_page(page_a.path, page_a.frontmatter, body_a)
|
||||
except OSError as exc:
|
||||
fail(f"Failed to write '{a}': {exc}. '{b}' was not touched - fix the write failure and retry once.")
|
||||
if not changed:
|
||||
success(f"'{a}' already declares {rel} -> '{b}'; nothing changed.")
|
||||
return
|
||||
|
||||
try:
|
||||
write_page(page_b.path, page_b.frontmatter, body_b)
|
||||
write_page(page_a.path, page_a.frontmatter, body)
|
||||
except OSError as exc:
|
||||
fail(
|
||||
f"'{a}' was updated but writing '{b}' failed: {exc}. The link is now one-directional - "
|
||||
f"fix the write failure, then re-run `xref add --a \"{a}\" --b \"{b}\"` (idempotent, safe to retry)."
|
||||
)
|
||||
success(f"Linked '{a}' <-> '{b}' ({rel_a} / {rel_b})")
|
||||
fail(f"Failed to write '{a}': {exc}")
|
||||
success(f"'{a}' {rel} '{b}'")
|
||||
|
||||
|
||||
def remove_related(frontmatter: dict, other_title: str) -> bool:
|
||||
"""Drop other_title from frontmatter['related'] if present. Returns True if
|
||||
a change was made."""
|
||||
related = frontmatter.get("related")
|
||||
if not related or other_title not in related:
|
||||
return False
|
||||
frontmatter["related"] = [title for title in related if title != other_title]
|
||||
return True
|
||||
"""Drop every edge pointing at other_title. True if a change was made."""
|
||||
return links.remove(frontmatter, "related", other_title)
|
||||
|
||||
def remove_link_bullets(body: str, other_title: str) -> str:
|
||||
"""Remove the whole-line Relationships/See Also bullets `xref add` writes -
|
||||
@@ -223,14 +234,19 @@ def xref_remove(
|
||||
page_a = _find_page(pages, a)
|
||||
page_b = pages.get(b)
|
||||
|
||||
body_a = remove_link_bullets(page_a.body, b)
|
||||
changed_a = strip_frontmatter_ref(page_a, b) or body_a != page_a.body
|
||||
# The frontmatter first, then the region re-rendered from it - the body is a
|
||||
# rendering, so editing the bullet out directly would leave an empty region
|
||||
# behind and, worse, put the two out of step.
|
||||
changed_a = strip_frontmatter_ref(page_a, b)
|
||||
body_a = apply_links_block(page_a, remove_link_bullets(page_a.body, b))
|
||||
changed_a = changed_a or body_a != page_a.body
|
||||
|
||||
changed_b = False
|
||||
body_b = ""
|
||||
if page_b is not None:
|
||||
body_b = remove_link_bullets(page_b.body, a)
|
||||
changed_b = strip_frontmatter_ref(page_b, a) or body_b != page_b.body
|
||||
changed_b = strip_frontmatter_ref(page_b, a)
|
||||
body_b = apply_links_block(page_b, remove_link_bullets(page_b.body, a))
|
||||
changed_b = changed_b or body_b != page_b.body
|
||||
|
||||
if dry_run:
|
||||
typer.echo(f"[dry-run] '{a}': {'would update' if changed_a else 'no reference to remove'}")
|
||||
@@ -278,7 +294,11 @@ def xref_link_source(
|
||||
sources = page.frontmatter.setdefault("sources", [])
|
||||
if source not in sources:
|
||||
sources.append(source)
|
||||
body = add_see_also_bullet(page.body, source)
|
||||
# No body bullet. `sources:` *is* the record, and the See Also bullet
|
||||
# this used to add was the reciprocal half of a bidirectional model
|
||||
# that no longer exists - 353 of the corpus's 555 such bullets were
|
||||
# provably redundant with an edge that already said the same thing.
|
||||
body = page.body
|
||||
|
||||
# The way back. Until this existed the command wrote only the targets,
|
||||
# so a source page's own `entities:`/`concepts:` stayed as `new` left
|
||||
|
||||
Reference in new issue
Block a user