Chemenu 2.1.0 - deterministischer Wissenskompiler
CI / verify (push) Failing after 32s
Release / release (push) Successful in 38s

Chemenu kompiliert Rohnotizen zu einem verlinkten, quellengebundenen Wiki:
raw/ -> types/ + tools/ -> kb/ -> reports/. Was mechanisch ist, macht
tools/wikitool; was Urteil braucht, macht ein Agent unter Contracts, deren
Grenzen in Code durchgesetzt sind statt im Prompt.

Dieser Commit ist der Startpunkt der oeffentlichen Historie. Die vorherige
Entwicklung fand in einer privaten Instanz statt und ist nicht Teil dieses
Repositorys; ihre Erzaehlung steht vollstaendig in CHANGES.md, das mit 44
Eintraegen von 0.1.0 bis 2.1.0 erhalten geblieben ist.

Der mitgelieferte Korpus ist ein Testbett und eine Demo: 170 Seiten ueber den
Stack selbst - Gates, Lint, Versionierung, Suche, das Wiki-Muster. Er
dokumentiert das Werkzeug mit den eigenen Mitteln des Werkzeugs.

Lizenz: AGPL-3.0 fuer den Stack (tools/, types/), CC-BY-4.0 fuer die Inhalte.
Die Grenze zwischen beiden ist der Dateiplan, den dist export berechnet -
siehe NOTICE.
This commit is contained in:
torben committed 2026-09-01 16:26:14 +02:00
commit 18ae28f918
368 files changed
+50628

No files matched your search

View File
Whitespace-only changes.
+192
View File
@@ -0,0 +1,192 @@
"""Shared helpers for wikitool subcommands."""
from __future__ import annotations
import re
from datetime import date
from pathlib import Path
from typing import Any, Dict, Optional
import typer
from rich.console import Console
console = Console()
# A third outcome alongside success (0) and validation error (1): the command
# is refusing until a *human* has seen its output and cleared it. It exists as
# its own code so the caller - an agent, a harness hook, a CI job, a trajectory
# scorer - can tell "stop and ask the user" apart from "your input was wrong,
# fix it and retry". Nothing about *why* clearance is needed lives in the agent
# instructions: the command's own output carries the reason, the evidence, and
# the exact re-run line.
EXIT_NEEDS_CLEARANCE = 42
def success(msg: str) -> None:
console.print(f"[green]OK[/green] {msg}")
# Set by `fail()`, read once per process by the CLI entry point. Exit 1 raised
# through `fail()` means the command declined and did the thing it was asked
# for: the argument was rejected, or a read-only check reported findings.
# Neither is an iteration step on the wiki, so the Iteration Budget Gate gives
# the slot back (see run_budget.refund). A command that has already done its
# work and then reports a non-zero result - `lint --fail-on-error` writes its
# report first - raises `typer.Exit(1)` directly and stays counted.
_declined = False
def declined() -> bool:
"""Whether this process left through `fail()`."""
return _declined
def fail(msg: str) -> None:
global _declined
_declined = True
console.print(f"[bold red]ERROR[/bold red] {msg}")
raise typer.Exit(code=1)
def needs_clearance(msg: str) -> None:
"""Refuse with EXIT_NEEDS_CLEARANCE. The message is written to be shown to
a human verbatim - it is the whole user-facing artifact of this gate."""
console.print(f"[bold yellow]NEEDS USER CLEARANCE[/bold yellow] {msg}")
raise typer.Exit(code=EXIT_NEEDS_CLEARANCE)
# A comma preceded by a backslash is a literal comma, not a separator.
_UNESCAPED_COMMA = re.compile(r"(?<!\\),")
def parse_list(value: str | None) -> list[str]:
"""Split a comma-separated CLI value into list elements.
`\\,` is an escaped literal comma: it survives the split and lands inside
the element. Without it a list format simply cannot express an element
that contains a comma - and shell quoting is no help, because the quotes
are gone long before this sees the string. Paths and page titles carry
commas often enough for that to matter: it once cost a `raw/` file its
original name, which `raw/CONTRACT.md` forbids.
"""
if not value:
return []
parts = (part.replace("\\,", ",").strip() for part in _UNESCAPED_COMMA.split(value))
return [part for part in parts if part]
def coerce_set_value(raw_value: str, field_schema: Optional[Dict[str, Any]]) -> Any:
"""Coerce a `--set field=value` string to the type its schema declares.
Arrays are comma-split (see `parse_list` for the escape), numbers are
parsed as float/int, booleans as true/false; everything else stays a
string. Unknown fields (no schema entry) pass through as strings and are
then caught by schema validation's `additionalProperties: false`.
"""
declared = (field_schema or {}).get("type")
if declared == "array":
return parse_list(raw_value)
if declared == "number":
try:
return float(raw_value)
except ValueError:
return raw_value
if declared == "integer":
try:
return int(raw_value)
except ValueError:
return raw_value
if declared == "boolean":
if raw_value.lower() in ("true", "false"):
return raw_value.lower() == "true"
return raw_value
def parse_set_fields(
set_fields: Optional[list[str]], schema: Optional[Dict[str, Any]], flag: str = "--set"
) -> Dict[str, Any]:
"""Parse repeated `<flag> field=value` pairs into a frontmatter dict,
coercing each value by the field's declared schema type.
Repeating the flag for an *array* field appends rather than replaces, so
`--set raw_files=a --set raw_files=b` yields both. That is the form that
needs no separator at all, and therefore the one to reach for when an
element contains a comma; `\\,` inside a single value does the same job
for a one-liner. Repeating a scalar field still means "last one wins" -
there is nothing to append to.
Note that this is per *invocation*. What a parsed value then means for a
page already on disk is the caller's decision: `new` writes it as the
page's initial value, while `touch` replaces, extends or subtracts
depending on which flag it came from.
"""
explicit: Dict[str, Any] = {}
properties = (schema or {}).get("properties", {})
for pair in set_fields or []:
if "=" not in pair:
fail(f"{flag} expects field=value, got: {pair}")
field_name, raw_value = pair.split("=", 1)
field_name = field_name.strip()
if not field_name:
fail(f"{flag} expects field=value, got: {pair}")
value = coerce_set_value(raw_value, properties.get(field_name))
previous = explicit.get(field_name)
if isinstance(value, list) and isinstance(previous, list):
previous.extend(value)
else:
explicit[field_name] = value
return explicit
def check_raw_files_exist(raw_files: Any) -> None:
"""Verify every `raw_files:` entry is an existing file.
This is the one validation that genuinely cannot live in the schema:
it is filesystem I/O, not a data-shape constraint. Cardinality
(`minItems: 1`) is already enforced by the schema itself, so only
existence and file-vs-directory are checked here.
Shared by `new` and `touch` - both write the field, and a page pointing at
a raw file that is not there is the same defect whichever wrote it.
"""
from chemenu import config
for raw_path in raw_files or []:
full_path = config.ROOT / raw_path
if not full_path.exists():
fail(
f"raw_files path does not exist: {raw_path}\n"
" This is one element after splitting the value on commas. If the real "
"filename contains a comma, escape it as `\\,` or pass one `--set "
"raw_files=<path>` per file - never rename the raw file to fit the flag."
)
if full_path.is_dir():
fail(f"raw_files must be a file, not a directory: {raw_path}")
def today_iso() -> str:
return date.today().isoformat()
def rel_path(path: Path) -> str:
"""Format a path relative to the repo root for display, falling back to
the raw path if it lies outside the root (e.g. in tests)."""
from chemenu import config
try:
return str(Path(path).relative_to(config.ROOT))
except ValueError:
return str(path)
def check_collision(name: str) -> None:
"""Fail if any page under wiki/ already has `name` as its filename stem.
The stem *is* the page title and wikilinks resolve by title alone, so two
files sharing a stem in different directories are indistinguishable to
every link in the wiki. Shared by `new` and `rename`.
"""
from chemenu import config
for path in config.KB_DIR.rglob("*.md"):
if path.stem == name:
fail(f"A page titled '{name}' already exists at {rel_path(path)}")
+220
View File
@@ -0,0 +1,220 @@
"""`wikitool cite ...` - real GFM footnote citations.
A citation marker is `[^cite-id]` in a page's prose, resolved by a
`[^cite-id]: [[Source - X]]` (or `[[Source - X|file.md]]`) definition line in
the page's trailing Footnotes block (see chemenu.provenance for the
regexes and cite_id() derivation). AGENTS.md invariant 1 forbids hand-writing
generated structure, and a cite-id is exactly that - an author must never
compute or paste one by hand. `cite add` is the only way to get one onto a
page; `cite sync` is the only way to reconcile a page's block after prose
edits changed which ids are actually referenced.
None of this writes the inline `[^cite-id]` reference into prose: where a
citation belongs in a sentence is an editorial call, same as the prose itself
(see tools/CONTRACT.md's design notes). `cite add` prints the marker to paste
in; the LLM places it.
"""
from __future__ import annotations
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import fail, rel_path, success
from chemenu.frontmatter_io import write_page
from chemenu.page import Page
from chemenu.kb_scan import load_kb_pages
from chemenu.provenance import (
CITE_REF_RE,
cite_id,
cite_block_heading,
render_page_body,
split_cite_block,
unique_cite_id,
)
app = typer.Typer(help="Manage [^cite-id] footnote citations and their Footnotes definition blocks.")
def _find_page(pages: dict[str, Page], title: str) -> Page:
if title not in pages:
fail(f"No page titled '{title}' found under wiki/.")
return pages[title]
@app.command("id")
def cite_id_command(
title: str = typer.Option(..., "--title", help="Source page title, e.g. 'Source - Docker Cheatsheet'"),
file: Optional[str] = typer.Option(None, "--file", help="Qualifier for a multi-file source, e.g. 'storage-model.md'"),
):
"""Print the deterministic id cite_id() would derive for (--title, --file).
Read-only preview - does not check the id is actually free on any given
page (two pages, or two distinct pairs on one page, can share this base
id; `cite add`/`cite sync` are what apply the real -2/-3 suffixing).
"""
typer.echo(cite_id(title, file))
def upsert_citation(page: Page, source_title: str, qualifier: Optional[str]) -> tuple[str, str, bool]:
"""Ensure `page` has a Footnotes definition for (source_title, qualifier)
and that source_title is in its frontmatter `sources:`. Returns
(cite_id_to_use, new_body, changed) - reuses an existing definition for
the same pair instead of minting a duplicate id."""
head, definitions = split_cite_block(page.body)
existing_id = next(
(cid for cid, pair in definitions.items() if pair == (source_title, qualifier)),
None,
)
if existing_id is not None:
marker_id = existing_id
block_changed = False
else:
marker_id = unique_cite_id(set(definitions), source_title, qualifier)
definitions[marker_id] = (source_title, qualifier)
block_changed = True
sources = page.frontmatter.setdefault("sources", [])
sources_changed = source_title not in sources
if sources_changed:
sources.append(source_title)
new_body = render_page_body(head, definitions, cite_block_heading(page.body))
changed = block_changed or sources_changed or new_body != page.body
return marker_id, new_body, changed
@app.command("add")
def cite_add(
page_title: str = typer.Option(..., "--page", help="Exact title of the page to add a citation on"),
source: str = typer.Option(..., "--source", help="Exact title of the source page being cited, e.g. 'Source - X'"),
file: Optional[str] = typer.Option(None, "--file", help="Qualifier for a multi-file source, e.g. 'storage-model.md'"),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview instead of writing"),
):
"""Upsert a Footnotes definition for `--source` (reusing it if the page
already cites the same source/file pair) and ensure `--source` is in the
page's frontmatter `sources:`. Prints the `[^cite-id]` marker to paste
into the prose - placing it is still the caller's job.
"""
pages = load_kb_pages(config.KB_DIR)
page = _find_page(pages, page_title)
if source not in pages:
fail(f"No page titled '{source}' found under wiki/ - citing a page that doesn't exist would be a dangling reference.")
marker_id, new_body, changed = upsert_citation(page, source, file)
marker = f"[^{marker_id}]"
if dry_run:
state = "would update" if changed else "already up to date"
typer.echo(f"[dry-run] '{page_title}': {state}")
typer.echo(f"marker: {marker}")
typer.echo("No files written (--dry-run).")
return
if changed:
write_page(page.path, page.frontmatter, new_body)
typer.echo(f"marker: {marker}")
success(
f"{'Updated' if changed else 'Already up to date:'} '{page_title}' cites '{source}'"
+ (f" ({file})" if file else "")
+ f". Paste {marker} at the point in the prose the fact appears."
)
def sync_page(page: Page) -> tuple[str, bool, list[str], list[str]]:
"""Reconcile one page's Footnotes block against its actual `[^id]`
references: prune definitions nothing references any more, and re-render
the block in first-reference order. Never mints or recomputes an id from
a title - a reference with no definition is reported, not guessed at.
Returns (new_body, changed, pruned_ids, undefined_ref_ids).
"""
head, definitions = split_cite_block(page.body)
referenced_ids = [m.group(1) for m in CITE_REF_RE.finditer(head)]
referenced_set = set(referenced_ids)
if not definitions and not referenced_ids:
# No citation content at all - leave the page's whitespace exactly as
# it is. Without this, re-rendering an empty block still normalizes
# trailing newlines, which would make `cite sync --all` rewrite every
# page in the wiki instead of just the ones it actually has work to do.
return page.body, False, [], []
pruned = [cid for cid in definitions if cid not in referenced_set]
undefined = sorted({cid for cid in referenced_ids if cid not in definitions})
ordered: dict[str, tuple[str, Optional[str]]] = {}
seen: set[str] = set()
for cid in referenced_ids:
if cid in definitions and cid not in seen:
ordered[cid] = definitions[cid]
seen.add(cid)
new_body = render_page_body(head, ordered, cite_block_heading(page.body))
changed = new_body != page.body
return new_body, changed, pruned, undefined
@app.command("sync")
def cite_sync(
page_title: Optional[str] = typer.Option(None, "--page", help="Sync just this page"),
all_pages: bool = typer.Option(False, "--all", help="Sync every page under wiki/"),
dry_run: bool = typer.Option(False, "--dry-run", help="Report what would change instead of writing"),
):
"""Prune orphan Footnotes definitions and re-render each page's block in
first-reference order. Reports any `[^id]` reference left with no
definition - that is an editorial gap (a citation whose `cite add` never
ran, or a hand-typed id), not something this command can fix."""
if bool(page_title) == bool(all_pages):
fail("Provide exactly one of --page or --all")
pages = load_kb_pages(config.KB_DIR)
targets = [_find_page(pages, page_title)] if page_title else sorted(pages.values(), key=lambda p: p.path)
touched: list[str] = []
undefined_report: dict[str, list[str]] = {}
failed: list[str] = []
for page in targets:
new_body, changed, pruned, undefined = sync_page(page)
title = page.path.stem
if undefined:
undefined_report[title] = undefined
if not changed:
continue
touched.append(title)
if dry_run:
continue
try:
write_page(page.path, page.frontmatter, new_body)
except OSError as exc:
failed.append(f"{title} ({exc})")
if failed:
fail(
f"Synced {len(touched) - len(failed)}/{len(touched)} page(s) before a write failed: "
f"{', '.join(failed)}. Safe to retry - each page's re-render is idempotent."
)
verb = "Would update" if dry_run else "Updated"
if touched:
typer.echo(f"{verb} {len(touched)} page(s):")
for title in touched:
typer.echo(f" - {title}")
else:
typer.echo("No pages needed a Footnotes block change.")
if undefined_report:
typer.echo("")
typer.echo("Undefined [^id] reference(s) - run `cite add` for these, or fix the typo:")
for title, ids in undefined_report.items():
typer.echo(f" - {title}: {', '.join(ids)}")
if dry_run:
typer.echo("No files written (--dry-run).")
return
if not touched and not undefined_report:
success("Every Footnotes block already matches its page's references.")
+159
View File
@@ -0,0 +1,159 @@
"""Apply the confidence decay formula defined in the wiki contract's
"Confidence Scoring" section: confidence decays at 1% per month since last
confirmation (the page's `modified` / `date` / `created` field), floored at 0.2.
This is pure arithmetic - previously left to the LLM's judgment even though
the contract specifies it exactly. Dry-run by default; `--apply` writes changes.
`confidence` is a *derived* field: it is always recomputed as
`confidence_base * (1 - 0.01 * months)`, never from its own previous value.
Keeping the undecayed anchor in `confidence_base` is what makes repeated runs
idempotent - decaying the stored `confidence` in place (the pre-2026-08-13
behavior) compounded on every run, because the elapsed-months factor kept
growing while the multiplicand had already shrunk.
"""
from __future__ import annotations
import datetime
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import success
from chemenu.frontmatter_io import write_page
from chemenu.kb_scan import load_kb_pages
app = typer.Typer(help="Apply confidence decay per the wiki contract's Confidence Scoring formula.")
DECAY_RATE_PER_MONTH = 0.01
FLOOR = 0.2
DAYS_PER_MONTH = 30.44
def _parse_date(value) -> Optional[datetime.date]:
if isinstance(value, datetime.datetime):
return value.date()
if isinstance(value, datetime.date):
return value
if isinstance(value, str):
try:
return datetime.date.fromisoformat(value)
except ValueError:
return None
return None
def compute_decay(confidence: float, last_confirmed: datetime.date, today: datetime.date) -> float:
months = max(0.0, (today - last_confirmed).days / DAYS_PER_MONTH)
decayed = confidence * (1 - DECAY_RATE_PER_MONTH * months)
return round(max(FLOOR, decayed), 2)
def _last_confirmed(frontmatter: dict) -> Optional[datetime.date]:
return (
_parse_date(frontmatter.get("modified"))
or _parse_date(frontmatter.get("date"))
or _parse_date(frontmatter.get("created"))
)
@app.command("init-base")
def confidence_init_base(
apply: bool = typer.Option(False, "--apply", help="Write changes; default is dry-run (preview only)"),
):
"""Backfill `confidence_base` from the current `confidence` on pages that
don't have one yet.
Needed once, when a wiki predates the derived-`confidence` model. Pages
already carrying a base are left untouched, so this is safe to re-run.
"""
pages = load_kb_pages(config.KB_DIR)
changes = []
for title, page in sorted(pages.items()):
confidence = page.frontmatter.get("confidence")
if confidence is None or page.frontmatter.get("confidence_base") is not None:
continue
changes.append((title, page, float(confidence)))
if not changes:
success("Every page with a confidence already has a confidence_base.")
return
for title, page, base in changes:
typer.echo(f"{title}: confidence_base <- {base:.2f}")
if apply:
_set_after(page.frontmatter, "confidence", "confidence_base", round(base, 2))
write_page(page.path, page.frontmatter, page.body)
if apply:
success(f"Set confidence_base on {len(changes)} page(s).")
else:
typer.echo(f"\n{len(changes)} page(s) would change. Re-run with --apply to write.")
def _set_after(frontmatter: dict, after_key: str, key: str, value) -> None:
"""Insert `key` immediately after `after_key`, preserving frontmatter order
(write_page serializes in dict insertion order, and the schemas list
confidence_base right after confidence)."""
if key in frontmatter or after_key not in frontmatter:
frontmatter[key] = value
return
items = list(frontmatter.items())
frontmatter.clear()
for existing_key, existing_value in items:
frontmatter[existing_key] = existing_value
if existing_key == after_key:
frontmatter[key] = value
@app.command("decay")
def confidence_decay(
apply: bool = typer.Option(False, "--apply", help="Write changes; default is dry-run (preview only)"),
):
pages = load_kb_pages(config.KB_DIR)
today = datetime.date.today()
changes = []
missing_base = []
for title, page in sorted(pages.items()):
confidence = page.frontmatter.get("confidence")
if confidence is None:
continue
base = page.frontmatter.get("confidence_base")
if base is None:
missing_base.append(title)
continue
last_confirmed = _last_confirmed(page.frontmatter)
if last_confirmed is None:
continue
new_confidence = compute_decay(float(base), last_confirmed, today)
if abs(new_confidence - round(float(confidence), 2)) >= 0.01:
changes.append((title, page, float(confidence), new_confidence))
if missing_base:
typer.echo(
f"Skipped {len(missing_base)} page(s) with a confidence but no confidence_base "
"- run `wikitool confidence init-base --apply` first:"
)
for title in missing_base[:10]:
typer.echo(f" - {title}")
if len(missing_base) > 10:
typer.echo(f" ... and {len(missing_base) - 10} more")
typer.echo("")
if not changes:
success("No confidence values need decaying.")
return
for title, page, old, new in changes:
typer.echo(f"{title}: {old:.2f} -> {new:.2f}")
if apply:
page.frontmatter["confidence"] = new
write_page(page.path, page.frontmatter, page.body)
if apply:
success(f"Updated confidence on {len(changes)} page(s).")
else:
typer.echo(f"\n{len(changes)} page(s) would change. Re-run with --apply to write.")
+442
View File
@@ -0,0 +1,442 @@
"""`wikitool dist export` - build a distributable, contentless copy of this
repo's machinery.
`export` copies the pipeline's schema/compiler/control-plane layers (types/,
tools/, instructions/, the stage contracts, every kb/*/COLLECTION.md) into an
empty target, with no kb/ pages, no raw/ content, and no git history - see
instructions/setup-instance.md for what happens after. It never calls git.
Three independent exclusion mechanisms feed the plan, for three different
shapes of "does not belong in someone else's instance":
- Every copied text file passes through `strip_markers()`, which removes any
region between `<!-- dist:strip-start -->` and `<!-- dist:strip-end -->`,
markers included - for dev-only *content inside* a file that is otherwise
shipped (e.g. a routing line in AGENTS.md).
- `instructions/dev/` is pruned from the copy wholesale - for dev-only
*whole files* (procedures and the skill that switches an agent into
tool-development mode). One-way: nothing reconstructs it in a distributed
instance, on purpose - see instructions/dev/ itself for the current
contents and AGENTS.md's routing line for what a dev instance sees instead.
- Build output under `tools/` is dropped, by directory (`TOOLS_EXCLUDE_DIRS`)
where it has one, and by filename (`_is_coverage_output`) where it does not.
Not dev-only but *derived*: recomputable, and measured against this repo's
own test run rather than the receiving instance's.
"""
from __future__ import annotations
import hashlib
import json
import os
import re
import stat
from pathlib import Path
from typing import Callable, NamedTuple, Optional, Union
import typer
from chemenu import config, kb_collections, kb_state, version as version_mod
from chemenu.commands._util import fail, rel_path, success, today_iso
app = typer.Typer(help="Build a distributable copy of the wiki machinery.")
MARKER_START = "<!-- dist:strip-start -->"
MARKER_END = "<!-- dist:strip-end -->"
DIST_TEMPLATES_DIR = Path(__file__).resolve().parent.parent / "dist_templates"
# Root files copied verbatim (after marker-stripping). INSTALL.md is optional
# here: it does not exist until the distribution docs land, and `export`
# must not fail just because a later stage of the same repo hasn't shipped
# yet.
#
# The personalization *templates* ship; the filled `USER.md`/`SOUL.md` never
# do. This allowlist is what makes that split automatic - a file is copied
# because it is named here, so an instance's own personalization is excluded
# by construction rather than by a rule someone has to remember.
#
# `ENVIRONMENT.md.template` rides the same split for the same reason: a
# distribution can describe what the file is for, but never what a particular
# checkout's harness, MCP servers and remotes are. The filled `ENVIRONMENT.md`
# is additionally gitignored, so it is excluded twice over.
#
# `CLAUDE.md` is harness glue, not a second control plane: Claude Code loads it
# and does not load `AGENTS.md`, so it ships for the same reason
# `.claude/settings.json` does - a distributed instance running that harness
# would otherwise start every session without the control plane.
ROOT_FILES = (
"AGENTS.md", "CLAUDE.md", "README.md", "EVALS.md", "INSTALL.md", ".gitignore", "VERSION",
*config.LICENSE_FILES,
*config.PERSONALIZATION_TEMPLATES,
config.ENVIRONMENT_TEMPLATE,
)
# The one part of ROOT_FILES that may not be quietly skipped. Every other entry
# copies only `if source.is_file()`, which is right for `INSTALL.md` (it did not
# exist until the distribution docs landed) and wrong for a licence: an export
# that silently omits it hands the receiving instance the AGPL-covered `tools/`
# tree with no licence text, which is a violation the moment that instance is
# pushed anywhere public. Missing means the export is broken, not minimal.
REQUIRED_ROOT_FILES = config.LICENSE_FILES
# Harness-specific session-tracing config: generic machinery (feeds
# tools/chemenu/telemetry/ and tools/trace_ingest.py via EVALS.md), not
# personal state - unlike `.obsidian/`/`.vscode/`, which are never copied.
HOOK_DIRS = (".github/hooks", ".vibe")
# tools/ subpaths never copied - build/venv/cache artifacts, not machinery.
# `htmlcov/` is coverage.py's HTML report: derived output, and a large tree of
# it, measured against the source repo's own test run. `.coveragerc` beside it
# *does* ship, the same way `pytest.ini` does - it is configuration, not output.
TOOLS_EXCLUDE_DIRS = {".venv", "__pycache__", ".pytest_cache", ".wikitool_session", "htmlcov"}
# The rest of coverage's output lands beside the code rather than in a directory
# of its own - `.coverage`, `coverage.xml`, and `.coverage.<host>.<pid>` under a
# parallel run - so a directory exclusion cannot reach it. Same argument as
# `reports/`: derived, recomputable, and about the source repo rather than about
# the instance that would receive it.
COVERAGE_OUTPUT_NAMES = frozenset({".coverage", "coverage.xml"})
def _is_coverage_output(filename: str) -> bool:
return filename in COVERAGE_OUTPUT_NAMES or filename.startswith(".coverage.")
# instructions/dev/ holds stack-development-only procedures and the skill
# that switches an agent into tool-development mode - never shipped to a
# distributed instance. One-way: there is no `enable-dev`-style command that
# reconstructs it afterwards, unlike the marker-block content below.
INSTRUCTIONS_EXCLUDE_DIRS = {"dev"}
# Fixed by raw/CONTRACT.md's routing table, unlike kb/'s areas (which are
# organic - see kb/CONTRACT.md - so `export` does not manufacture them).
RAW_SUBDIRS = ("articles", "documents", "notes", "assets")
# Stage contracts that are not collections and carry no pages: copied as a
# single file each, nothing else from their directory.
CONTRACT_ONLY_STAGES = ("raw/CONTRACT.md", "reports/CONTRACT.md", "work/CONTRACT.md")
# Single tracked files copied out of an otherwise-untouched, partially-ignored
# directory. `.claude/` holds the harness's own session-tracing config
# (`settings.json`, tracked) alongside generated skill copies and personal
# untracked state (`.claude/skills/`, `.claude/settings.local.json`) - neither
# of which belongs in a distribution. Adding `.claude` to HOOK_DIRS would copy
# the whole directory, skills included; a single-file entry avoids that
# without needing an exclude set HOOK_DIRS doesn't otherwise carry.
SINGLE_FILES = (".claude/settings.json",)
Content = Union[str, bytes]
class PlannedFile(NamedTuple):
content: Content
executable: bool = False
_MARKER_TOKEN_RE = re.compile(re.escape(MARKER_START) + "|" + re.escape(MARKER_END))
# The leading/trailing `\n?` consume the blank line on each side of the
# block - the convention is that a marker block always sits as its own
# paragraph. Without eating both, a strip leaves two blank lines where the
# clean file only ever had one.
_MARKER_BLOCK_RE = re.compile(
r"\n?" + re.escape(MARKER_START) + r".*?" + re.escape(MARKER_END) + r"\n?", re.DOTALL
)
def _validate_markers(text: str, label: str) -> None:
"""A marker file must be a sequence of well-formed, non-nested
start/end pairs. Malformed markers would make `strip_markers` remove
either too little or too much, silently - this fails loudly instead."""
depth = 0
for match in _MARKER_TOKEN_RE.finditer(text):
if match.group() == MARKER_START:
if depth != 0:
fail(f"{label}: nested dist:strip-start markers are not supported")
depth = 1
else:
if depth != 1:
fail(f"{label}: dist:strip-end without a matching dist:strip-start")
depth = 0
if depth != 0:
fail(f"{label}: dist:strip-start without a matching dist:strip-end")
def strip_markers(text: str) -> str:
"""Remove every marked region, markers included. Generic by design: it
does not matter what is inside, or how many regions a file has."""
return _MARKER_BLOCK_RE.sub("", text)
def _is_executable(path: Path) -> bool:
return bool(path.stat().st_mode & stat.S_IXUSR)
def _read_planned_file(path: Path, label: str) -> PlannedFile:
executable = _is_executable(path)
try:
text = path.read_text(encoding="utf-8")
except UnicodeDecodeError:
return PlannedFile(path.read_bytes(), executable)
# Marker syntax is an HTML/Markdown comment convention, scoped to .md
# files on purpose: applying it to every text file would let the marker
# strings themselves - inline here as Python string literals - match as
# a region in this file's own source when tools/ gets copied, and eat
# the code between them.
if path.suffix != ".md":
return PlannedFile(text, executable)
_validate_markers(text, label)
return PlannedFile(strip_markers(text), executable)
def _copy_tree(
source_root: Path,
dest_prefix: str,
exclude_dirs: frozenset[str],
exclude_file: Optional[Callable[[str], bool]] = None,
) -> dict[str, PlannedFile]:
"""Every file under source_root, marker-stripped, keyed by its
destination-relative path. Excluded directories are pruned during the
walk rather than filtered after, so a large `.venv/` is never read.
`exclude_file` drops individual files by name, for output that lands
beside the code instead of in a directory a prune could catch."""
files: dict[str, PlannedFile] = {}
if not source_root.is_dir():
return files
for dirpath, dirnames, filenames in os.walk(source_root):
dirnames[:] = sorted(d for d in dirnames if d not in exclude_dirs)
for filename in sorted(filenames):
if exclude_file is not None and exclude_file(filename):
continue
path = Path(dirpath) / filename
relative = path.relative_to(source_root).as_posix()
dest_rel = f"{dest_prefix}/{relative}"
files[dest_rel] = _read_planned_file(path, dest_rel)
return files
def _digest(content: Content) -> str:
data = content if isinstance(content, bytes) else content.encode("utf-8")
return "sha256:" + hashlib.sha256(data).hexdigest()
def build_stamp(plan: dict[str, PlannedFile], origin: "Origin") -> str:
"""The release stamp written into every export.
Two jobs. The version and origin fields are what `version check` compares
against a release feed - without them an instance cannot tell which stack
it is running. The per-file digests are for the update *after* detection:
they record what the machinery looked like when it was installed, which is
the only way a later upgrade can tell a file the instance edited from one
it merely received. Nothing reads them today; writing them now is what
keeps that upgrade from needing a format change.
"""
stamp = {
"schema": version_mod.STAMP_SCHEMA,
"version": str(version_mod.read_version()),
"exported_at": today_iso(),
"source_repo": origin.source_repo,
"source_commit": origin.source_commit,
"release_url": origin.release_url,
"update_url": origin.update_url or version_mod.DEFAULT_UPDATE_URL,
"files": {relative: _digest(planned.content) for relative, planned in sorted(plan.items())},
}
return json.dumps(stamp, indent=2, sort_keys=False) + "\n"
class Origin(NamedTuple):
"""Where this export came from. Supplied by the caller (the release
workflow knows the commit and the release URL); `dist export` itself never
calls git, so it cannot discover any of it."""
source_repo: Optional[str] = None
source_commit: Optional[str] = None
release_url: Optional[str] = None
update_url: Optional[str] = None
def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
"""Every (destination-relative path -> planned file) the export writes."""
plan: dict[str, PlannedFile] = {}
missing_licences = [
name for name in REQUIRED_ROOT_FILES if not (config.ROOT / name).is_file()
]
if missing_licences:
fail(
"export would ship code without its licence: "
+ ", ".join(missing_licences)
+ " missing from the source tree. Restore them before exporting - a "
"distribution carrying tools/ without LICENSE is a copyleft violation "
"the moment the receiving instance is published."
)
for name in ROOT_FILES:
source = config.ROOT / name
if source.is_file():
plan[name] = _read_planned_file(source, name)
plan.update(_copy_tree(config.INSTRUCTIONS_DIR, "instructions", frozenset(INSTRUCTIONS_EXCLUDE_DIRS)))
plan.update(_copy_tree(config.TYPES_DIR, "types", frozenset()))
plan.update(_copy_tree(
config.ROOT / "tools", "tools", frozenset(TOOLS_EXCLUDE_DIRS), _is_coverage_output
))
for hook_dir in HOOK_DIRS:
plan.update(_copy_tree(config.ROOT / hook_dir, hook_dir, frozenset()))
kb_contract = config.KB_DIR / "CONTRACT.md"
if kb_contract.is_file():
plan["kb/CONTRACT.md"] = _read_planned_file(kb_contract, "kb/CONTRACT.md")
for collection in kb_collections.iter_kb_collections():
rel = f"kb/{collection.name}/COLLECTION.md"
plan[rel] = _read_planned_file(collection / "COLLECTION.md", rel)
for relative in CONTRACT_ONLY_STAGES:
source = config.ROOT / relative
if source.is_file():
plan[relative] = _read_planned_file(source, relative)
for relative in SINGLE_FILES:
source = config.ROOT / relative
if source.is_file():
plan[relative] = _read_planned_file(source, relative)
for sub in RAW_SUBDIRS:
plan[f"raw/{sub}/.gitkeep"] = PlannedFile("")
plan["kb/log.md"] = PlannedFile((DIST_TEMPLATES_DIR / "log.md").read_text(encoding="utf-8"))
plan["CHANGES.md"] = PlannedFile((DIST_TEMPLATES_DIR / "CHANGES.md").read_text(encoding="utf-8"))
# A fresh instance's content is empty, so it is trivially in the shape this
# machinery expects - which is exactly what makes the initial declaration
# safe to write here rather than leaving it to `migrate baseline`. Only an
# instance predating this file has to answer that question by hand.
plan[kb_state.KB_STATE_FILENAME] = PlannedFile(
kb_state.render_kb_state(version_mod.read_version(), [])
)
# Last, so it can digest everything above it. It is the one file in the
# export that describes the export rather than being copied into it.
plan[version_mod.RELEASE_STAMP_FILENAME] = PlannedFile(
build_stamp(plan, origin or Origin())
)
return plan
# Content that must never appear in a plan, expressed structurally rather than
# by matching text. Three allowlists feed `build_plan`, and each one holds only
# because someone remembered the rule when they edited it - nothing re-checks
# the result. This does.
#
# The checks are deliberately structural: a filled personalization file, a kb
# page, a raw source, a dev-only instruction. A text-pattern scan (hostnames,
# IP literals) was considered and rejected - the project's own host legitimately
# appears in INSTALL.md and version.py, so such a scan would either whitelist
# the very string it is looking for or cry wolf on every export.
_CONTENT_PREFIXES = ("kb/", "raw/")
_CONTENT_ALLOWED_NAMES = ("CONTRACT.md", "COLLECTION.md", "log.md", ".gitkeep")
def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
"""Planned paths that carry one instance's own data instead of machinery."""
leaks: list[str] = []
for relative in sorted(plan):
name = relative.rsplit("/", 1)[-1]
if name in config.PERSONALIZATION_FILES or name == config.ENVIRONMENT_FILE:
leaks.append(f"{relative} (one instance's own personalization)")
elif relative.startswith("instructions/dev/"):
leaks.append(f"{relative} (stack-development only)")
elif relative.startswith(_CONTENT_PREFIXES) and name not in _CONTENT_ALLOWED_NAMES:
leaks.append(f"{relative} (wiki content, not machinery)")
return leaks
def _write_plan(target: Path, plan: dict[str, PlannedFile]) -> None:
for relative, planned in plan.items():
dest = target / relative
dest.parent.mkdir(parents=True, exist_ok=True)
if isinstance(planned.content, bytes):
dest.write_bytes(planned.content)
else:
dest.write_text(planned.content, encoding="utf-8")
if planned.executable:
dest.chmod(dest.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
@app.command("export")
def export_command(
target: Path = typer.Argument(
..., help="Directory to write the distribution into. Must not exist, or must be empty."
),
dry_run: bool = typer.Option(
False, "--dry-run", help="List what would be written, without writing anything."
),
source_repo: Optional[str] = typer.Option(
None, "--source-repo", help="Repository this export was built from (recorded in the stamp)"
),
source_commit: Optional[str] = typer.Option(
None, "--source-commit", help="Commit this export was built from (recorded in the stamp)"
),
release_url: Optional[str] = typer.Option(
None, "--release-url", help="Release page this export ships as (recorded in the stamp)"
),
update_url: Optional[str] = typer.Option(
None, "--update-url", help="Release feed `version check` should ask (recorded in the stamp)"
),
):
"""Export a contentless, distributable copy of this repo's machinery:
AGENTS.md/README.md (dev-instance-only marker blocks removed),
instructions/ (no instructions/dev/), types/, tools/ (no venv/caches),
the .github/hooks/+.vibe session-tracing config plus .claude/settings.json,
every kb/*/COLLECTION.md (no pages, no areas), empty
raw/{articles,documents,notes,assets}/, VERSION, the USER.md/SOUL.md
personalization templates (never the filled files), and a
.wikitool-release.json stamp. The --source-*/--release-url/--update-url
options only fill fields in that stamp: `export` never calls git and cannot
discover them. See instructions/setup-instance.md for what comes next."""
run_export(
target,
dry_run=dry_run,
origin=Origin(
source_repo=source_repo,
source_commit=source_commit,
release_url=release_url,
update_url=update_url,
),
)
def run_export(target: Path, dry_run: bool = False, origin: Optional[Origin] = None) -> None:
"""The export itself, free of Typer's option objects so it can be called
directly - by the command above, and by the tests."""
target = target.resolve()
if target.exists():
if not target.is_dir():
fail(f"{target} exists and is not a directory.")
if any(target.iterdir()):
fail(f"{target} is not empty. `dist export` refuses to write into a non-empty directory.")
try:
plan = build_plan(origin)
except version_mod.VersionError as exc:
fail(f"{exc} - a distribution must carry the version it ships.")
return
leaks = find_leaks(plan)
if leaks:
fail(
"export would carry this instance's own data, not just machinery:\n "
+ "\n ".join(leaks)
+ "\nThis is an allowlist bug in dist_cmd.py, not something to work "
"around - fix the allowlist rather than deleting files from the target."
)
return
if dry_run:
for relative in sorted(plan):
typer.echo(f"write {relative}")
success(f"Dry run: would write {len(plan)} file(s) to {target}. Nothing written.")
return
_write_plan(target, plan)
success(f"Exported {len(plan)} file(s) to {rel_path(target)}.")
+499
View File
@@ -0,0 +1,499 @@
"""`wikitool docs verify` - machine-check the documentation copies that can be
re-derived from the code and the repo layout.
The wiki's own rule is that a derived copy of recomputable truth must be
checked or absent. Three such copies survive on purpose because they earn
their keep as reading material:
1. `tools/CONTRACT.md`'s command table (re-derivable from the Typer app)
2. the collection and stage contracts (their existence and placement, not
their content)
3. the absence of pre-type-system `type: entity` frontmatter in the
contract docs - the exact drift that left a stale comparison template
sitting in AGENTS.md for months after the type migration
A fourth check has a different shape: `.gitignore` is not documentation, but
it is the one file that can silently un-publish content. A pattern excluding a
file under `raw/` or `kb/` is a data-loss bug - `sources coverage` reads the
filesystem and reports the file as covered, while `publish` (`git add -A`)
never commits it, so a fresh clone has a broken `raw_files:` reference. The
same check runs in reverse over `reports/`, where a *missing* ignore rule would
start committing derived output.
A fifth has the same shape as the fourth: `VERSION` is not documentation
either, but it is the one number a release stamps into every distributed
instance, and a version raised without a changelog entry ships release notes
that describe the previous release.
Everything here is a hard oracle: a set comparison or a regex, no judgment.
Content quality of the contracts themselves stays with the LLM.
"""
from __future__ import annotations
import re
import subprocess
from pathlib import Path
from typing import Optional
import typer
from chemenu import config, kb_collections, version as version_mod
from chemenu.commands._util import fail, rel_path, success
app = typer.Typer(help="Verify documentation that mirrors the code or repo layout.")
# Contracts that are not COLLECTION.md files, because their directories are not
# collections. Each is the authoring contract for one stage or layer.
STAGE_CONTRACTS = (
"raw/CONTRACT.md",
"kb/CONTRACT.md",
"types/type-spec.md",
"reports/CONTRACT.md",
"work/CONTRACT.md",
"tools/CONTRACT.md",
"instructions/CONTRACT.md",
)
# Directories whose contents are the repository's reason to exist, and which
# therefore may never be excluded by an ignore rule. `work/` is here because a
# workshop is the only record of a multi-session run: unlike `reports/`, losing
# it loses judgment that nothing can recompute.
CONTENT_DIRS = ("raw", "kb", "work")
# Paths that must never be ignored. They deliberately do not have to exist:
# `git check-ignore --no-index` answers about the *pattern set*, not the
# filesystem, so these catch a trap before a real file ever falls into it.
# Every entry corresponds to a pattern that was genuinely swallowing content
# before the 2026-08-13 `.gitignore` rewrite.
IGNORE_CANARIES = (
"raw/notes/template.md", # was caught by `*temp*`
"raw/notes/temperature-sensors.md", # was caught by `*temp*`
"raw/notes/scratch.md", # was caught by `*scratch*`
"raw/assets/build.log", # was caught by `*.log`
"raw/documents/go.mod", # was caught by `go.mod`
"raw/assets/bin/tool.txt", # was caught by `bin/`
"raw/assets/diagram.orig", # was caught by `*.orig`
"kb/concepts/Template Method.md", # was caught by `*temp*`
"kb/entities/tools/core.md", # was caught by `core`
"kb/entities/tools/tags.md", # was caught by `tags`
# `work/` is tracked on purpose: unlike `reports/`, a workshop holds
# judgment in progress that nothing can recompute, so an ignore rule
# reaching it would silently discard a multi-session run's only record.
"work/ingest-documents-example/extract-00-architecture.md",
)
# The mirror image of IGNORE_CANARIES. `reports/` holds derived output that must
# stay *out* of git, so an ignore rule going missing there is as much a bug as an
# ignore rule appearing over content - it would start committing a second,
# drifting copy of something `wikitool lint` recomputes on demand. The contract
# is the one file that must survive the rule.
#
# The skill directories are here for a different reason: they are copies of
# `instructions/<name>/SKILL.md`, published by `wikitool instructions sync`.
# Committing them would create exactly the drifting second copy this repo
# refuses to keep anywhere else.
#
# `ENVIRONMENT.md` is a third reason again: it is per-checkout, so committing
# one working copy's harness, MCP servers and remotes would hand every other
# clone a file that is confidently wrong rather than honestly absent. Its
# `.template` sits in REQUIRED_TRACKED_PATHS below, because the obvious
# careless pattern (`ENVIRONMENT.md*`) would swallow both.
#
# The coverage paths are the reports/ argument applied to `pytest --cov`
# output: derived, recomputable, and in the way of `publish`'s `git add -A`.
REQUIRED_IGNORE_CANARIES = (
"reports/Lint Report 2026-01-01.md",
".agents/skills/wiki-query/SKILL.md",
".claude/skills/wiki-query/SKILL.md",
"ENVIRONMENT.md",
"tools/coverage.xml",
"tools/htmlcov/index.html",
)
REQUIRED_TRACKED_PATHS = (
"reports/CONTRACT.md",
"instructions/CONTRACT.md",
"instructions/wiki-query/SKILL.md",
"ENVIRONMENT.md.template",
)
CLI_README = config.ROOT / "tools" / "CONTRACT.md"
# The root README is the "absent" half of the checked-or-absent rule: it used to
# carry its own copy of the command table, which drifted because nothing
# compared it to anything. It now points at tools/CONTRACT.md instead, and this
# check keeps it that way.
ROOT_README = config.ROOT / "README.md"
# `README.md` is for humans, `CONTRACT.md` is the agent-facing contract, and a
# stage may carry both. The split only holds while the README stays prose: the
# first thing that drifted last time was a second copy of the command table, and
# tools/README.md is exactly the file it drifted in. INSTALL.md is here for the
# same reason: it is human-facing prose about installing an instance, and the
# command reference lives exactly once, in tools/CONTRACT.md.
STAGE_READMES = ("tools/README.md", "INSTALL.md")
# Docs that must not re-introduce the pre-migration bare-enum `type:` form.
# The per-collection contracts are appended at call time, since which ones exist
# is a filesystem question rather than a constant.
TYPE_GUARD_DOCS = ("AGENTS.md", "README.md", *STAGE_CONTRACTS)
LEGACY_TYPE_RE = re.compile(r"^type:\s*(entity|concept|source|comparison)\s*$", re.MULTILINE)
# First backticked cell of a markdown table row, e.g. "| `xref add --a ...` | ... |"
TABLE_CELL_RE = re.compile(r"^\|\s*`([^`]+)`", re.MULTILINE)
def registered_commands() -> set[str]:
"""Every command path the CLI exposes, e.g. {'new', 'xref add', ...}.
Imported lazily: `chemenu.cli` imports this module, so a top-level
import would be circular.
"""
from chemenu import cli
paths: set[str] = set()
for command in cli.app.registered_commands:
name = command.name or (command.callback.__name__.replace("_", "-") if command.callback else None)
if name:
paths.add(name)
for group in cli.app.registered_groups:
group_name = group.name
sub_app = group.typer_instance
if not group_name or sub_app is None:
continue
for command in sub_app.registered_commands:
name = command.name or (command.callback.__name__.replace("_", "-") if command.callback else None)
if name:
paths.add(f"{group_name} {name}")
return paths
def top_level_names() -> set[str]:
return {path.split(" ", 1)[0] for path in registered_commands()}
def documented_commands(readme_text: str) -> list[str]:
return [match.group(1).strip() for match in TABLE_CELL_RE.finditer(readme_text)]
def check_cli_readme() -> list[str]:
"""Every registered command must appear in tools/CONTRACT.md's command
table, and every command documented there must exist.
The reverse check matches a documented cell against the full registered
command path (e.g. `xref add`, `confidence init-base`), not just its first
token - checking only the top-level word would let a typo'd or invented
subcommand (`xref frobnicate`) sit undetected next to a real command group
(`xref`) forever.
"""
if not CLI_README.exists():
return [f"{CLI_README.relative_to(config.ROOT)} is missing"]
text = CLI_README.read_text(encoding="utf-8")
cells = documented_commands(text)
issues = []
registered = sorted(registered_commands())
for command_path in registered:
if not any(cell == command_path or cell.startswith(command_path + " ") for cell in cells):
issues.append(f"command `{command_path}` is not documented in tools/CONTRACT.md")
for cell in cells:
if not any(cell == cp or cell.startswith(cp + " ") for cp in registered):
first_token = cell.split(" ", 1)[0]
issues.append(f"tools/CONTRACT.md documents `{cell}`, but `{first_token}` is not a wikitool command")
return issues
def check_collection_contracts() -> list[str]:
"""The three structural rules that define what a collection is.
Collections are discovered by contract presence rather than listed here, so
`mkdir kb/<name>` + a COLLECTION.md is all it takes to add one. That only
works if the inverse is also checked: a directory under kb/ *without* a
contract is an unclaimed subtree whose pages obey no local rules, and a
contract outside kb/ quietly widens "collection" back out to "any directory".
"""
issues = []
collections = {path.name for path in kb_collections.iter_kb_collections()}
if config.KB_DIR.is_dir():
for child in sorted(config.KB_DIR.iterdir()):
if child.is_dir() and child.name not in collections:
issues.append(
f"kb/{child.name}/ has no COLLECTION.md - every directory under kb/ is a "
"collection and needs its own authoring contract"
)
for stray in kb_collections.stray_collection_contracts():
relative = stray.relative_to(config.ROOT)
if kb_collections.kb_collection_of(stray.parent) is not None:
issues.append(
f"{relative} is nested inside a collection - a subdirectory is an area and "
"inherits the enclosing contract"
)
else:
issues.append(
f"{relative} is outside kb/ - only kb/ holds collections; other directories "
"carry a CONTRACT.md instead"
)
for relative_path in STAGE_CONTRACTS:
if not (config.ROOT / relative_path).exists():
issues.append(f"{relative_path} is missing - it is the authoring contract for its stage")
return issues
def check_legacy_type_blocks() -> list[str]:
issues = []
guarded = [
*TYPE_GUARD_DOCS,
*(
str((path / "COLLECTION.md").relative_to(config.ROOT))
for path in kb_collections.iter_kb_collections()
),
]
for relative_path in guarded:
path = config.ROOT / relative_path
if not path.exists():
continue
for match in LEGACY_TYPE_RE.finditer(path.read_text(encoding="utf-8")):
line_number = path.read_text(encoding="utf-8")[: match.start()].count("\n") + 1
issues.append(
f"{relative_path}:{line_number} uses the pre-migration `type: {match.group(1)}` form "
f"- pages reference types by path (`types/{match.group(1)}.md`)"
)
return issues
def command_table_free_readmes() -> list[Path]:
"""Every README that must not carry a copy of the command table.
Built at call time rather than at import, so a test can point ROOT_README at
a fixture.
"""
return [ROOT_README, *(config.ROOT / relative for relative in STAGE_READMES)]
def check_readmes_have_no_command_table() -> list[str]:
"""No README may re-list wikitool commands in a table.
A derived copy of recomputable truth is either checked or absent. The
command table is checked in tools/CONTRACT.md, so a second copy in a README
has to be absent - otherwise it drifts silently, which is exactly what it
did.
"""
known_top_level = top_level_names()
issues = []
for readme in command_table_free_readmes():
if not readme.exists():
continue
offenders = sorted(
{
cell
for cell in documented_commands(readme.read_text(encoding="utf-8"))
if cell.split(" ", 1)[0] in known_top_level
}
)
issues += [
f"{rel_path(readme)} has a table row for `{cell}` - the command reference lives in "
"tools/CONTRACT.md, which `docs verify` checks; link to it instead of copying it"
for cell in offenders
]
return issues
def _git(args: list[str], stdin: Optional[str] = None) -> Optional[subprocess.CompletedProcess]:
"""Run a git command in the repo root, or return None if git is unavailable
or this is not a checkout. Returning None (rather than raising) keeps
`docs verify` usable in a source tree without git, where the ignore rules
are unknowable rather than wrong."""
try:
return subprocess.run(
["git", *args], cwd=config.ROOT, capture_output=True, text=True, input=stdin
)
except OSError:
return None
def _check_ignore(paths: tuple[str, ...]) -> Optional[list[str]]:
"""The subset of `paths` the repo's ignore rules would exclude, or None if
git cannot answer.
`--no-index` makes this a pure question about the pattern set: it does not
matter whether the path exists or is tracked, only whether a rule would
swallow it. That is what turns a latent trap into a failing check.
None and `[]` have to stay distinguishable. For the forward canaries an
unknowable answer and an empty answer both mean "no finding", but the
reverse canaries assert that a path *is* ignored - so collapsing None into
`[]` would turn a missing git binary into a fabricated failure.
"""
result = _git(["check-ignore", "--no-index", "-z", "--stdin"], stdin="\0".join(paths))
if result is None or result.returncode not in (0, 1):
return None
return [path for path in result.stdout.split("\0") if path]
def ignored_canaries(canaries: tuple[str, ...] = IGNORE_CANARIES) -> list[str]:
"""The subset of `canaries` the ignore rules would exclude; empty if
unknowable."""
return _check_ignore(canaries) or []
def ignored_content_files() -> list[str]:
"""Files that actually exist under a CONTENT_DIRS directory but are ignored,
and so would never be committed by `wikitool publish`."""
result = _git(
["ls-files", "--others", "--ignored", "--exclude-standard", "-z", "--", *CONTENT_DIRS]
)
if result is None or result.returncode != 0:
return []
return [path for path in result.stdout.split("\0") if path]
def check_ignored_content() -> list[str]:
"""No file under `raw/`, `kb/` or `work/` may be excluded by an ignore rule,
and everything under `reports/` except its README must be."""
issues = [
f"`{path}` exists but is gitignored - `wikitool publish` will never commit it"
for path in ignored_content_files()
]
issues += [
f"an ignore rule would swallow `{path}` - anchor the pattern in .gitignore "
"(see its header note) so content cannot be silently un-published"
for path in ignored_canaries()
]
still_ignored = _check_ignore(REQUIRED_IGNORE_CANARIES)
if still_ignored is not None:
issues += [
f"`{path}` is NOT ignored - generated reports must stay out of git, or they become "
"a second copy of what `wikitool lint` recomputes on demand"
for path in REQUIRED_IGNORE_CANARIES
if path not in still_ignored
]
wrongly_ignored = _check_ignore(REQUIRED_TRACKED_PATHS)
if wrongly_ignored is not None:
issues += [
f"`{path}` is ignored - it must survive the reports/ ignore rule"
for path in REQUIRED_TRACKED_PATHS
if path in wrongly_ignored
]
return issues
def check_version_changelog() -> list[str]:
"""`VERSION` must parse, and the newest versioned `CHANGES.md` entry must
name it.
This is the check that makes `version bump` more than a convenience: a
version raised with nothing written about it would ship a release whose
notes describe the previous one. A changelog with *no* versioned entry at
all is fine - that is a fresh distribution, and this repo's own pre-
versioning history, neither of which claims to describe the current
version.
"""
version_path = config.ROOT / version_mod.VERSION_FILENAME
if not version_path.is_file():
return [
f"{version_mod.VERSION_FILENAME} is missing - the stack has no version for "
"`dist export` to stamp or `version check` to compare"
]
try:
declared = version_mod.Version.parse(version_path.read_text(encoding="utf-8"))
except version_mod.VersionError as exc:
return [f"{version_mod.VERSION_FILENAME}: {exc}"]
changes_path = config.ROOT / version_mod.CHANGES_FILENAME
if not changes_path.is_file():
return [f"{version_mod.CHANGES_FILENAME} is missing - a version has nowhere to be explained"]
documented = version_mod.top_changes_version(changes_path.read_text(encoding="utf-8"))
if documented is not None and documented != declared:
return [
f"{version_mod.VERSION_FILENAME} says {declared}, but the newest versioned "
f"{version_mod.CHANGES_FILENAME} entry is {documented} - run "
"`wikitool version bump` (which writes both), or fix whichever is wrong"
]
return []
def _second_changes_version(text: str) -> Optional["version_mod.Version"]:
"""The version named by the second-newest versioned entry, or None."""
seen = [
version_mod.Version.parse(match.group(1))
for match in version_mod._CHANGES_ENTRY_RE.finditer(text)
]
return seen[1] if len(seen) > 1 else None
def check_migration_for_boundary() -> list[str]:
"""A version that crosses the compatibility boundary must say how to cross it.
`version check` tells an instance that it must migrate. Without this, that
is where the trail ends - the instance knows it is behind and nothing tells
it what to do. So a boundary-crossing version needs either a migration
document targeting it, or an explicit statement in its changelog entry that
no content has to change.
Only the newest entry is checked. Older boundaries were either satisfied
when they were written or cannot be fixed retroactively, and re-reporting
them forever would make the check noise.
"""
from chemenu import kb_state
changes_path = config.ROOT / version_mod.CHANGES_FILENAME
version_path = config.ROOT / version_mod.VERSION_FILENAME
if not changes_path.is_file() or not version_path.is_file():
return [] # already reported by check_version_changelog
text = changes_path.read_text(encoding="utf-8")
current = version_mod.top_changes_version(text)
previous = _second_changes_version(text)
if current is None or previous is None:
return [] # the first versioned entry has no predecessor to cross from
if current.compat_key == previous.compat_key:
return []
if version_mod.MIGRATION_NONE_MARKER in (version_mod.changes_section(text, current) or ""):
return []
if any(m.target == current for m in kb_state.load_migrations()):
return []
return [
f"{current} crosses the compatibility boundary from {previous}, so every existing "
f"instance must migrate - but no document under "
f"{rel_path(kb_state.migrations_dir())}/ targets it, and its {version_mod.CHANGES_FILENAME} "
f"entry does not carry `{version_mod.MIGRATION_NONE_MARKER}`. Write the migration "
"(instructions/migrate-corpus.md), or record why none is needed"
]
@app.command("verify")
def verify():
"""Check the CLI/README command tables, contract presence, type-form drift, ignore rules, and version/changelog agreement."""
issues = (
check_cli_readme()
+ check_readmes_have_no_command_table()
+ check_collection_contracts()
+ check_legacy_type_blocks()
+ check_ignored_content()
+ check_version_changelog()
+ check_migration_for_boundary()
)
if issues:
fail("Documentation issues found:\n" + "\n".join(f"- {i}" for i in issues))
success(
f"Docs verified: {len(registered_commands())} command(s) documented, "
f"{len(kb_collections.iter_kb_collections())} collection(s) and "
f"{len(STAGE_CONTRACTS)} stage contract(s) present, no legacy type blocks, "
f"{len(IGNORE_CANARIES)} ignore canaries clear, "
f"{version_mod.CHANGES_FILENAME} documents version "
f"{(config.ROOT / version_mod.VERSION_FILENAME).read_text(encoding='utf-8').strip()}."
)
+387
View File
@@ -0,0 +1,387 @@
"""`wikitool doctor` - one deterministic health check for a wiki instance.
Read-only, never writes. Exists to back `instructions/setup-instance.md` (and
any other instance-setup procedure) with a single command instead of ten
individual checks spelled out in prose - the same reasoning that keeps
mechanical work in code everywhere else in this repo. Each check reports
`OK`, `WARN`, or `FAIL` plus, on anything but `OK`, the command to fix it.
Only a `FAIL` makes the overall exit code non-zero: a fresh instance with no
remote yet, or no `WIKITOOL_SESSION_ID` set, is a valid state, not a fault.
"""
from __future__ import annotations
import json as _json
import shutil
import subprocess
import sys
from dataclasses import dataclass
from typing import Optional
import typer
from rich.console import Console
from chemenu import config, kb_collections, version as version_mod
from chemenu.commands import instructions_cmd
from chemenu.commands._util import rel_path
from chemenu.session import ENV_VAR as SESSION_ENV_VAR
console = Console()
@dataclass
class Check:
name: str
status: str # "OK" | "WARN" | "FAIL"
detail: str
fix: Optional[str] = None
def _git(args: list[str]) -> Optional[subprocess.CompletedProcess]:
try:
return subprocess.run(
["git", *args], cwd=config.ROOT, capture_output=True, text=True, timeout=5
)
except (OSError, subprocess.SubprocessError):
return None
def check_python() -> Check:
version = sys.version_info
if version < (3, 11):
return Check(
"python", "FAIL", f"Python {version.major}.{version.minor} found, need >= 3.11",
"Install Python 3.11+ and recreate tools/.venv",
)
return Check("python", "OK", f"Python {version.major}.{version.minor}.{version.micro}")
def check_ripgrep() -> Check:
if shutil.which("rg"):
return Check("ripgrep", "OK", "rg found on PATH")
return Check(
"ripgrep", "FAIL", "rg not found on PATH - `search` and `sources coverage` need it",
"Install ripgrep (e.g. `apt install ripgrep` / `brew install ripgrep`)",
)
def check_author() -> Check:
author = config.default_author()
if author is None:
return Check(
"author", "FAIL", "Neither $WIKI_AUTHOR nor `git config user.name` resolves",
"Run `git config user.name \"<Your Name>\"`, or export WIKI_AUTHOR",
)
import os
source = "WIKI_AUTHOR" if os.environ.get("WIKI_AUTHOR", "").strip() else "git config user.name"
return Check("author", "OK", f"'{author}' (from {source})")
def check_git_repo() -> list[Check]:
checks: list[Check] = []
inside = _git(["rev-parse", "--is-inside-work-tree"])
if inside is None or inside.returncode != 0 or inside.stdout.strip() != "true":
checks.append(
Check(
"git-repo", "FAIL", "Not inside a git working tree",
"Run `git init -b main`",
)
)
return checks
checks.append(Check("git-repo", "OK", "Inside a git working tree"))
name = _git(["config", "user.name"])
email = _git(["config", "user.email"])
if not name or not name.stdout.strip():
checks.append(
Check("git-identity", "FAIL", "`git config user.name` is not set",
"Run `git config user.name \"<Your Name>\"`")
)
elif not email or not email.stdout.strip():
checks.append(
Check("git-identity", "FAIL", "`git config user.email` is not set",
"Run `git config user.email \"<you@example.com>\"`")
)
else:
checks.append(Check("git-identity", "OK", f"{name.stdout.strip()} <{email.stdout.strip()}>"))
branch = _git(["rev-parse", "--abbrev-ref", "HEAD"])
branch_name = branch.stdout.strip() if branch and branch.returncode == 0 else ""
if not branch_name or branch_name == "HEAD":
checks.append(
Check("git-branch", "WARN", "No commit yet, or detached HEAD",
"Make the first commit via `publish` once ready")
)
else:
checks.append(Check("git-branch", "OK", f"On branch '{branch_name}'"))
remote = _git(["remote", "get-url", "origin"])
if remote and remote.returncode == 0 and remote.stdout.strip():
checks.append(Check("git-remote", "OK", remote.stdout.strip()))
else:
checks.append(
Check(
"git-remote", "WARN", "No 'origin' remote configured",
"A local-only instance is valid - `git remote add origin <url>` if you want one. "
"Every `publish` needs --no-push until then",
)
)
return checks
def check_skills() -> Check:
sources = instructions_cmd.skill_dirs()
if not sources:
return Check("skills", "FAIL", "No skills found under instructions/", None)
target_dirs = instructions_cmd.target_dirs()
missing = 0
drifted: list[str] = []
for target_root in target_dirs:
for source in sources:
difference = instructions_cmd.drift(source, target_root / source.name)
if difference == "missing":
missing += 1
elif difference:
drifted.append(f"{rel_path(target_root / source.name)}: {difference}")
expected = len(sources) * len(target_dirs)
if missing == expected and not drifted:
return Check(
"skills", "FAIL", "No skills published yet",
"Run `tools/wikitool instructions sync`",
)
if drifted:
return Check(
"skills", "FAIL", f"{len(drifted)} published copy/copies drifted from source",
"Run `tools/wikitool instructions sync`",
)
return Check("skills", "OK", f"{expected} published copy/copies match their source")
def check_structure() -> Check:
missing = []
for relative_path in (
"kb/CONTRACT.md", "raw/CONTRACT.md", "reports/CONTRACT.md",
"work/CONTRACT.md", "instructions/CONTRACT.md", "types/type-spec.md",
):
if not (config.ROOT / relative_path).exists():
missing.append(relative_path)
collections = kb_collections.iter_kb_collections()
if not collections:
missing.append("kb/*/COLLECTION.md")
if missing:
return Check(
"structure", "FAIL", f"Missing: {', '.join(missing)}",
"Re-run `dist export`, or restore the missing contract(s) from the source repo",
)
return Check(
"structure", "OK", f"{len(collections)} collection(s), all stage contracts present"
)
def check_personalization() -> Check:
"""Whether this instance knows who it works for, and how it sounds.
`USER.md` and `SOUL.md` are read every session, so an instance without
them runs a generic agent against a wiki built for one person - which is
a fault, not a preference, hence `FAIL` rather than `WARN`. They are also
the one pair of required files a distribution cannot ship filled: their
content is personal, so `dist export` carries the templates and the
Personalization step of `setup-instance.md` writes the real ones. That
makes a still-templated file the second failure mode worth naming
separately - it looks present and answers nothing.
"""
missing: list[str] = []
unfilled: list[str] = []
for name in config.PERSONALIZATION_FILES:
path = config.ROOT / name
if not path.is_file():
missing.append(name)
elif config.TEMPLATE_SENTINEL in path.read_text(encoding="utf-8"):
unfilled.append(name)
fix = (
"Run the Personalization step of instructions/setup-instance.md - it interviews you "
f"along {' and '.join(config.PERSONALIZATION_TEMPLATES)} and writes your answers verbatim"
)
if missing:
return Check("personalization", "FAIL", f"Missing: {', '.join(missing)}", fix)
if unfilled:
return Check(
"personalization", "FAIL",
f"Still the unfilled template: {', '.join(unfilled)}", fix,
)
return Check("personalization", "OK", f"{', '.join(config.PERSONALIZATION_FILES)} present and filled")
def check_environment() -> Check:
"""Whether this checkout records the environment it works through.
`ENVIRONMENT.md` names the harness, the published skills, the MCP
servers, the connectors and the git remotes this working copy actually
uses. Missing it costs a session some questions, not correctness, so this
check never FAILs - the whole point of the file is that it is optional,
and a FAIL would make it mandatory by the back door.
The one thing worth reporting is the failure mode the personalization
check already knows: a template renamed but not filled in. That file is
present, is loaded into every session, and answers nothing - worse than
absence, because absence is honest.
"""
path = config.ROOT / config.ENVIRONMENT_FILE
if not path.is_file():
return Check(
"environment", "OK", f"{config.ENVIRONMENT_FILE} absent (optional)",
)
if config.TEMPLATE_SENTINEL in path.read_text(encoding="utf-8"):
return Check(
"environment", "WARN",
f"{config.ENVIRONMENT_FILE} is still the unfilled template",
f"Fill it in along {config.ENVIRONMENT_TEMPLATE}'s sections and drop the "
f"`{config.TEMPLATE_SENTINEL}` line, or delete the file - it is optional",
)
return Check("environment", "OK", f"{config.ENVIRONMENT_FILE} present and filled")
def check_generated_files() -> Check:
missing = [
rel_path(path)
for path in (config.INDEX_FILE, config.LOG_FILE, config.PROVENANCE_FILE)
if not path.exists()
]
if missing:
return Check(
"generated-files", "FAIL", f"Missing: {', '.join(missing)}",
"Run `index rebuild` and `sources rebuild-index`",
)
return Check("generated-files", "OK", "kb/index.md, kb/log.md, kb/provenance.md present")
def check_session_id() -> Check:
import os
if os.environ.get(SESSION_ENV_VAR, "").strip():
return Check("session-id", "OK", f"{SESSION_ENV_VAR}={os.environ[SESSION_ENV_VAR]}")
return Check(
"session-id", "WARN", f"{SESSION_ENV_VAR} is not set - budget falls back to the parent PID",
"See instructions/session-setup.md",
)
def check_stack_version() -> Check:
"""Which stack this instance runs, and where it came from.
A missing `VERSION` is a WARN, not a FAIL: instances exported before the
stack was versioned are still perfectly functional - they just cannot
answer `version check`. A malformed one is a FAIL, because then something
edited a generated fact by hand and every comparison built on it is wrong.
"""
try:
current = version_mod.read_version()
except version_mod.VersionError as exc:
if not version_mod.version_file().is_file():
return Check(
"stack-version", "WARN", "No VERSION file - this instance predates stack versioning",
"Re-export from a current origin, or write the version this instance corresponds to",
)
return Check("stack-version", "FAIL", str(exc), f"Fix {version_mod.VERSION_FILENAME} by hand - it holds one semantic version, nothing else")
try:
stamp = version_mod.read_stamp()
except version_mod.VersionError as exc:
return Check(
"stack-version", "FAIL", str(exc),
f"Delete {version_mod.RELEASE_STAMP_FILENAME} or restore it from the release it came from",
)
origin = "development tree" if stamp is None else f"distribution, exported {stamp.get('exported_at', 'unknown')}"
return Check("stack-version", "OK", f"{current} ({origin})")
def check_kb_version() -> Check:
"""Whether the content is in the shape this machinery expects.
A `WARN` when the content lags: that is the normal, transient state in the
middle of an upgrade, not a fault - and `migrate status` names the chain
that closes it. A missing declaration is also a `WARN` (an instance from
before the file existed still works), an unreadable one a `FAIL`.
"""
from chemenu import kb_state
try:
stack = version_mod.read_version()
except version_mod.VersionError:
return Check(
"kb-version", "WARN", "No stack version to compare the content against",
"See the stack-version check above",
)
try:
kb_version = kb_state.read_kb_version()
except version_mod.VersionError as exc:
return Check(
"kb-version", "FAIL", str(exc),
f"Restore or delete {kb_state.KB_STATE_FILENAME}, then "
"`tools/wikitool migrate baseline <version>`",
)
if kb_version is None:
return Check(
"kb-version", "WARN",
f"{kb_state.KB_STATE_FILENAME} is missing - the content's shape is undeclared",
f"Run `tools/wikitool migrate baseline {stack}` if this instance's content has "
"never lagged behind its machinery",
)
if kb_version < stack:
pending = kb_state.chain(kb_state.load_migrations(), kb_version, stack)
if pending:
return Check(
"kb-version", "WARN",
f"Content is at {kb_version}, machinery at {stack} - "
f"{len(pending)} migration(s) outstanding",
"Run `tools/wikitool migrate status`",
)
return Check("kb-version", "OK", f"{kb_version} (nothing outstanding up to {stack})")
return Check("kb-version", "OK", f"{kb_version}")
def run_doctor() -> list[Check]:
checks: list[Check] = [
check_python(),
check_ripgrep(),
check_author(),
check_stack_version(),
check_kb_version(),
*check_git_repo(),
check_skills(),
check_structure(),
check_personalization(),
check_environment(),
check_generated_files(),
check_session_id(),
]
return checks
def doctor_command(
json_out: bool = typer.Option(False, "--json", help="Print the checks as JSON"),
):
"""Check that this instance is correctly configured: dependencies, author,
git identity/remote, published skills, structure, personalization,
generated files, and session scoping. Read-only. Exits 1 only if a check
FAILs."""
checks = run_doctor()
if json_out:
typer.echo(_json.dumps([c.__dict__ for c in checks], indent=2))
else:
for check in checks:
color = {"OK": "green", "WARN": "yellow", "FAIL": "bold red"}[check.status]
line = f"[{color}]{check.status}[/{color}] {check.name}: {check.detail}"
if check.fix and check.status != "OK":
line += f"\n fix: {check.fix}"
typer.echo(line) if False else None
from rich.console import Console
Console().print(line)
if any(check.status == "FAIL" for check in checks):
raise typer.Exit(code=1)
+96
View File
@@ -0,0 +1,96 @@
"""`wikitool eval` - score what a session did against what it left behind.
Read-only over `kb/`: the command runs lint's checks in-process and reads a
trace. It writes only into `reports/evals/`, which is gitignored like the rest of
that stage.
"""
from __future__ import annotations
import json
from datetime import date
from pathlib import Path
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import fail, rel_path, success
from chemenu.evals import scorecard
from chemenu.session import session_id as current_session_id
from chemenu.telemetry import reader
from chemenu.telemetry.writer import trace_root
app = typer.Typer(help="Score a traced session (see EVALS.md).")
EVALS_DIR = config.REPORTS_DIR / "evals"
@app.command("sessions")
def sessions_command(
json_out: bool = typer.Option(False, "--json", help="Print the list as JSON"),
):
"""List sessions that have a trace, most recent first."""
found = reader.sessions()
if json_out:
typer.echo(json.dumps(found, indent=2))
return
if not found:
success(f"No traces yet under {rel_path(trace_root())}.")
return
for name in found:
records = reader.read_trace(name)
first = records[0]["ts"][:19] if records else "-"
typer.echo(f"{name:40} {len(records):5d} event(s) since {first}")
@app.command("score")
def score_command(
session: Optional[str] = typer.Option(
None, "--session",
help="Session to score. Defaults to this shell's session, the same id the "
"budget gate uses.",
),
json_out: bool = typer.Option(False, "--json", help="Print the scorecard as JSON"),
markdown_out: Optional[Path] = typer.Option(
None, "--markdown",
help="Write a markdown scorecard here, conventionally under reports/evals/.",
),
save: bool = typer.Option(
False, "--save",
help="Write the scorecard to reports/evals/<date>/<session>.{json,md}.",
),
fail_on_error: bool = typer.Option(
False, "--fail-on-error",
help="Exit non-zero when the tree has hard errors or an invariant was violated.",
),
):
"""Score one session: structural state (L1) plus trajectory rules (L2)."""
target = session or current_session_id()
records = reader.read_trace(target)
if not records:
fail(
f"No trace for session '{target}'. `wikitool eval sessions` lists the ones "
"that exist; a session records nothing when WIKI_TRACE=0."
)
card = scorecard.score(target, records)
if save:
directory = EVALS_DIR / date.today().isoformat()
directory.mkdir(parents=True, exist_ok=True)
stem = target.replace("/", "__")
(directory / f"{stem}.json").write_text(json.dumps(card, indent=2), encoding="utf-8")
(directory / f"{stem}.md").write_text(
scorecard.render_markdown(card) + "\n", encoding="utf-8"
)
success(f"Wrote {rel_path(directory / stem)}.json/.md")
if markdown_out:
markdown_out.write_text(scorecard.render_markdown(card) + "\n", encoding="utf-8")
success(f"Wrote {rel_path(markdown_out)}")
if json_out:
typer.echo(json.dumps(card, indent=2))
if not json_out and not markdown_out and not save:
typer.echo(scorecard.render_markdown(card))
if fail_on_error and scorecard.failed(card):
raise typer.Exit(code=1)
File diff suppressed because it is too large. Load diff
+325
View File
@@ -0,0 +1,325 @@
"""Deterministically regenerate the wiki's catalog from every page's frontmatter.
This replaces manual statistics counting and manual sorted-row insertion, which
was a repeated source of errors (miscounts, wrong alphabetical position) when
done by hand.
The catalog is **sharded**, not one file. `kb/index.md` is a map: statistics,
one row per collection and per area, and a link to the shard that lists those
pages. The tables themselves live in a generated `INDEX.md` inside each
collection, and an area that grows past `SHARD_THRESHOLD` rows gets its own.
Why: a single flat catalog has to be read in full to answer any question about
it, so its cost grows with the wiki while the answer being looked for does not.
At a few hundred pages that is tens of thousands of tokens spent to learn three
filenames. The wiki's own `Index Scaling` page sets the threshold used here.
The map stays small enough to browse; `wikitool search` answers everything else.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from datetime import date
from pathlib import Path
import typer
from chemenu import config
from chemenu.commands._util import rel_path, success
from chemenu.kb_collections import iter_kb_collections
from chemenu.page import Page
from chemenu.kb_scan import GENERATED_INDEX, load_kb_pages
from chemenu.type_resolver import resolver
app = typer.Typer(help="Manage the generated wiki catalog (kb/index.md + per-collection INDEX.md).")
TABLE_HEADER = "| Page | Type | Summary | Last Modified |"
TABLE_SEP = "|------|------|---------|----------------|"
SUMMARY_HEADINGS = ("Description", "Definition", "Summary")
# Rows per area before it is split into its own shard. From the wiki's own
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
# number so growth is handled by arithmetic rather than by a judgment call.
SHARD_THRESHOLD = 50
# Display title for pages sitting directly in a collection root rather than in
# an area subdirectory.
UNGROUPED_TITLE = "All"
DO_NOT_EDIT = "<!-- Generated by `wikitool index rebuild`. Do not hand-edit. -->"
def _summary(page: Page) -> str:
fm_summary = page.frontmatter.get("summary")
if fm_summary:
return str(fm_summary).strip()
for heading in SUMMARY_HEADINGS:
match = re.search(rf"^## {heading}\s*\n+(.+)", page.body, re.MULTILINE)
if match:
line = match.group(1).strip().splitlines()[0].strip()
if line and not line.upper().startswith("TODO"):
return line[:117] + "..." if len(line) > 120 else line
return "TODO: add summary"
def _last_modified(page: Page) -> str:
for key in ("modified", "date", "created"):
value = page.frontmatter.get(key)
if value:
return str(value)
return date.fromtimestamp(page.path.stat().st_mtime).isoformat()
def _type_label(page: Page) -> str:
return page.subtype or page.kind or "unknown"
def _table(pages: list[Page]) -> list[str]:
lines = [TABLE_HEADER, TABLE_SEP]
for page in sorted(pages, key=lambda p: p.title.lower()):
lines.append(
f"| [[{page.title}]] | {_type_label(page)} | {_summary(page)} | {_last_modified(page)} |"
)
return lines
def _anchor(title: str) -> str:
"""GitHub-style heading anchor, so the map can deep-link into a shard."""
slug = re.sub(r"[^a-z0-9\s-]", "", title.lower())
return re.sub(r"\s+", "-", slug.strip())
@dataclass
class Area:
"""One grouping inside a collection: a subdirectory, or the collection root
for pages that sit directly in it."""
name: str
title: str
pages: list[Page] = field(default_factory=list)
own_shard: bool = False
@property
def count(self) -> int:
return len(self.pages)
@dataclass
class Collection:
name: str
areas: list[Area] = field(default_factory=list)
@property
def count(self) -> int:
return sum(area.count for area in self.areas)
def _area_titles() -> dict[str, str]:
"""Display titles for entity areas, taken from the entity type-spec's own
`layout:` rather than a hardcoded map - so a new subtype names its own
section by adding a type-spec, with no code change."""
layout = resolver.get_layout(resolver.find_type_by_name("entity")) or {}
return {spec.get("dir", key): spec.get("title", key.title()) for key, spec in layout.items()}
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
"""Group pages by their physical location: collection directory, then area
subdirectory.
Location rather than `kind` because a shard lives in the directory it
describes, and the two agree by construction: a type-spec's `base_dir:` is
what put the page there.
"""
titles = _area_titles()
grouped: dict[str, dict[str, Area]] = {}
# Seed from the collections that exist on disk, not only from the ones that
# happen to hold pages: an empty collection is a real (if unfilled) part of
# the wiki, and dropping it from the map would hide it from every reader.
for collection_dir in iter_kb_collections(kb_dir):
grouped.setdefault(collection_dir.name, {})
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
try:
parts = page.path.relative_to(kb_dir).parts
except ValueError: # pragma: no cover - pages always live under kb_dir
continue
if len(parts) < 2:
collection_name, area_name = "(kb root)", ""
else:
collection_name = parts[0]
area_name = parts[1] if len(parts) > 2 else ""
areas = grouped.setdefault(collection_name, {})
area = areas.get(area_name)
if area is None:
title = titles.get(area_name, area_name.title()) if area_name else UNGROUPED_TITLE
area = Area(name=area_name, title=title)
areas[area_name] = area
area.pages.append(page)
collections = []
for name in sorted(grouped):
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
for area in ordered:
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
collections.append(Collection(name=name, areas=ordered))
return collections
def build_area_shard(area: Area) -> str:
lines = [DO_NOT_EDIT, "", f"# {area.title}", "", f"{area.count} page(s).", ""]
lines.extend(_table(area.pages))
lines.append("")
return "\n".join(lines) + "\n"
def build_collection_shard(collection: Collection) -> str:
lines = [DO_NOT_EDIT, "", f"# kb/{collection.name}/ - Index", ""]
lines.append(f"{collection.count} page(s). Regenerated by `wikitool index rebuild`.")
lines.append("")
for area in collection.areas:
lines.append(f"## {area.title}")
lines.append("")
if area.own_shard:
lines.append(
f"{area.count} page(s) - listed in "
f"[{area.name}/{GENERATED_INDEX}]({area.name}/{GENERATED_INDEX})."
)
else:
lines.extend(_table(area.pages))
lines.append("")
return "\n".join(lines) + "\n"
def build_index_map(collections: list[Collection]) -> str:
"""The root catalog: counts and pointers, no page rows.
Deliberately carries no summaries. A summary is what makes a hit worth
opening, and that judgment belongs where the hit is produced - `search` and
the shards - not in a file every reader pays for in full.
"""
totals = {c.name: c.count for c in collections}
total = sum(totals.values())
lines = [
DO_NOT_EDIT,
"",
"# Wiki Index",
"",
"A map of the wiki, not a catalog of it: counts and pointers only.",
"",
"To *find* a page, search instead of reading this file:",
"",
'- `tools/wikitool search "<text>"` - ranked text search, with summaries',
"- `tools/wikitool search --field entity_type=system --field 'confidence<0.6'`"
" - structured query over frontmatter",
"",
"The page tables live in a generated `INDEX.md` inside each collection, linked below.",
"",
"## Statistics",
"",
f"- **Total Pages:** {total}",
]
for name in sorted(totals):
lines.append(f"- **{name.title()}:** {totals[name]}")
lines.append(f"- **Last Updated:** {date.today().isoformat()}")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## Collections")
lines.append("")
lines.append("| Collection | Pages | Index |")
lines.append("|------------|------:|-------|")
for collection in collections:
target = f"{collection.name}/{GENERATED_INDEX}"
lines.append(f"| `{collection.name}/` | {collection.count} | [{target}]({target}) |")
lines.append("")
for collection in collections:
listed = [area for area in collection.areas if area.name]
if not listed:
continue
lines.append(f"### {collection.name}/")
lines.append("")
lines.append("| Area | Pages | Index |")
lines.append("|------|------:|-------|")
for area in listed:
if area.own_shard:
target = f"{collection.name}/{area.name}/{GENERATED_INDEX}"
else:
target = f"{collection.name}/{GENERATED_INDEX}#{_anchor(area.title)}"
lines.append(f"| {area.title} | {area.count} | [{target}]({target}) |")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## Notes")
lines.append("")
lines.append(
"This map and every `INDEX.md` under `kb/` are generated by "
"`wikitool index rebuild`. Do not hand-edit them."
)
lines.append("")
lines.append("To add a new page, run `wikitool new ...`, then `wikitool index rebuild`.")
return "\n".join(lines) + "\n"
def plan_index(kb_dir: Path) -> dict[Path, str]:
"""Every file the catalog consists of, as {path: content}.
Returning the whole plan instead of writing as it goes is what makes the
stale-shard sweep possible: anything named `INDEX.md` that is not in the
plan is a leftover from a collection or area that no longer exists.
"""
collections = group_pages(kb_dir, load_kb_pages(kb_dir))
plan: dict[Path, str] = {kb_dir / "index.md": build_index_map(collections)}
for collection in collections:
collection_dir = kb_dir / collection.name
if not collection_dir.is_dir():
continue
plan[collection_dir / GENERATED_INDEX] = build_collection_shard(collection)
for area in collection.areas:
if area.own_shard:
plan[collection_dir / area.name / GENERATED_INDEX] = build_area_shard(area)
return plan
def stale_shards(kb_dir: Path, plan: dict[Path, str]) -> list[Path]:
"""Generated shards on disk that the current plan does not produce."""
return sorted(p for p in kb_dir.rglob(GENERATED_INDEX) if p not in plan)
def build_index(kb_dir: Path) -> str:
"""The root map. Kept as a named function because callers (and tests) ask
for "the index" meaning the entry point, not the whole plan."""
return build_index_map(group_pages(kb_dir, load_kb_pages(kb_dir)))
@app.command("rebuild")
def index_rebuild(
dry_run: bool = typer.Option(
False, "--dry-run", help="Print what would be written instead of writing it"
),
):
plan = plan_index(config.KB_DIR)
stale = stale_shards(config.KB_DIR, plan)
if dry_run:
for path in sorted(plan):
typer.echo(f"--- {rel_path(path)}")
# nl=False: the content already ends in a newline.
typer.echo(plan[path], nl=False)
for path in stale:
typer.echo(f"--- would remove stale shard: {rel_path(path)}")
return
for path, content in plan.items():
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
for path in stale:
path.unlink()
shards = len(plan) - 1
removed = f", removed {len(stale)} stale" if stale else ""
success(f"Rebuilt {rel_path(config.INDEX_FILE)} and {shards} shard(s){removed}")
+500
View File
@@ -0,0 +1,500 @@
"""`wikitool instructions sync|verify|list` - the instruction layer.
`instructions/` is the single source for everything an agent is told to do. It
holds two forms, told apart structurally rather than by any flag:
- `instructions/<name>.md` - an instruction, reached by link or on request
- `instructions/<name>/SKILL.md` - a skill, published into the harness directories
Publication is by **copy**, into `.agents/skills/` (read natively by GitHub
Copilot, Codex CLI and Mistral Vibe) and `.claude/skills/` (Claude Code reads
nothing else). This reverses an earlier design that used relative symlinks. The
symlink argument was that a link cannot go stale; the counter-arguments that
won are that symlinks are unreliable on Windows checkouts and do not survive
being archived or copied, and that this repo is meant to stay reproducible
elsewhere. The price of a copy is drift, so `verify` checks every copy byte for
byte against its source - the copy is derived truth, and derived truth is either
checked or absent.
Both target directories are gitignored. A fresh clone has no skills until `sync`
runs; `instructions/bootstrap.md` is the procedure, and `verify` says so rather
than reporting an error when *every* copy is missing, because that is the
expected state of a clean checkout rather than a fault.
"""
from __future__ import annotations
import filecmp
import shutil
from pathlib import Path
import typer
import yaml
from chemenu import config
from chemenu.commands import dist_cmd
from chemenu.commands._util import fail, rel_path, success
from chemenu.type_resolver import resolver
app = typer.Typer(help="Manage instructions/ and publish its skills into the harness directories.")
SKILL_FILE = "SKILL.md"
CONTRACT_FILE = "CONTRACT.md"
INSTRUCTION_TYPE = "types/instruction.md"
# instructions/dev/ is a second, purpose-scoped root nested one level in:
# stack-development-only instructions and (nested one level further) the
# skill that switches a session into that mode. `dist export` prunes it
# wholesale (see dist_cmd.INSTRUCTIONS_EXCLUDE_DIRS) - discovery below treats
# it as a second location to scan, not as ordinary recursion.
DEV_SUBDIR = "dev"
# Root files an agent actually loads, and from which a link therefore *reaches*
# it. CLAUDE.md sits alongside AGENTS.md rather than being folded into one name:
# AGENTS.md is read natively by every other harness (Codex CLI, GitHub Copilot
# CLI, Mistral Vibe), while CLAUDE.md is read only by Claude Code, which does not
# load AGENTS.md on its own (see CLAUDE.md itself, and AGENTS.md's file-naming
# table). A link that belongs in only one harness's auto-loaded file still has to
# count - see automatic_load_paths() below for the matching half of this split.
AGENT_ROOT_FILES = ("AGENTS.md", "CLAUDE.md")
# README.md and CHANGES.md are deliberately NOT above. They describe the stack
# to humans: AGENTS.md's file-naming table defines README.md as "never by an
# agent as instruction", and CHANGES.md is a changelog. An instruction whose only
# mention is in one of them deploys to no one, so counting either as a reference
# would be a false green by construction - `verify` would go on reporting the
# layer healthy while the instruction had become unreachable.
#
# The dev-boundary check asks a different question - what would *dangle* in a
# distributed instance - so it scans README.md too, because `dist export` copies
# it verbatim (dist_cmd.ROOT_FILES). CHANGES.md stays out even there; see
# dev_only_forbidden_references for why.
SHIPPED_DOC_ROOT_FILES = ("README.md",)
def skill_dirs(instructions_dir: Path | None = None) -> list[Path]:
"""Directories under instructions/ (including instructions/dev/) that contain a SKILL.md."""
root = instructions_dir or config.INSTRUCTIONS_DIR
if not root.is_dir():
return []
candidates = list(root.iterdir())
dev_root = root / DEV_SUBDIR
if dev_root.is_dir():
candidates += list(dev_root.iterdir())
return sorted(p for p in candidates if p.is_dir() and (p / SKILL_FILE).exists())
def instruction_files(instructions_dir: Path | None = None) -> list[Path]:
"""Flat instruction files - everything except the layer's own contract.
Includes instructions/dev/*.md, the stack-development-only subset
`dist export` excludes wholesale."""
root = instructions_dir or config.INSTRUCTIONS_DIR
if not root.is_dir():
return []
files = list(root.glob("*.md"))
dev_root = root / DEV_SUBDIR
if dev_root.is_dir():
files += list(dev_root.glob("*.md"))
return sorted(p for p in files if p.name != CONTRACT_FILE)
def is_dev_only(path: Path, instructions_dir: Path | None = None) -> bool:
"""Whether `path` sits under instructions/dev/."""
root = instructions_dir or config.INSTRUCTIONS_DIR
try:
path.relative_to(root / DEV_SUBDIR)
except ValueError:
return False
return True
def target_dirs() -> list[Path]:
return [config.AGENTS_SKILLS_DIR, config.CLAUDE_SKILLS_DIR]
def _read_frontmatter(path: Path) -> tuple[dict | None, str]:
"""Return (frontmatter, error). Exactly one is meaningful."""
text = path.read_text(encoding="utf-8")
if not text.startswith("---"):
return None, "has no YAML frontmatter"
end = text.find("---", 3)
if end == -1:
return None, "frontmatter block is not closed with `---`"
try:
frontmatter = yaml.safe_load(text[3:end]) or {}
except yaml.YAMLError as exc:
return None, f"frontmatter is not valid YAML ({exc})"
if not isinstance(frontmatter, dict):
return None, "frontmatter is not a mapping"
return frontmatter, ""
def _copy_skill(source: Path, target: Path, force: bool) -> None:
if target.is_symlink():
# Left over from the symlink-based mirror this replaced.
target.unlink()
elif target.exists():
if not _looks_like_a_published_skill(target) and not force:
fail(
f"{rel_path(target)} is not a published skill (no SKILL.md). Syncing would "
"delete it and everything in it. Check whether it holds anything worth "
"keeping, then re-run with --force."
)
shutil.rmtree(target)
shutil.copytree(source, target)
def _looks_like_a_published_skill(target: Path) -> bool:
"""Whether this target directory is a skill slot sync owns.
Deliberately a *structural* test, not a comparison against the source. An
edited copy also differs from its source, and re-running `sync` is the
documented fix for exactly that - so refusing on difference would refuse
the repair. What sync must not silently delete is a directory that was
never a published skill at all.
"""
return (target / SKILL_FILE).exists()
def drift(source: Path, target: Path) -> str | None:
"""Describe how a published copy differs from its source, or None."""
if not target.exists():
return "missing"
if target.is_symlink():
return "is a symlink, not a copy"
comparison = filecmp.dircmp(str(source), str(target))
if comparison.diff_files:
return f"differs in {', '.join(sorted(comparison.diff_files))}"
if comparison.left_only:
return f"missing {', '.join(sorted(comparison.left_only))}"
if comparison.right_only:
return f"has extra {', '.join(sorted(comparison.right_only))}"
return None
def _reference_haystacks(
root: Path, *, include_dev: bool, root_files: tuple[str, ...]
) -> list[Path]:
"""Every file that could mention an instruction: the given `root_files`,
every CONTRACT.md, every COLLECTION.md, every skill's SKILL.md, every flat
instruction. `include_dev=False` restricts the skill/instruction portion to
files outside instructions/dev/ - what `dev_only_forbidden_references`
needs, since it specifically asks about mentions from outside that
boundary.
`root_files` is the axis the two callers actually differ on, and it is
passed explicitly rather than defaulted because getting it wrong is silent
in both directions: too wide, and a human-only document keeps a dead
instruction looking alive; too narrow, and a reference that would dangle in
a distributed instance goes unreported. See AGENT_ROOT_FILES and
SHIPPED_DOC_ROOT_FILES."""
haystacks: list[Path] = []
for name in root_files:
candidate = config.ROOT / name
if candidate.exists():
haystacks.append(candidate)
haystacks += sorted(config.ROOT.rglob(CONTRACT_FILE))
haystacks += sorted(config.KB_DIR.rglob("COLLECTION.md")) if config.KB_DIR.is_dir() else []
skills = skill_dirs(root)
instructions = instruction_files(root)
if not include_dev:
skills = [d for d in skills if not is_dev_only(d, root)]
instructions = [p for p in instructions if not is_dev_only(p, root)]
haystacks += [d / SKILL_FILE for d in skills]
haystacks += instructions
return haystacks
def referenced_names(instructions_dir: Path | None = None) -> set[str]:
"""Every instruction filename referenced from somewhere that loads it.
Scans only what an agent can actually reach: AGENTS.md/CLAUDE.md, the
contracts and collection files, the skills, and the other instructions.
README.md and CHANGES.md are deliberately not in the haystack - a mention
there documents an instruction to a human without deploying it to anyone.
"""
root = instructions_dir or config.INSTRUCTIONS_DIR
haystacks = _reference_haystacks(root, include_dev=True, root_files=AGENT_ROOT_FILES)
referenced: set[str] = set()
for path in haystacks:
try:
text = path.read_text(encoding="utf-8")
except OSError: # pragma: no cover - unreadable file
continue
for candidate in instruction_files(root):
if candidate == path:
continue # a file referencing itself is not a reference
if candidate.name in text:
referenced.add(candidate.name)
return referenced
def automatic_load_paths(instructions_dir: Path | None = None) -> list[Path]:
"""Where an agent encounters a link *without* asking for it by name:
AGENTS.md (loaded every session by every other harness) and CLAUDE.md
(loaded every session, but only by Claude Code, which does not load
AGENTS.md on its own), plus every published skill's SKILL.md (loaded
once by the harness, then followed as live procedure).
Deliberately narrower than `referenced_names()`'s haystack: a mention in
a CONTRACT.md, a COLLECTION.md, or another instruction's "see also" is
documentation a reader opts into, not something that runs on its own -
this is what `manual: true` (see instructions/CONTRACT.md) checks against.
"""
root = instructions_dir or config.INSTRUCTIONS_DIR
agents_md = config.ROOT / "AGENTS.md"
claude_md = config.ROOT / "CLAUDE.md"
haystacks: list[Path] = [p for p in (agents_md, claude_md) if p.exists()]
haystacks += [d / SKILL_FILE for d in skill_dirs(root)]
return haystacks
def manual_forbidden_references(instructions_dir: Path | None = None) -> set[str]:
"""Instruction filenames mentioned somewhere they would be picked up
automatically - forbidden for a `manual: true` instruction."""
root = instructions_dir or config.INSTRUCTIONS_DIR
referenced: set[str] = set()
for path in automatic_load_paths(root):
try:
text = path.read_text(encoding="utf-8")
except OSError: # pragma: no cover - unreadable file
continue
for candidate in instruction_files(root):
if candidate.name in text:
referenced.add(candidate.name)
return referenced
def dev_only_forbidden_references(instructions_dir: Path | None = None) -> set[str]:
"""Instruction/skill names under instructions/dev/ mentioned from outside
it - forbidden. `dist export` prunes instructions/dev/ wholesale
(dist_cmd.INSTRUCTIONS_EXCLUDE_DIRS), so a reference from outside it
would either dangle in a distributed instance or leak a dev-only
procedure into a regular content path.
A mention inside a <!-- dist:strip-start/end --> block is exempt: it is
stripped from the haystack text before the scan (`dist_cmd.strip_markers`,
the same utility `dist export` itself uses), because `dist export` drops
that block and instructions/dev/ together - nothing is left dangling.
The haystack is wider here than in `referenced_names()`: README.md is
scanned too, because this check is about what would *dangle* in a shipped
document rather than about what an agent can reach, and `dist export`
copies README.md verbatim (dist_cmd.ROOT_FILES). CHANGES.md is the one
shipped-looking file left out - `dist export` always replaces it wholesale
with dist_templates/CHANGES.md regardless of its content, so a historical
mention there never reaches a distributed instance in the first place."""
root = instructions_dir or config.INSTRUCTIONS_DIR
dev_names = {p.name for p in instruction_files(root) if is_dev_only(p, root)}
dev_names |= {d.name for d in skill_dirs(root) if is_dev_only(d, root)}
if not dev_names:
return set()
referenced: set[str] = set()
haystacks = _reference_haystacks(
root, include_dev=False, root_files=AGENT_ROOT_FILES + SHIPPED_DOC_ROOT_FILES
)
for path in haystacks:
try:
text = dist_cmd.strip_markers(path.read_text(encoding="utf-8"))
except OSError: # pragma: no cover - unreadable file
continue
for name in dev_names:
if name in text:
referenced.add(name)
return referenced
@app.command("sync")
def sync(
force: bool = typer.Option(
False,
"--force",
help="Replace a target directory whose contents differ from the source and was not "
"generated by sync. Without this, sync refuses instead of deleting content it did "
"not create.",
),
):
"""Publish every instructions/<name>/SKILL.md into the harness skill directories."""
sources = skill_dirs()
if not sources:
fail(f"No skills found under {rel_path(config.INSTRUCTIONS_DIR)}.")
names = {source.name for source in sources}
published = []
removed = []
for target_root in target_dirs():
target_root.mkdir(parents=True, exist_ok=True)
for source in sources:
_copy_skill(source, target_root / source.name, force)
# A skill that no longer exists must not keep being offered.
for stale in sorted(target_root.iterdir()):
if stale.name not in names:
removed.append(rel_path(stale))
if stale.is_dir() and not stale.is_symlink():
shutil.rmtree(stale)
else:
stale.unlink()
published.append(rel_path(target_root))
suffix = f", removed {len(removed)} stale" if removed else ""
success(f"Published {len(sources)} skill(s) to {' and '.join(published)}{suffix}")
@app.command("verify")
def verify():
"""Check instructions/ against its type, and every published copy against its source."""
sources = skill_dirs()
instructions = instruction_files()
if not sources and not instructions:
fail(f"Nothing found under {rel_path(config.INSTRUCTIONS_DIR)}.")
issues: list[str] = []
manual: set[str] = set()
# 1. Flat instructions validate against the instruction type-spec.
for path in instructions:
frontmatter, error = _read_frontmatter(path)
if frontmatter is None:
issues.append(f"{path.name}: {error}")
continue
if frontmatter.get("type") != INSTRUCTION_TYPE:
issues.append(f"{path.name}: `type:` should be {INSTRUCTION_TYPE}")
continue
try:
resolver.validate_frontmatter(frontmatter, INSTRUCTION_TYPE)
except ValueError as exc:
issues.append(f"{path.name}: {exc}")
continue
if frontmatter.get("name") != path.stem:
issues.append(
f"{path.name}: frontmatter `name` ({frontmatter.get('name')!r}) "
"does not match the filename"
)
if frontmatter.get("manual"):
manual.add(path.name)
# 2. Skills carry the frontmatter the harness reads. That frontmatter is
# the harness's contract, not this repo's type system's, so it is checked
# directly rather than against a schema.
for source in sources:
frontmatter, error = _read_frontmatter(source / SKILL_FILE)
if frontmatter is None:
issues.append(f"{source.name}: SKILL.md {error}")
continue
if frontmatter.get("name") != source.name:
issues.append(
f"{source.name}: SKILL.md `name` ({frontmatter.get('name')!r}) "
"does not match the folder name"
)
if not frontmatter.get("description"):
issues.append(f"{source.name}: SKILL.md is missing (or has an empty) `description`")
# 3. Published copies match their sources. Missing *everywhere* is a clean
# checkout, not a fault - say what to run instead of reporting drift.
expected = len(sources) * len(target_dirs())
missing = 0
drifted: list[str] = []
for target_root in target_dirs():
for source in sources:
difference = drift(source, target_root / source.name)
if difference == "missing":
missing += 1
elif difference:
drifted.append(f"{rel_path(target_root / source.name)}: {difference}")
if target_root.is_dir():
for extra in sorted(target_root.iterdir()):
if extra.name not in {s.name for s in sources}:
drifted.append(f"{rel_path(extra)}: published but has no source")
issues.extend(drifted)
bootstrap_needed = missing and missing == expected and not drifted
if missing and not bootstrap_needed:
issues.append(f"{missing} published copy/copies missing - run `wikitool instructions sync`")
# 4. An instruction nothing loads is inert. Nothing else would report it -
# unless it is `manual: true`, which inverts the rule over a narrower
# haystack: that instruction must not be linked from AGENTS.md or a
# skill (automatic pickup), though a CONTRACT.md mentioning it by name
# as documentation is fine and expected.
referenced = referenced_names()
forbidden = manual_forbidden_references()
for path in instructions:
if path.name in manual:
if path.name in forbidden:
issues.append(
f"{path.name}: marked `manual` but linked from AGENTS.md, CLAUDE.md, or a "
"skill - that would load it automatically, exactly what `manual` exists to "
"prevent. Remove the link, or drop `manual: true` if it should run routinely."
)
continue
if path.name not in referenced:
issues.append(
f"{path.name}: nothing references it - it deploys to no one. "
"Link it from a skill, a contract, AGENTS.md, or CLAUDE.md, or delete it."
)
# 5. instructions/dev/ is a hard boundary: `dist export` prunes it whole,
# so nothing outside it may depend on something inside it staying
# around in a distributed instance. See dev_only_forbidden_references's
# docstring for the dist:strip exemption.
for name in sorted(dev_only_forbidden_references()):
issues.append(
f"{name}: lives under instructions/dev/ but is referenced from outside it and "
"outside a dist:strip block - `dist export` removes instructions/dev/ wholesale, so "
"that reference would dangle in a distributed instance. Remove the reference, or "
"wrap it in a <!-- dist:strip-start/end --> block if it belongs only to this dev "
"instance."
)
if issues:
fail("Instruction layer issues:\n - " + "\n - ".join(issues))
if bootstrap_needed:
success(
f"{len(instructions)} instruction(s) and {len(sources)} skill(s) valid. "
"No skills published yet - run `wikitool instructions sync` "
"(see instructions/bootstrap.md)."
)
return
success(
f"{len(instructions)} instruction(s) and {len(sources)} skill(s) valid, "
f"{expected} published copy/copies match their source."
)
@app.command("list")
def list_instructions(
json_out: bool = typer.Option(False, "--json", help="Print the listing as JSON"),
):
"""List the flat instructions with their descriptions.
This is how the layer is discovered. `wikitool search` deliberately covers
`kb/` only: a page is found by what it says, an instruction by what it is
for, and that is exactly what `description` carries.
"""
import json as _json
rows = []
for path in instruction_files():
frontmatter, _error = _read_frontmatter(path)
rows.append(
{
"name": (frontmatter or {}).get("name", path.stem),
"path": rel_path(path),
"description": (frontmatter or {}).get("description", ""),
}
)
if json_out:
typer.echo(_json.dumps(rows, indent=2))
return
if not rows:
typer.echo("No instructions found.")
return
for row in rows:
typer.echo(f"{row['name']} ({row['path']})")
typer.echo(f" {row['description']}")
typer.echo("")
typer.echo(f"{len(rows)} instruction(s). Skills are listed by the agent harness itself.")
+470
View File
@@ -0,0 +1,470 @@
"""Deterministic structural health checks for the wiki.
This intentionally covers only what can be computed mechanically: broken
wikilinks, orphan pages, index/page drift, frontmatter schema gaps, and
filename/title mismatches. Semantic judgment (contradictions, staleness,
what's worth writing about next) stays with the LLM - this report gives it a
verified factual foundation instead of requiring it to re-derive these facts
by reading every page.
"""
from __future__ import annotations
import json
from datetime import date
from pathlib import Path
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import rel_path, success
from chemenu.frontmatter_io import frontmatter_error
from chemenu.markdown_code import strip_code_spans
from chemenu.provenance import broken_raw_refs as find_broken_raw_refs
from chemenu.provenance import duplicate_raw_file_owners as find_duplicate_raw_file_owners
from chemenu.provenance import extract_inline_cites
from chemenu.provenance import legacy_citation_markers as find_legacy_citation_markers
from chemenu.provenance import legacy_source_pages as find_legacy_source_pages
from chemenu.provenance import orphan_footnote_defs as find_orphan_footnote_defs
from chemenu.provenance import uncovered_raw_files as find_uncovered_raw_files
from chemenu.provenance import undefined_footnote_refs as find_undefined_footnote_refs
from chemenu.kb_scan import (
GENERATED_INDEX,
WIKILINK_RE,
build_link_graph,
find_duplicate_title_paths,
inbound_links,
load_kb_pages,
)
from chemenu.type_resolver import resolver
# Style guide's one mechanically-checkable rule (hard oracle: a plain count).
# The rest of the style guide (tone, AI-phrase avoidance) is a soft/proxy judgment
# and stays with the LLM - see wiki-manage/wiki-ingest skill guidance, not lint.
#
# The unit is a quote, not a `>` line. It used to be the line, which measured
# the wrap width the rule has no opinion about: one quotation written long
# counted 1 and the same quotation wrapped at 100 columns counted 4. An author
# who took the finding seriously made the page harder to read to quiet it.
QUOTE_LIMIT = 2
# How many hub pages `most_linked` reports. Purely informational (wiki-status
# surfaces it); not a finding, so the cutoff only bounds report size.
MOST_LINKED_COUNT = 10
def count_quote_blocks(body: str) -> int:
"""How many distinct blockquotes `body` carries.
A run of consecutive `>` lines is one quote; a blank line or any
non-quoted line ends it. Code is masked out first, so a `>` inside a
fenced shell transcript is a prompt, not a quotation.
Lazy continuation - a quote whose wrapped lines drop the `>` - reads here
as two quotes rather than one. That over-counts in the direction the limit
already errs on, and the corpus prefixes every line, so the alternative
(tracking paragraph state) buys nothing.
"""
count, in_quote = 0, False
for line in strip_code_spans(body).splitlines():
is_quote = line.lstrip().startswith(">")
if is_quote and not in_quote:
count += 1
in_quote = is_quote
return count
def run_lint(kb_dir: Path) -> dict:
pages = load_kb_pages(kb_dir)
duplicate_titles = find_duplicate_title_paths(kb_dir, config.ROOT)
# Pages whose frontmatter can't be parsed read back as `{}` everywhere
# else, which would let them slip past every frontmatter-driven check
# below with no finding at all - so they are detected explicitly.
frontmatter_errors = []
for title, page in sorted(pages.items()):
reason = frontmatter_error(page.path)
if reason is None and not page.frontmatter.get("type"):
reason = "missing `type:` field"
if reason is not None:
frontmatter_errors.append({"page": title, "error": reason})
graph = build_link_graph(pages)
broken_links = [
{"page": title, "target": target}
for title, targets in graph.items()
for target in sorted(targets)
if target not in pages
]
inbound = inbound_links({t: v for t, v in graph.items() if t != "index"})
orphan_pages = sorted(
title
for title, sources in inbound.items()
if not sources
and title not in ("index", "log")
# comparison pages are not linked to by design; index.md is sufficient coverage
and pages[title].kind != "comparison"
)
# Same link graph, opposite end: the most-linked-to pages are the wiki's
# hubs. Reported (not judged) so `wiki-status` can show them without
# re-deriving the graph.
inbound_counts = {title: len(sources) for title, sources in inbound.items()}
most_linked = [
{"page": title, "inbound": count}
for title, count in sorted(inbound_counts.items(), key=lambda kv: (-kv[1], kv[0]))
if count > 0
][:MOST_LINKED_COUNT]
# The catalog is sharded: `kb/index.md` is a map carrying counts and links,
# and the page rows live in a generated INDEX.md per collection/area. Both
# halves have to be read, or every page reads as missing from the index.
index_text = "".join(
path.read_text(encoding="utf-8")
for path in [kb_dir / "index.md", *sorted(kb_dir.rglob(GENERATED_INDEX))]
if path.exists()
)
index_links = {m.group(1).strip() for m in WIKILINK_RE.finditer(index_text)}
missing_from_index = sorted(set(pages) - index_links - {"index", "log"})
dangling_index_entries = sorted(index_links - set(pages))
title_mismatches = []
for title, page in sorted(pages.items()):
if page.kind not in ("entity", "concept"):
continue
h1 = page.h1_title
if h1 is not None and h1 != title:
title_mismatches.append({"page": title, "h1": h1})
unmarked_provenance = []
for title, page in sorted(pages.items()):
if page.kind not in ("entity", "concept"):
continue
sources_list = page.frontmatter.get("sources") or []
if not sources_list and page.frontmatter.get("provenance") != "general":
unmarked_provenance.append(title)
citation_frontmatter_drift = []
for title, page in sorted(pages.items()):
sources_list = set(page.frontmatter.get("sources") or [])
cited = {cited_title for cited_title, _file in extract_inline_cites(page.body)}
cited.discard(title) # a source page citing itself for a specific file within it is not drift
for missing_source in sorted(cited - sources_list):
citation_frontmatter_drift.append({"page": title, "cited_but_not_in_sources": missing_source})
legacy_citation_markers = find_legacy_citation_markers(pages)
undefined_footnote_refs = find_undefined_footnote_refs(pages)
orphan_footnote_defs = find_orphan_footnote_defs(pages)
# The frontmatter half of the link graph. `broken_links` above only walks
# `[[wikilinks]]` in page *bodies*, so a `related:`/`sources:`/`entities:`
# entry naming a page that does not exist - a rename that was not
# propagated, a deleted page, or a URL pasted where a title belongs - used
# to pass every check. Which fields hold page titles is declared by each
# type-spec's `page_ref_fields:`, not hardcoded here.
dangling_frontmatter_refs = []
for title, page in sorted(pages.items()):
type_path = page.frontmatter.get("type")
if not type_path:
continue
try:
ref_fields = resolver.get_page_ref_fields(type_path, page.path)
except ValueError:
continue # unresolvable type is already reported as type_resolution_errors
for field in ref_fields:
for target in page.frontmatter.get(field) or []:
if target not in pages:
dangling_frontmatter_refs.append(
{"page": title, "field": field, "target": target}
)
quote_limit_violations = []
for title, page in sorted(pages.items()):
quote_count = count_quote_blocks(page.body)
if quote_count > QUOTE_LIMIT:
quote_limit_violations.append({"page": title, "quote_count": quote_count})
# Type system validation. Lint reports are not validated here: they are
# written to `reports/` outside kb/ and are never pages, so nothing this
# loop scans can be one.
invalid_type_paths = []
type_resolution_errors = []
schema_validation_errors = []
for title, page in sorted(pages.items()):
type_path = page.frontmatter.get("type")
if not type_path:
continue
# Check if type path is valid
if not type_path.endswith('.md'):
invalid_type_paths.append({"page": title, "type": type_path, "error": "Type path must end with .md"})
continue
# Try to resolve and validate the type
try:
resolver.load_type_spec(type_path, page.path)
# Try schema validation
try:
resolver.validate_frontmatter(page.frontmatter, type_path, page.path)
except ValueError as schema_error:
schema_validation_errors.append({"page": title, "type": type_path, "error": str(schema_error)})
except ValueError as resolution_error:
type_resolution_errors.append({"page": title, "type": type_path, "error": str(resolution_error)})
return {
"generated": date.today().isoformat(),
"page_count": len(pages),
"frontmatter_errors": frontmatter_errors,
"broken_links": broken_links,
"orphan_pages": orphan_pages,
"most_linked": most_linked,
"inbound_counts": inbound_counts,
"missing_from_index": missing_from_index,
"dangling_index_entries": dangling_index_entries,
"title_mismatches": title_mismatches,
"duplicate_titles": duplicate_titles,
"uncovered_raw_files": find_uncovered_raw_files(config.RAW_DIR, pages),
"broken_raw_refs": find_broken_raw_refs(pages),
"duplicate_raw_file_owners": find_duplicate_raw_file_owners(pages),
"legacy_source_pages": find_legacy_source_pages(pages),
"unmarked_provenance": unmarked_provenance,
"citation_frontmatter_drift": citation_frontmatter_drift,
"legacy_citation_markers": legacy_citation_markers,
"undefined_footnote_refs": undefined_footnote_refs,
"orphan_footnote_defs": orphan_footnote_defs,
"dangling_frontmatter_refs": dangling_frontmatter_refs,
"quote_limit_violations": quote_limit_violations,
"invalid_type_paths": invalid_type_paths,
"type_resolution_errors": type_resolution_errors,
"schema_validation_errors": schema_validation_errors,
}
def _section(lines: list[str], title: str, items: list, formatter) -> None:
lines.append(f"## {title}")
lines.append("")
if not items:
lines.append("None found.")
else:
for item in items:
lines.append(f"- {formatter(item)}")
lines.append("")
def render_markdown(report: dict) -> str:
lines = [f"# Structural Lint Report ({report['generated']})", ""]
lines.append(f"Scanned {report['page_count']} pages under `wiki/`. This report covers only")
lines.append("mechanically-verifiable structural issues; see the Semantic Review section")
lines.append("below for judgment calls the LLM should complete.")
lines.append("")
_section(
lines, "Unreadable Frontmatter", report["frontmatter_errors"],
lambda i: f"[[{i['page']}]] - {i['error']}",
)
_section(
lines, "Broken Wikilinks", report["broken_links"],
lambda i: f"[[{i['page']}]] links to missing [[{i['target']}]]",
)
_section(lines, "Orphan Pages (no inbound links)", report["orphan_pages"], lambda i: f"[[{i}]]")
_section(
lines, f"Most-Linked Pages (top {MOST_LINKED_COUNT} hubs)", report["most_linked"],
lambda i: f"[[{i['page']}]] - {i['inbound']} inbound link(s)",
)
_section(lines, "Pages Missing from index.md", report["missing_from_index"], lambda i: f"[[{i}]]")
_section(lines, "Dangling index.md Entries", report["dangling_index_entries"], lambda i: f"[[{i}]]")
_section(
lines, "Duplicate Titles (naming collisions)", report["duplicate_titles"],
lambda i: f"`{i['stem']}` -> {', '.join(f'`{p}`' for p in i['paths'])}",
)
_section(
lines, "Filename / H1 Title Mismatches", report["title_mismatches"],
lambda i: f"[[{i['page']}]] H1 is '{i['h1']}'",
)
_section(
lines, "Uncovered Raw Files (no source page)", report["uncovered_raw_files"],
lambda i: f"`{i}`",
)
_section(
lines, "Broken raw_files References", report["broken_raw_refs"],
lambda i: f"[[{i['page']}]] -> `{i['raw_path']}` (does not exist)",
)
_section(
lines, "Raw Files With More Than One Owner", report["duplicate_raw_file_owners"],
lambda i: f"`{i['raw_file']}` is claimed by " + ", ".join(f"[[{t}]]" for t in i["owners"]),
)
_section(
lines, "Legacy source: Field (not yet migrated to raw_files:)", report["legacy_source_pages"],
lambda i: f"[[{i['page']}]] source: `{i['source']}` ({i['reason']})",
)
_section(
lines, "Pages Missing provenance: general Marker", report["unmarked_provenance"],
lambda i: f"[[{i}]] has no sources and is not marked `provenance: general`",
)
_section(
lines, "Citation / Frontmatter Drift", report["citation_frontmatter_drift"],
lambda i: f"[[{i['page']}]] cites [[{i['cited_but_not_in_sources']}]] inline but it is missing from frontmatter `sources:`",
)
_section(
lines, "Legacy Citation Markers (pre-migration `^[[...]]`)", report["legacy_citation_markers"],
lambda i: f"[[{i['page']}]] still has `{i['marker']}` - run `wikitool cite add` and replace it with the `[^cite-id]` it prints",
)
_section(
lines, "Undefined Footnote References", report["undefined_footnote_refs"],
lambda i: f"[[{i['page']}]] references `[^{i['ref']}]`, which has no `[^{i['ref']}]: [[...]]` definition",
)
_section(
lines, "Orphan Footnote Definitions", report["orphan_footnote_defs"],
lambda i: f"[[{i['page']}]] defines `[^{i['id']}]` (-> [[{i['source']}]]) but nothing references it - run `wikitool cite sync`",
)
_section(
lines, "Dangling Frontmatter References", report["dangling_frontmatter_refs"],
lambda i: f"[[{i['page']}]] `{i['field']}:` names `{i['target']}`, which is not a page",
)
_section(
lines, "Invalid Type Paths", report["invalid_type_paths"],
lambda i: f"[[{i['page']}]] has type: `{i['type']}` - {i['error']}",
)
_section(
lines, "Type Resolution Errors", report["type_resolution_errors"],
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
)
_section(
lines, "Schema Validation Errors", report["schema_validation_errors"],
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
)
_section(
lines, f"Pages Exceeding Quote Limit (>{QUOTE_LIMIT}/page)", report["quote_limit_violations"],
lambda i: f"[[{i['page']}]] has {i['quote_count']} quotes - trim or confirm they're load-bearing",
)
lines.append("## Semantic Review (LLM to complete)")
lines.append("")
lines.append("- Contradictions across pages: TODO")
lines.append("- Stale claims (unconfirmed >6 months): TODO")
lines.append("- Suggested new pages / missing cross-references: TODO")
lines.append("")
return "\n".join(lines)
# Sections that always carry content but are not findings, so the summary
# handles them separately: a hub list is a statistic, and the semantic review
# is the checklist that follows the report rather than part of it.
INFORMATIONAL_SECTIONS = ("Most-Linked Pages",)
SEMANTIC_REVIEW_SECTION = "Semantic Review"
def _split_sections(markdown: str) -> tuple[str, list[tuple[str, str]]]:
"""Cut a rendered report into its preamble and (title, body) sections."""
preamble, *rest = markdown.split("\n## ")
sections = []
for part in rest:
title, _, body = part.partition("\n")
sections.append((title.strip(), body.strip()))
return preamble.rstrip(), sections
def render_summary(report: dict) -> str:
"""The same report with the empty sections removed.
On a healthy corpus the full report is better than 90% "None found.", so
reading it in the terminal means paging past the answer. The file on disk
stays complete - this is what gets printed, and the written path underneath
it is how the rest is reached without running lint a second time.
"""
preamble, sections = _split_sections(render_markdown(report))
findings, trailing = [], []
for title, body in sections:
if title.startswith(SEMANTIC_REVIEW_SECTION):
trailing.append((title, body))
elif body != "None found." and not title.startswith(INFORMATIONAL_SECTIONS):
findings.append((title, body))
lines = [preamble, ""]
if not findings:
lines += ["No structural findings.", ""]
for title, body in findings + trailing:
lines += [f"## {title}", "", body, ""]
return "\n".join(lines)
def default_report_path(report: dict) -> Path:
"""Where a report goes when the caller names no path.
`reports/` is derived and gitignored ([reports/CONTRACT.md]), so writing
here by default costs the tree nothing.
"""
return config.ROOT / "reports" / f"Lint Report {report['generated']}.md"
# Findings that make a tree structurally wrong rather than merely untidy.
# `orphan_pages` is deliberately absent: many pages are validly reachable
# through the index or navigation only. `quote_limit_violations` is advisory
# too - it flags a habit, not a broken tree.
#
# One definition, used by `lint --fail-on-error` and by the eval scorecard: if
# the two disagreed, a run could pass its score while lint refused it.
HARD_ERROR_KEYS = (
"frontmatter_errors",
"broken_links",
"dangling_index_entries",
"duplicate_titles",
"broken_raw_refs",
"duplicate_raw_file_owners",
"legacy_source_pages",
"citation_frontmatter_drift",
"legacy_citation_markers",
"undefined_footnote_refs",
"orphan_footnote_defs",
"dangling_frontmatter_refs",
"invalid_type_paths",
"type_resolution_errors",
"schema_validation_errors",
)
def has_hard_errors(report: dict) -> bool:
return any(report.get(key) for key in HARD_ERROR_KEYS)
def lint_command(
json_out: bool = typer.Option(False, "--json", help="Print the raw findings as JSON and write no report"),
markdown_out: Optional[Path] = typer.Option(None, "--markdown", help="Write the markdown report here instead of the default reports/Lint Report <date>.md"),
full: bool = typer.Option(False, "--full", help="Print the whole report instead of only the sections with findings"),
fail_on_error: bool = typer.Option(False, "--fail-on-error", help="Exit non-zero if hard errors were found"),
):
"""Run structural lint checks against kb/.
Unless `--json` is given, the full report is always written to a file and
its path is printed. That path is the point: a lint report is long, and an
agent that only saw it on stdout had no way back to the part it scrolled
past except by running lint again - two budget slots for one look at the
corpus.
"""
report = run_lint(config.KB_DIR)
if json_out:
typer.echo(json.dumps(report, indent=2))
if fail_on_error and has_hard_errors(report):
raise typer.Exit(code=1)
return
typer.echo(render_markdown(report) if full else render_summary(report))
target = markdown_out or default_report_path(report)
frontmatter = (
"---\n"
"type: types/lint-report.md\n"
f"created: {report['generated']}\n"
f"summary: Structural lint report - {report['page_count']} pages scanned\n"
"---\n\n"
)
target.parent.mkdir(parents=True, exist_ok=True)
target.write_text(frontmatter + render_markdown(report) + "\n", encoding="utf-8")
success(f"Full report written to {rel_path(target)}")
if fail_on_error and has_hard_errors(report):
raise typer.Exit(code=1)
+84
View File
@@ -0,0 +1,84 @@
"""Append correctly-formatted entries to wiki/log.md."""
from __future__ import annotations
import re
from pathlib import Path
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import fail, rel_path, success, today_iso
app = typer.Typer(help="Manage wiki/log.md.")
VALID_OPS = ["ingest", "query", "lint", "create", "update", "delete", "rename"]
# Matches the "## [YYYY-MM-DD] op | title" heading `format_log_entry` writes,
# in file order (oldest first, since entries are appended).
LOG_ENTRY_RE = re.compile(r"^## \[(\d{4}-\d{2}-\d{2})\] (\S+) \| (.+)$", re.MULTILINE)
def format_log_entry(op: str, title: str, body: str = "", today: Optional[str] = None) -> str:
today = today or today_iso()
entry = f"## [{today}] {op} | {title}\n"
if body.strip():
entry += f"\n{body.strip()}\n"
entry += "\n---\n"
return entry
def parse_log_entries(text: str) -> list[tuple[str, str, str]]:
"""Return every logged (date, op, title) entry in file order."""
return [(m.group(1), m.group(2), m.group(3)) for m in LOG_ENTRY_RE.finditer(text)]
def ingests_since_last_lint(entries: list[tuple[str, str, str]]) -> int:
"""Count `ingest` entries logged after the most recent `lint` entry (or
since the start of the log, if it has never been linted).
This is the deterministic count behind the Maintenance Schedule's "every
10 sources" full-lint cadence: nothing else in the system tracks it, so
without this the claim was prose with no enforcement - an agent (or user)
had to remember to count."""
count = 0
for _date, op, _title in entries:
if op == "lint":
count = 0
elif op == "ingest":
count += 1
return count
@app.command("append")
def log_append(
op: str = typer.Option(..., "--op", help="|".join(VALID_OPS)),
title: str = typer.Option(..., "--title", help="Brief description, e.g. a source path"),
body: str = typer.Option("", "--body", help="Optional multi-line details"),
body_file: Optional[Path] = typer.Option(None, "--body-file", help="Read the body from a file instead of --body"),
):
if op not in VALID_OPS:
fail(f"--op must be one of {VALID_OPS}")
text = body
if body_file:
text = body_file.read_text(encoding="utf-8")
entry = format_log_entry(op, title, text)
with config.LOG_FILE.open("a", encoding="utf-8") as f:
f.write("\n" + entry)
success(f"Appended log entry to {rel_path(config.LOG_FILE)}")
@app.command("status")
def log_status():
"""Report how many `ingest` operations have been logged since the last
`lint` - the deterministic trigger for the Maintenance Schedule's "every
10 sources" full-lint cadence. Read-only."""
if not config.LOG_FILE.exists():
success("No wiki/log.md yet; nothing logged.")
return
entries = parse_log_entries(config.LOG_FILE.read_text(encoding="utf-8"))
count = ingests_since_last_lint(entries)
typer.echo(f"Ingests since last lint: {count}")
if count >= 10:
typer.echo("Threshold reached (>=10) - run the wiki-lint skill (`tools/wikitool lint ...`) next.")
success(f"{len(entries)} total log entries in {rel_path(config.LOG_FILE)}.")
+381
View File
@@ -0,0 +1,381 @@
"""`wikitool migrate` - content migrations: what this instance still owes, and
whether a bulk rewrite broke anything.
Five commands around two facts. `.wikitool-kb.json` (see `chemenu/kb_state.py`)
records what shape the content is in, so the chain of outstanding migrations is
computed rather than guessed. `migrate verify` compares the corpus against a git
revision on the invariants a migration must not change (see
`chemenu/corpus_diff.py`).
**There is no `migrate run`.** An `assisted` migration is a procedure an agent
carries out page by page; the tool keeps the books and checks the result. A
`run` would claim an ability that does not exist - it arrives when mechanical
primitives do.
"""
from __future__ import annotations
import json as _json
import subprocess
from pathlib import Path
from typing import Optional
import typer
from chemenu import config, corpus_diff, kb_scan, kb_state, version as version_mod
from chemenu.commands._util import console, fail, rel_path, success, today_iso
from chemenu.frontmatter_io import read_page
from chemenu.page import Page
from chemenu.version import Version, VersionError
app = typer.Typer(help="Content migrations: outstanding chain, bookkeeping, and verification.")
# --- shared state loading --------------------------------------------------
def _versions() -> tuple[Version, Optional[Version]]:
"""(stack version, kb version).
An unreadable VERSION or a corrupt state file exits 1 here (via `fail`,
which raises); a *missing* kb version returns None, because that is a
state each command explains in its own words rather than an error."""
try:
return version_mod.read_version(), kb_state.read_kb_version()
except VersionError as exc:
fail(str(exc))
raise # unreachable: fail() raises typer.Exit
# --- migrate list ----------------------------------------------------------
@app.command("list")
def list_command(
json_out: bool = typer.Option(False, "--json", help="Print the migrations as JSON"),
):
"""List every migration document, oldest target first. Read-only."""
migrations = kb_state.load_migrations()
if json_out:
typer.echo(
_json.dumps(
[
{
"name": m.name,
"migrates_to": str(m.target),
"migration_kind": m.kind,
"description": m.description,
"path": m.relative_path,
}
for m in migrations
],
indent=2,
)
)
return
if not migrations:
success(f"No migration documents under {rel_path(kb_state.migrations_dir())}.")
return
for migration in migrations:
console.print(f"[bold]{migration.target}[/bold] {migration.name} ({migration.kind})")
if migration.description:
console.print(f" {migration.description}")
# --- migrate status --------------------------------------------------------
@app.command("status")
def status_command(
json_out: bool = typer.Option(False, "--json", help="Print the chain as JSON"),
):
"""Show the migrations this instance still owes, in the order they run.
Read-only. Exit 1 only when the KB version is undeclared - that is a
question the tool refuses to answer by guessing."""
stack, kb_version = _versions()
migrations = kb_state.load_migrations()
if kb_version is None:
if json_out:
typer.echo(
_json.dumps(
{"stack_version": str(stack), "kb_version": None, "pending": None}, indent=2
)
)
fail(
f"{kb_state.KB_STATE_FILENAME} is missing - this instance has never declared what "
f"shape its content is in, and guessing would be wrong exactly when it matters.\n"
f"Declare it once: `wikitool migrate baseline <version>` (use {stack} if this "
f"instance's content has never been migrated behind its machinery)."
)
return
pending = kb_state.chain(migrations, kb_version, stack)
if json_out:
typer.echo(
_json.dumps(
{
"stack_version": str(stack),
"kb_version": str(kb_version),
"pending": [
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
for m in pending
],
},
indent=2,
)
)
return
console.print(f"stack {stack}, content {kb_version}")
if not pending:
if kb_version < stack:
console.print(
f"[green]Nothing outstanding[/green] - no migration targets the range "
f"({kb_version}, {stack}]."
)
else:
console.print("[green]Nothing outstanding[/green] - content matches the machinery.")
return
console.print(f"[cyan]{len(pending)} migration(s) outstanding, in this order:[/cyan]")
for position, migration in enumerate(pending, start=1):
console.print(f" {position}. {migration.target} {migration.name} ({migration.kind})")
if migration.description:
console.print(f" {migration.description}")
console.print(f" {migration.relative_path}")
console.print(
f"\nRun the first one, then record it: `wikitool migrate done {pending[0].target}`.\n"
"The procedure is instructions/migrate-corpus.md."
)
# --- migrate done / baseline ----------------------------------------------
@app.command("done")
def done_command(
version: str = typer.Argument(..., help="The migration's target version, e.g. 1.4.0"),
pages: Optional[int] = typer.Option(None, "--pages", help="How many pages it touched"),
dry_run: bool = typer.Option(False, "--dry-run", help="Report without writing"),
):
"""Record one migration as applied, advancing the KB version to its target.
Refuses any version that is not the *next* link in the chain: skipping a
migration is how a corpus ends up in a shape no version describes, and an
interrupted multi-step upgrade has to be resumable rather than guessable."""
stack, kb_version = _versions()
if kb_version is None:
fail(
f"{kb_state.KB_STATE_FILENAME} is missing - run `wikitool migrate baseline <version>` "
"before recording a migration."
)
return
try:
target = Version.parse(version)
except VersionError as exc:
fail(str(exc))
return
migrations = kb_state.load_migrations()
expected = kb_state.next_link(migrations, kb_version, stack)
if expected is None:
fail(
f"Nothing is outstanding: content is at {kb_version}, machinery at {stack}, and no "
f"migration targets the range in between."
)
return
if expected.target != target:
fail(
f"{target} is not the next migration. The chain from {kb_version} continues with "
f"{expected.target} ({expected.name}) - applying them out of order leaves the corpus "
f"in a shape no version describes.\nRun `wikitool migrate status` to see the order."
)
return
state = kb_state.read_kb_state() or {}
applied = list(state.get("applied") or [])
entry = {"migration": expected.name, "at": today_iso()}
if pages is not None:
entry["pages"] = pages
applied.append(entry)
if dry_run:
success(f"Dry run: content {kb_version} -> {target} ({expected.name}). Nothing written.")
return
kb_state.write_kb_state(target, applied)
remaining = kb_state.chain(migrations, target, stack)
success(
f"Content is now {target} ({expected.name}). "
+ (
f"{len(remaining)} migration(s) still outstanding - next is {remaining[0].target}."
if remaining
else "Nothing outstanding."
)
)
@app.command("baseline")
def baseline_command(
version: str = typer.Argument(..., help="The shape this instance's content is already in"),
force: bool = typer.Option(
False, "--force", help="Overwrite an existing declaration (not a substitute for `done`)"
),
):
"""Declare the KB version once, for an instance that never had one.
Only for a tree predating `.wikitool-kb.json`. Advancing the version after
a migration is `migrate done`, which checks the chain; this command does
not, which is why it refuses to overwrite silently."""
try:
target = Version.parse(version)
except VersionError as exc:
fail(str(exc))
return
try:
existing = kb_state.read_kb_version()
except VersionError as exc:
fail(str(exc))
return
if existing is not None and not force:
fail(
f"This instance already declares content version {existing}. Use "
f"`wikitool migrate done <version>` to advance it after a migration, or --force "
f"if the declaration itself is wrong."
)
return
state = kb_state.read_kb_state() or {}
kb_state.write_kb_state(target, list(state.get("applied") or []))
success(f"Content version declared as {target}.")
# --- migrate verify --------------------------------------------------------
def _git_show(rev: str, relative: str) -> Optional[str]:
result = subprocess.run(
["git", "show", f"{rev}:{relative}"],
cwd=config.ROOT,
capture_output=True,
text=True,
)
return result.stdout if result.returncode == 0 else None
def _paths_at(rev: str) -> Optional[list[str]]:
result = subprocess.run(
["git", "ls-tree", "-r", "--name-only", "-z", rev, "--", "kb"],
cwd=config.ROOT,
capture_output=True,
text=True,
)
if result.returncode != 0:
return None
# The same page/not-a-page rule the working tree is read with. Answering it
# differently on the two sides reported every COLLECTION.md and INDEX.md as
# a page that had since disappeared.
return [
path
for path in result.stdout.split("\0")
if path.startswith("kb/") and kb_scan.is_page_path(path[len("kb/"):])
]
def _shapes_at_revision(rev: str, wanted: set[str]) -> dict[str, corpus_diff.PageShape]:
"""Page shapes as of `rev`, keyed by repo-relative path."""
import tempfile
shapes: dict[str, corpus_diff.PageShape] = {}
paths = _paths_at(rev)
if paths is None:
fail(f"`git show {rev}` failed - is {rev} a revision in this repository?")
return shapes
with tempfile.TemporaryDirectory() as tmp:
for relative in paths:
if wanted and not any(relative.startswith(prefix) for prefix in wanted):
continue
text = _git_show(rev, relative)
if text is None:
continue
# read_page owns frontmatter parsing (and its error contract), so the
# historical blob is materialised under its real filename - the stem
# is the page title, which PageShape compares.
scratch = Path(tmp) / Path(relative).name
scratch.write_text(text, encoding="utf-8")
try:
frontmatter, body = read_page(scratch)
except Exception: # noqa: BLE001 - an unparseable historical page is not this tool's error
continue
shapes[relative] = corpus_diff.PageShape.of(Page(Path(relative), frontmatter, body))
return shapes
def _shapes_now(wanted: set[str]) -> dict[str, corpus_diff.PageShape]:
shapes: dict[str, corpus_diff.PageShape] = {}
for path in kb_scan.iter_kb_pages(config.KB_DIR):
relative = path.relative_to(config.ROOT).as_posix()
if wanted and not any(relative.startswith(prefix) for prefix in wanted):
continue
try:
frontmatter, body = read_page(path)
except Exception: # noqa: BLE001 - lint reports unreadable frontmatter
continue
shapes[relative] = corpus_diff.PageShape.of(Page(path, frontmatter, body))
return shapes
@app.command("verify")
def verify_command(
from_rev: str = typer.Option(..., "--from", help="Git revision to compare against, e.g. HEAD"),
path: Optional[list[str]] = typer.Option(
None, "--path", help="Limit to a subtree, repeatable (e.g. kb/concepts)"
),
expect_body_change: bool = typer.Option(
False, "--expect-body-change", help="Also report pages whose body did not change at all"
),
json_out: bool = typer.Option(False, "--json", help="Print the diff as JSON"),
fail_on_error: bool = typer.Option(
False, "--fail-on-error", help="Exit 1 if any invariant changed"
),
):
"""Compare kb/ against a git revision on the invariants a content migration
must not change: wikilink and citation *counts*, footnote definitions, H1,
and structural frontmatter.
Not migration-specific - worth running after any bulk rewrite. `lint` cannot
answer this: it reads one revision, so a reference that went missing is
invisible to it."""
wanted = {p.rstrip("/") for p in (path or [])}
before = _shapes_at_revision(from_rev, wanted)
after = _shapes_now(wanted)
diff = corpus_diff.compare(before, after, expect_body_change=expect_body_change)
if json_out:
typer.echo(
_json.dumps(
{
"from": from_rev,
"compared": diff.compared,
"added": diff.added,
"removed": diff.removed,
"findings": [
{"path": f.path, "kind": f.kind, "detail": f.detail} for f in diff.findings
],
},
indent=2,
)
)
else:
typer.echo(corpus_diff.render_report(diff, from_rev))
if diff.findings and fail_on_error:
raise typer.Exit(code=1)
+331
View File
@@ -0,0 +1,331 @@
"""Scaffold new wiki pages from type-spec templates.
These commands produce structurally-correct frontmatter and a body
skeleton with TODO placeholders by loading templates from type-spec files.
The prose (Description, Summary, Key Takeaways, ...) is still written by the
LLM afterwards with its normal edit tool.
The split between type-spec templates and LLM-provided prose is intentional:
type definitions (naming, frontmatter shape, directory placement, templates) are
deterministic and stored in /types/; the content is judgment and provided by the LLM.
Frontmatter defaults, enum validity, and required-ness all come from the
type's `.schema.yaml` (via `TypeResolver`) - nothing here re-declares them.
Directory placement for subtype-driven types (currently just entities) also
comes from the type-spec, via its `layout:` frontmatter (see
`TypeResolver.get_layout`) - not a hand-maintained Python dict.
"""
from __future__ import annotations
from pathlib import Path
import datetime
from typing import Any, Dict, Optional
import re
import typer
from chemenu import config
from chemenu.commands._util import (
check_collision,
check_raw_files_exist,
fail,
parse_set_fields,
rel_path,
success,
)
from chemenu.frontmatter_io import write_page
from chemenu.type_resolver import resolver
def _default_summary(summary: str) -> str:
"""Scaffold-time placeholder for an unfilled --summary, so schema
validation's `summary` minLength requirement doesn't block page creation.
Matches the same "TODO: add summary" text index_build.py already falls
back to when a page has no frontmatter summary."""
return summary.strip() or "TODO: add summary"
def _resolve_type_and_get_template(type_path: str, source_dir: Path = None):
"""Resolve a type path, load the type-spec, and extract its template."""
type_spec = resolver.load_type_spec(type_path, source_dir)
template = resolver.extract_template(type_spec)
return type_spec, template
def _enum_help(type_path: str, field_name: str) -> str:
"""Build a help string from the schema's own enum, so any text listing a
field's valid values can never drift from what the schema accepts."""
return f"One of: {'|'.join(resolver.get_enum(type_path, field_name))}"
def _parse_date(field_name: str, text: str) -> datetime.date:
"""A `--set <date field>=<value>` as a real date, or a refusal naming it."""
try:
return datetime.date.fromisoformat(text)
except ValueError:
fail(f"--set {field_name}={text!r} must be YYYY-MM-DD.")
def _build_frontmatter(
type_path: str, schema: Optional[Dict[str, Any]], today: datetime.date, explicit: Dict[str, Any]
) -> Dict[str, Any]:
"""Build a page's frontmatter dict in schema-declaration order.
`explicit` supplies every CLI-derived value the caller already has;
fields not in `explicit` get a type-appropriate default (today's date for
date-formatted fields, the scaffold placeholder for `summary`, the
schema's own `default:` where declared, an empty list for arrays), or are
omitted entirely if optional with no sensible default (e.g.
`source_url`). This is what lets frontmatter shape - and scaffold-time
defaults like `provenance: general` or `confidence: 0.5` - follow the
schema instead of being hand-declared per CLI command.
"""
frontmatter: Dict[str, Any] = {"type": type_path}
for field_name, field_schema in (schema or {}).get("properties", {}).items():
if field_name == "type":
continue
if field_name in explicit:
value = explicit[field_name]
# A `--set date=2026-08-23` arrives as a string; the corpus stores
# dates as `datetime.date`, and the schema already says which fields
# those are. Converting here keeps one representation on disk instead
# of leaving it to whoever wrote the CLI call.
if field_schema.get("format") == "date" and isinstance(value, str):
value = _parse_date(field_name, value)
frontmatter[field_name] = value
elif field_name == "summary":
frontmatter[field_name] = _default_summary("")
elif field_name == "author":
resolved_author = config.default_author()
if resolved_author is None:
fail(
"No author configured for this instance. Set `git config user.name`, "
"or export WIKI_AUTHOR to override it, then retry."
)
frontmatter[field_name] = resolved_author
elif field_schema.get("format") == "date":
frontmatter[field_name] = today
elif "default" in field_schema:
frontmatter[field_name] = field_schema["default"]
elif field_schema.get("type") == "array":
frontmatter[field_name] = []
# Carry through any caller-supplied field the schema doesn't declare,
# rather than silently dropping it: a typo'd `--set` must surface as a
# validation error (via `additionalProperties: false`) instead of being
# quietly ignored, and a schema that does allow extra properties should
# keep them.
for field_name, value in explicit.items():
if field_name not in frontmatter:
frontmatter[field_name] = value
return frontmatter
def _filter_bullets(value: Any) -> str:
"""Render a list frontmatter field as `- [[item]]` bullet lines."""
items = value or []
return "\n".join(f"- [[{item}]]" for item in items) if items else "- None identified"
def _filter_join(value: Any) -> str:
"""Comma-join a list frontmatter field."""
return ", ".join(value or [])
def _filter_capitalize(value: Any) -> str:
return str(value).capitalize()
def _filter_table_header(value: Any) -> str:
"""Render an array field as wikilinked markdown table column headers."""
return " | ".join(f"[[{item}]]" for item in (value or []))
def _filter_table_sep(value: Any) -> str:
"""Render the markdown table separator row for an array field, one
column per item."""
return "|".join("--------" for _ in (value or [])) or "--------"
def _filter_table_cells(value: Any) -> str:
"""Render a placeholder table body row, one cell per array item."""
return " | ".join("..." for _ in (value or []))
_TEMPLATE_FILTERS = {
"bullets": _filter_bullets,
"join": _filter_join,
"capitalize": _filter_capitalize,
"table_header": _filter_table_header,
"table_sep": _filter_table_sep,
"table_cells": _filter_table_cells,
}
def _apply_template_variables(template: str, variables: Dict[str, Any]) -> str:
"""Apply variable substitutions to a template string.
Supports:
- `{field}` - plain substitution from `variables[field]`
- `{field|filter}` - apply a named filter (bullets, join, capitalize)
to `variables[field]`'s value, so templates can render list/enum
frontmatter fields directly instead of the caller precomputing a
separate display-only variable for each one
- `{field|literal text}` - literal fallback if `field` isn't in
`variables` at all and the suffix isn't a recognized filter name
"""
def replace_match(match: re.Match) -> str:
full_match = match.group(0)
var_name = match.group(1)
if '|' in var_name:
var_name, suffix = var_name.split('|', 1)
var_name = var_name.strip()
suffix = suffix.strip()
if var_name not in variables:
return suffix
value = variables[var_name]
filter_fn = _TEMPLATE_FILTERS.get(suffix)
return filter_fn(value) if filter_fn else str(value)
return str(variables.get(var_name, full_match))
pattern = r'\{([^}]+)\}'
return re.sub(pattern, replace_match, template)
def _page_subdir(subtype: Optional[str], type_path: str) -> Optional[str]:
"""Return the subtype-driven subdirectory under a type's `base_dir`, from
the type-spec's own `layout:` frontmatter. Returns None for types with no
`layout:` (flat directory). Falls back to `<subtype>s` for a subtype the
layout doesn't list, matching the previous hand-maintained behavior."""
if subtype is None:
return None
try:
layout = resolver.get_layout(type_path)
except ValueError:
layout = None
if layout is None:
return None
return layout.get(subtype, {}).get("dir", subtype + "s")
def _target_dir(type_path: str, frontmatter: Dict[str, Any]) -> Path:
"""Resolve where an instance of this type is written: `<root>/<base_dir>`,
plus a subtype subdirectory when the type declares a `layout:`.
`base_dir` is resolved against `config.KB_DIR` by default, so tests that
point KB_DIR at a temporary fixture wiki can never write into the real
`kb/`. A type-spec declaring `root: repo` resolves against `config.ROOT`
instead - for artifacts that are agent-directed material rather than
knowledge, and so live outside the knowledge layer."""
base_dir = resolver.get_base_dir(type_path)
if not base_dir:
fail(
f"Type {type_path} declares no `base_dir:` and cannot be "
f"instantiated as a page"
)
try:
root = resolver.get_root(type_path)
except ValueError as exc:
fail(str(exc))
target = (config.ROOT if root == "repo" else config.KB_DIR) / base_dir
subtype_field = resolver.get_subtype_field(type_path)
if subtype_field:
subdir = _page_subdir(frontmatter.get(subtype_field), type_path)
if subdir:
target = target / subdir
return target
def _validate_or_fail(frontmatter: Dict[str, Any], type_path: str, source_dir: Path) -> None:
"""Validate frontmatter against its type-spec schema, converting a
ValueError into the CLI's normal friendly-failure path instead of an
uncaught traceback. The schema's own error message (which already names
the offending field and, for enums, lists the valid values) is shown
as-is - there is no separate hand-maintained validity check to keep in
sync with it."""
try:
resolver.validate_frontmatter(frontmatter, type_path, source_dir)
except ValueError as exc:
fail(str(exc))
def _load_type_or_fail(type_path: str, source_dir: Path):
"""Resolve a type path and load its type-spec + template, converting an
unresolvable/invalid `--type` into the CLI's normal friendly-failure path."""
try:
return _resolve_type_and_get_template(type_path, source_dir)
except ValueError as exc:
fail(str(exc))
def new_page_command(
type_name: str = typer.Argument(
...,
help="Type name, e.g. entity|concept|source|comparison (see `wikitool types list`)",
),
name: str = typer.Option(..., "--name", help="Page title (a type's title_prefix is added automatically)"),
type_path_override: str = typer.Option(
"", "--type", help="Override the type-spec path (defaults to the one named by TYPE_NAME)"
),
set_fields: Optional[list[str]] = typer.Option(
None,
"--set",
help="Frontmatter field, repeatable: --set entity_type=tool --set tags=a,b. Array values split on commas (escape a literal one as \\,); repeating --set for an array field appends instead of replacing",
),
):
"""Scaffold a new wiki page of any type.
The type's own type-spec drives everything: which frontmatter fields
exist and are required (its `.schema.yaml`), their scaffold defaults
(schema `default:`), where the page is written (`base_dir` + `layout`),
what prefixes its title (`title_prefix`), and its body skeleton (the
type-spec's template). Adding a new type therefore needs no change here.
"""
type_path = type_path_override or resolver.find_type_by_name(type_name)
if not type_path:
available = sorted(fm.get("name") for _, fm in resolver.list_type_specs())
fail(f"No type-spec named '{type_name}'. Available: {', '.join(available)}")
today = datetime.date.today()
try:
title_prefix = resolver.get_title_prefix(type_path)
root = resolver.get_root(type_path)
except ValueError as exc:
fail(str(exc))
page_title = f"{title_prefix}{name}"
if root == "kb":
# Title collisions matter because wikilinks resolve by title alone, so
# two pages sharing a stem are indistinguishable to every link in the
# wiki. Artifacts outside kb/ are not addressed by title and are not
# part of that namespace, so the check does not apply to them.
check_collision(page_title)
schema = resolver.get_schema(type_path)
explicit = parse_set_fields(set_fields, schema)
declared = (schema or {}).get("properties", {})
if "summary" in declared:
explicit.setdefault("summary", _default_summary(""))
if "name" in declared:
# The CLI already has this value; a type that stores its own name in
# frontmatter should not have to be told it twice.
explicit.setdefault("name", name)
if "description" in declared:
explicit.setdefault("description", "TODO: add description")
frontmatter = _build_frontmatter(type_path, schema, today, explicit)
target_dir = _target_dir(type_path, frontmatter)
_type_spec, template = _load_type_or_fail(type_path, target_dir)
_validate_or_fail(frontmatter, type_path, target_dir)
if "raw_files" in frontmatter:
check_raw_files_exist(frontmatter["raw_files"])
path = target_dir / f"{page_title}.md"
body = _apply_template_variables(
template, {**frontmatter, "name": name, "today": today.isoformat()}
)
write_page(path, frontmatter, body)
success(f"Created {rel_path(path)}")
+377
View File
@@ -0,0 +1,377 @@
"""`wikitool rename` / `wikitool rm` - the two page mutations that had no command.
A page's title is the wiki's only identifier for it, so renaming or deleting a
page is never just a filesystem operation: the title appears in every other
page's body `[[wikilinks]]`, in the `[[Title]]` a `[^cite-id]` footnote
definition points at, and in page-reference frontmatter arrays (`related:`,
`sources:`, `entities:`, `concepts:`).
Doing this by hand is what left four pages citing
`Source - Docker Cheatsheet.md` when the page is
`Source - Docker Cheatsheet` - and because `lint`'s broken-link check
only walked page bodies, nothing ever reported it.
Which frontmatter fields hold page titles comes from each type-spec's
`page_ref_fields:`, so a new type needs no change here.
Neither command is atomic: both write one page at a time. Both are idempotent
per page, so a retry after a partial failure is safe.
"""
from __future__ import annotations
import re
from pathlib import Path
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import check_collision, fail, rel_path, success
from chemenu.frontmatter_io import write_page
from chemenu.page import Page
from chemenu.kb_scan import load_kb_pages
from chemenu.provenance import (
CITE_REF_RE,
cite_id,
cite_block_heading,
render_page_body,
split_cite_block,
unique_cite_id,
)
from chemenu.type_resolver import resolver
# `[[Target]]`, `[[Target|alias]]`, `[[Target#anchor]]` - including the
# `[[Target]]` inside a `[^cite-id]: [[Target]]` Footnotes definition, which
# is exactly what lets retarget_body() repoint a citation's link target on a
# rename. Group 1 is the target title; group 2 keeps any alias/anchor suffix
# untouched. The id itself is a separate concern - see retarget_cite_ids().
LINK_RE = re.compile(r"\[\[([^\[\]|#]+)((?:[|#][^\[\]]*)?)\]\]")
def page_ref_fields(page: Page) -> list[str]:
"""The page's declared page-title frontmatter fields, or [] if its type
can't be resolved (lint reports that separately)."""
type_path = page.frontmatter.get("type")
if not type_path:
return []
try:
return resolver.get_page_ref_fields(type_path, page.path)
except ValueError:
return []
def retarget_body(body: str, old: str, new: str) -> str:
"""Repoint every wikilink and citation marker aimed at `old` to `new`,
preserving any `|alias` or `#anchor` suffix."""
def replace(match: re.Match) -> str:
target, suffix = match.group(1), match.group(2)
if target.strip() != old:
return match.group(0)
return f"[[{new}{suffix}]]"
return LINK_RE.sub(replace, body)
def retarget_cite_ids(body: str, old: str, new: str) -> str:
"""After retarget_body() has already repointed a Footnotes definition's
`[[old]]` link target to `[[new]]`, also refresh a citation id that was
*derived* from `old`'s slug - `[^s-old-title]` -> `[^s-new-title]` - in
both the definition and every inline `[^id]` reference to it.
An id not derived from `old` (hand-picked, or a `-2`/`-3` collision
suffix from an unrelated pair) is left untouched; body is returned
unchanged if nothing needs renaming.
"""
head, definitions = split_cite_block(body)
if not definitions:
return body
renames: dict[str, str] = {}
new_definitions: dict[str, tuple[str, Optional[str]]] = {}
reserved = set(definitions)
for cid, (title, qualifier) in definitions.items():
if title == new and cid == cite_id(old, qualifier):
new_id = unique_cite_id(reserved - {cid}, new, qualifier)
renames[cid] = new_id
reserved.add(new_id)
new_definitions[new_id] = (title, qualifier)
else:
new_definitions[cid] = (title, qualifier)
if not renames:
return body
new_head = CITE_REF_RE.sub(lambda m: f"[^{renames.get(m.group(1), m.group(1))}]", head)
return render_page_body(new_head, new_definitions, cite_block_heading(body))
def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
"""Repoint `old` to `new` in every declared page-ref field. Returns True if
anything changed."""
changed = False
for field in page_ref_fields(page):
values = page.frontmatter.get(field)
if not values:
continue
updated = [new if value == old else value for value in values]
if updated != values:
page.frontmatter[field] = updated
changed = True
return changed
def _ref_fields_to_sweep(page: Page) -> list[str]:
"""Every field this page might hold a reference in.
The type's own declaration, plus any field already present on the page that
*some* type declares as a reference field. The second half exists because a
command can leave a reference in a field this type does not declare - `xref
add` wrote `related:` on source pages until 1.6.0 - and clearing exactly
that kind of leftover is what `xref remove` promises to be for. Sweeping
only declared fields made the state unreachable.
The extra names come from the type-specs rather than a constant here, so a
new reference field is swept without a code change.
"""
declared = page_ref_fields(page)
known = {
field
for _path, frontmatter in resolver.list_type_specs()
for field in (frontmatter.get("page_ref_fields") or [])
}
extra = [f for f in page.frontmatter if f not in declared and f in known]
return declared + extra
def strip_frontmatter_ref(page: Page, title: str) -> bool:
"""Drop `title` from every page-ref field. Returns True if anything changed.
An *undeclared* field that ends up empty is removed outright rather than
left as `field: []`: it was never valid for this type, and leaving the key
keeps the page failing schema validation for a reference that is gone.
"""
changed = False
declared = page_ref_fields(page)
for field in _ref_fields_to_sweep(page):
values = page.frontmatter.get(field)
if not values:
continue
updated = [value for value in values if value != title]
if updated == values:
continue
if not updated and field not in declared:
del page.frontmatter[field]
else:
page.frontmatter[field] = updated
changed = True
return changed
def strip_link_bullets(body: str, title: str) -> str:
"""Remove whole-line list bullets that exist only to point at `title` -
`- [[Title]]` (See Also) and `- **label:** [[Title]]` (Relationships).
Deliberately narrow: a bullet carrying prose alongside the link, and a
`[^cite-id]: [[Title]]` Footnotes definition line (which never starts
with `-`, so the pattern below cannot match it), are left alone. Removing
a citation is an editorial judgment about a claim, not a mechanical
de-linking.
"""
escaped = re.escape(title)
pattern = re.compile(
rf"^[ \t]*-[ \t]+(?:\*\*[^*\n]+:\*\*[ \t]+)?\[\[{escaped}\]\][ \t]*\n?",
re.MULTILINE,
)
return pattern.sub("", body)
def body_references(body: str, title: str) -> int:
"""How many wikilinks in `body` still point at `title`."""
return sum(1 for match in LINK_RE.finditer(body) if match.group(1).strip() == title)
def inbound_pages(pages: dict[str, Page], title: str) -> list[str]:
"""Every page (other than `title` itself) referencing it from its body or
from a declared page-ref frontmatter field."""
found = set()
for other_title, page in pages.items():
if other_title == title:
continue
if body_references(page.body, title):
found.add(other_title)
continue
if any(title in (page.frontmatter.get(f) or []) for f in page_ref_fields(page)):
found.add(other_title)
return sorted(found)
def rename_command(
old: str = typer.Option(..., "--from", help="Current page title, exactly as it appears"),
new: str = typer.Option(..., "--to", help="New page title"),
dry_run: bool = typer.Option(False, "--dry-run", help="List what would change instead of writing"),
):
"""Rename a page, or repoint references that name a page that never existed.
Two modes, chosen by whether `--from` is an actual page:
- `--from` is a page: it is renamed to `--to` (which must be free) and every
reference follows.
- `--from` is not a page but is referenced: references are repointed to
`--to`, which must already exist. This is the cleanup case - a reference
spelled `act_runner` when the page is `Act Runner`, or
`Source - X.md` when the page is `Source - X`. Nothing moves on disk.
"""
if old == new:
fail("--from and --to are the same title; nothing to rename.")
pages = load_kb_pages(config.KB_DIR)
target = pages.get(old)
references_only = target is None
if references_only:
if new not in pages:
fail(
f"Neither '{old}' nor '{new}' is a page under wiki/. Repointing references "
f"to '{new}' would just move the dangling reference; create the page first "
"with `wikitool new ...`, or drop the reference with `wikitool xref remove`."
)
elif not dry_run:
check_collision(new)
elif new in pages:
fail(f"A page titled '{new}' already exists at {rel_path(pages[new].path)}")
touched: list[str] = []
failed: list[str] = []
for title, page in sorted(pages.items()):
new_body = retarget_body(page.body, old, new)
new_body = retarget_cite_ids(new_body, old, new)
if title == old and page.h1_title == old:
new_body = re.sub(rf"^# {re.escape(old)}$", f"# {new}", new_body, count=1, flags=re.MULTILINE)
frontmatter_changed = retarget_frontmatter(page, old, new)
if new_body == page.body and not frontmatter_changed:
continue
touched.append(title)
if not dry_run:
try:
write_page(page.path, page.frontmatter, new_body)
except OSError as exc:
failed.append(f"{title} ({exc})")
if failed:
fail(
f"Updated references in {len(touched) - len(failed)}/{len(touched)} page(s) before a write "
f"failed: {', '.join(failed)}. Nothing was renamed on disk, so '{old}' is unchanged - check "
"`git status`, resolve the write failure (permissions/disk), then re-run the full `rename` "
"command (safe to retry - each page's rewrite is idempotent)."
)
for title in touched:
typer.echo(f" updated references in '{title}'")
if references_only:
if not touched:
success(f"Nothing references '{old}'; nothing to repoint.")
return
if dry_run:
typer.echo(f"[dry-run] would repoint {len(touched)} page(s) to '{new}'. No files written.")
return
success(
f"Repointed references from '{old}' to the existing page '{new}' in "
f"{len(touched)} page(s). No file was moved ('{old}' was not a page)."
)
return
new_path = target.path.parent / f"{new}.md"
if dry_run:
typer.echo(f"[dry-run] would rename {rel_path(target.path)} -> {rel_path(new_path)}")
typer.echo(f"[dry-run] would update {len(touched)} page(s). No files written.")
return
target.path.rename(new_path)
success(
f"Renamed '{old}' -> '{new}' ({rel_path(new_path)}); "
f"updated references in {len(touched)} page(s). "
"Run `wikitool index rebuild` and `wikitool sources rebuild-index` next."
)
def rm_command(
page_title: str = typer.Option(..., "--page", help="Exact title of the page to delete"),
yes: bool = typer.Option(
False,
"--yes",
"-y",
help="Confirm deletion of a page that other pages still reference. Only pass this after "
"a human has reviewed the inbound list - never set it automatically.",
),
dry_run: bool = typer.Option(False, "--dry-run", help="List what would change instead of writing"),
):
"""Delete a page and mechanically de-link it from the rest of the wiki."""
pages = load_kb_pages(config.KB_DIR)
target = pages.get(page_title)
if target is None:
fail(f"No page titled '{page_title}' found under wiki/.")
inbound = inbound_pages(pages, page_title)
if inbound and not yes:
listed = "\n".join(f"- {t}" for t in inbound)
fail(
f"'{page_title}' is still referenced by {len(inbound)} page(s). Deleting it will leave "
"their prose pointing at nothing. Show the user this list and only re-run with --yes "
f"once they have approved:\n{listed}"
)
touched: list[str] = []
failed: list[str] = []
leftover: list[tuple[str, int]] = []
for title, page in sorted(pages.items()):
if title == page_title:
continue
new_body = strip_link_bullets(page.body, page_title)
frontmatter_changed = strip_frontmatter_ref(page, page_title)
if new_body != page.body or frontmatter_changed:
touched.append(title)
if not dry_run:
try:
write_page(page.path, page.frontmatter, new_body)
except OSError as exc:
failed.append(f"{title} ({exc})")
remaining = body_references(new_body, page_title)
if remaining:
leftover.append((title, remaining))
if failed:
fail(
f"De-linked {len(touched) - len(failed)}/{len(touched)} page(s) before a write failed: "
f"{', '.join(failed)}. '{page_title}' was NOT deleted, so nothing is orphaned - check "
"`git status`, resolve the write failure, then re-run `rm` (safe to retry)."
)
for title in touched:
typer.echo(f" de-linked '{title}'")
if dry_run:
typer.echo(f"[dry-run] would delete {rel_path(target.path)}")
typer.echo(f"[dry-run] would update {len(touched)} page(s). No files written.")
else:
target.path.unlink()
if leftover:
typer.echo("")
typer.echo(
"Prose references left in place - these carry claims, so removing them is an "
"editorial call, not a mechanical one:"
)
for title, count in leftover:
typer.echo(f" - {title}: {count} remaining [[{page_title}]] reference(s)")
typer.echo("Fix them, then re-run `wikitool lint`.")
if dry_run:
return
success(
f"Deleted '{page_title}' ({rel_path(target.path)}); de-linked {len(touched)} page(s). "
"Run `wikitool index rebuild` and `wikitool sources rebuild-index` next."
)
+184
View File
@@ -0,0 +1,184 @@
"""`wikitool sources ...` - raw-file <-> wiki provenance tooling.
This is the deterministic backbone for citation backtracing: it never guesses
which raw file backs a claim, it only reports what the frontmatter and inline
`[^cite-id]` footnotes already declare. Filling in those declarations
correctly is still the LLM's job.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import fail, rel_path, success
from chemenu.provenance import (
broken_raw_refs,
citing_pages,
legacy_source_pages,
page_raw_files,
source_pages_by_raw_file,
source_raw_files,
uncovered_raw_files,
)
from chemenu.kb_scan import load_kb_pages
app = typer.Typer(help="Trace and lint raw-file <-> wiki-page provenance.")
def _normalize_raw_path(raw: str) -> str:
"""Accept an absolute path or a path relative to the repo root or to raw/,
and return it as a path relative to the repo root (matching how it is
stored in `raw_files:`)."""
candidate = Path(raw)
if candidate.is_absolute():
try:
return str(candidate.relative_to(config.ROOT))
except ValueError:
return str(candidate)
if candidate.exists():
return str(candidate)
if (config.ROOT / candidate).exists():
return str(candidate)
if (config.RAW_DIR / candidate).exists():
return str((Path("raw") / candidate))
return raw
@app.command("coverage")
def coverage(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
"""Report raw files with no source page, broken raw_files: references, and
source pages still using a legacy directory/URL-only `source:` field."""
pages = load_kb_pages(config.KB_DIR)
report = {
"uncovered_raw_files": uncovered_raw_files(config.RAW_DIR, pages),
"broken_raw_refs": broken_raw_refs(pages),
"legacy_source_pages": legacy_source_pages(pages),
}
if json_out:
typer.echo(json.dumps(report, indent=2))
return
typer.echo(f"Uncovered raw files: {len(report['uncovered_raw_files'])}")
for f in report["uncovered_raw_files"]:
typer.echo(f" - {f}")
typer.echo(f"Broken raw_files references: {len(report['broken_raw_refs'])}")
for item in report["broken_raw_refs"]:
typer.echo(f" - [[{item['page']}]] -> {item['raw_path']}")
typer.echo(f"Legacy (directory/URL-only) source pages: {len(report['legacy_source_pages'])}")
for item in report["legacy_source_pages"]:
typer.echo(f" - [[{item['page']}]] ({item['reason']}): {item['source']}")
@app.command("trace")
def trace(
raw: Optional[str] = typer.Option(None, "--raw", help="Raw file path to trace forward from"),
page: Optional[str] = typer.Option(None, "--page", help="Wiki page title to trace backward from"),
):
"""Trace provenance in either direction: --raw shows which source pages
cover a raw file and which wiki pages cite it; --page shows which sources
and raw files back a given wiki page."""
if bool(raw) == bool(page):
fail("Provide exactly one of --raw or --page")
pages = load_kb_pages(config.KB_DIR)
if raw:
raw_key = _normalize_raw_path(raw)
by_raw = source_pages_by_raw_file(pages)
source_titles = by_raw.get(raw_key, [])
if not source_titles:
typer.echo(f"No source page covers {raw_key}")
raise typer.Exit(code=1)
for source_title in source_titles:
typer.echo(f"{raw_key}")
typer.echo(f" covered by: [[{source_title}]]")
citers = citing_pages(pages, source_title)
if citers:
for c in citers:
typer.echo(f" cited by: [[{c}]]")
else:
typer.echo(" cited by: (nothing yet)")
return
target_page = pages.get(page)
if target_page is None:
fail(f"No page titled '{page}' found")
sources = target_page.frontmatter.get("sources") or []
typer.echo(f"[[{page}]]")
if not sources:
typer.echo(" sources: (none listed)")
for source_title in sources:
typer.echo(f" sources: [[{source_title}]]")
source_page = pages.get(source_title)
if source_page is None:
typer.echo(" (source page not found)")
continue
for raw_path in source_raw_files(source_page):
typer.echo(f" raw file: {raw_path}")
raw_files = page_raw_files(pages, target_page)
typer.echo(f" all raw files (incl. inline citations): {raw_files or '(none)'}")
def build_provenance_index(kb_dir: Path, raw_dir: Path) -> str:
pages = load_kb_pages(kb_dir)
by_raw = source_pages_by_raw_file(pages)
all_raw = sorted(str(p.relative_to(config.ROOT)) for p in config.iter_raw_files(raw_dir))
uncovered = uncovered_raw_files(raw_dir, pages)
lines: list[str] = []
lines.append("# Provenance Index")
lines.append("")
lines.append("Generated by `tools/wikitool sources rebuild-index`. Do not hand-edit.")
lines.append("")
lines.append("Maps every raw source file to the wiki source page(s) that cover it, and")
lines.append("every wiki page that cites that source (via frontmatter `sources:` or an")
lines.append("inline `[^cite-id]` footnote).")
lines.append("")
lines.append("## Coverage Summary")
lines.append("")
lines.append(f"- **Total raw files:** {len(all_raw)}")
lines.append(f"- **Covered:** {len(all_raw) - len(uncovered)}")
lines.append(f"- **Uncovered:** {len(uncovered)}")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## Raw Files")
lines.append("")
for raw_path in all_raw:
lines.append(f"### `{raw_path}`")
lines.append("")
source_titles = by_raw.get(raw_path, [])
if not source_titles:
lines.append("No source page covers this file yet.")
lines.append("")
continue
for source_title in source_titles:
lines.append(f"- Covered by: [[{source_title}]]")
citers = citing_pages(pages, source_title)
if citers:
lines.append(f" - Cited by: {', '.join(f'[[{c}]]' for c in citers)}")
else:
lines.append(" - Cited by: (nothing yet)")
lines.append("")
return "\n".join(lines) + "\n"
@app.command("rebuild-index")
def rebuild_index(
dry_run: bool = typer.Option(False, "--dry-run", help="Print the result instead of writing wiki/provenance.md"),
):
content = build_provenance_index(config.KB_DIR, config.RAW_DIR)
provenance_file = config.KB_DIR / "provenance.md"
if dry_run:
# nl=False so the preview is byte-identical to the file that would be
# written; see the same note in index_build.py.
typer.echo(content, nl=False)
return
provenance_file.write_text(content, encoding="utf-8")
success(f"Rebuilt {rel_path(provenance_file)}")
+355
View File
@@ -0,0 +1,355 @@
"""Iteration/cost budget gate: a hard, code-enforced cap on how many wikitool
commands a single agent session may run before requiring explicit human
confirmation, plus a loop-breaker that trips immediately if the last few
calls are near-identical (same command + same arguments).
This closes the gap documented in AGENTS.md's "Gates" section: unlike a
prompt instruction ("stop after N steps"), this check runs
in-process on every `wikitool` invocation and cannot be skipped by the
calling agent "politely trying again". It mirrors the Mass-Update Gate
pattern (see git_publish.py / wiki/concepts/Mass-Update Gate.md), but that
gate is scoped to the *size* of a single publish, while this one is scoped to
*iteration volume* across a whole session (e.g. a wiki-ingest or wiki-lint run
that could otherwise loop unbounded over many entity/concept pages).
Session scoping: a "session" is approximated by the parent process of this
CLI invocation (the agent's shell), via the WIKITOOL_SESSION_ID env var if the
caller sets one, otherwise os.getppid(). A new terminal/session therefore
starts with a fresh budget.
"""
from __future__ import annotations
import json
import os
import time
from contextlib import contextmanager
from pathlib import Path
import typer
from chemenu import config
from chemenu.commands._util import fail, success
from chemenu.session import session_id as _shared_session_id
from chemenu.session import session_id_source as _shared_session_id_source
from chemenu.telemetry import emit
app = typer.Typer(help="Session iteration/cost budget gate (see the tooling contract's 'Iteration and Cost Limits').")
STATE_DIR = config.ROOT / "tools" / ".wikitool_session"
STATE_FILE = STATE_DIR / "budget.json"
LOCK_FILE = STATE_DIR / "budget.lock"
# Calibration, measured in this instance rather than inherited: ~5-15 calls for
# a simple task, ~20-35 for a complex multi-tool workflow such as an ingest.
#
# The upper band used to read 15-25, taken from an industry rule of thumb (see
# kb/concepts/Iteration and Cost Limits.md, which still cites it as such). Four
# consecutive real ingests measured 24, 26, 29 and 30 calls - every one of them
# at or above the old band's ceiling while doing nothing unusual. A guideline
# that the normal case exceeds is not a guideline; it teaches an agent that the
# numbers are decorative.
#
# The limit sits well above the band on purpose. It is not a target but the
# point past which a session is presumed stuck. At 30 the ingest of 2026-08-30
# hit it on overhead alone - reading a report back, a corrected retry, checking
# the tree before publishing - which is the gate firing on the tool rather than
# on the task.
DEFAULT_CALL_LIMIT = 60
# Loop-breaker: abort if the last N calls all share the same command + args,
# even if the overall call limit hasn't been reached yet.
DEFAULT_LOOP_WINDOW = 3
# Sessions untouched for this long are dropped on the next write. Without this
# the state file grows one entry per session forever - and the getppid()
# fallback makes new keys cheap (a new shell is a new session).
SESSION_TTL_SECONDS = 7 * 24 * 3600
# Never gate the gate's *read* side, or reporting the situation to the user
# would become impossible exactly when the limit trips. `budget reset` is
# deliberately NOT exempt: it clears the counter, so exempting it would make
# the whole gate a formality an agent could step around by resetting first.
# It is gated on `--yes` instead, the same way `publish` is.
#
# `eval score` and `eval sessions` read a trace and re-run lint's checks in
# process. Reading back what a session already did is not iteration on the wiki,
# and charging for it would discourage checking one's own work.
# `version show`/`check`/`notes` only read - `VERSION`, the release stamp, the
# changelog, or a remote release feed. `version bump` writes two files and
# stays counted like every other mutation.
SKIP_COMMAND_PATHS = {
("budget", "status"),
("eval", "score"),
("eval", "sessions"),
("cite", "id"),
("version", "show"),
("version", "check"),
("version", "notes"),
# Bare `wikitool version` (and `version --json`) is an alias for `show`;
# the subcommand slot is empty, so it needs its own entry to be exempt
# alongside the command it delegates to.
("version", ""),
# `migrate list/status/verify` only read - the migration documents, the KB
# state file, and git history. `verify` especially: a migration runs it
# once per unit by design, and charging for the check would push an agent
# toward skipping the one step that catches a dropped reference.
# `migrate done`/`baseline` write the state file and stay counted.
("migrate", "list"),
("migrate", "status"),
("migrate", "verify"),
}
# Commands exempt regardless of their first argument, because that argument is
# a query rather than a subcommand. `search` is here because retrieval is
# reading, not iterating: the budget exists to stop an agent looping over the
# wiki's *state*, and charging for a search would penalise the one habit that
# lowers cost - looking before reading. `doctor` is here for the same reason:
# it only reads and reports, never mutates anything. Every command that
# mutates anything stays counted.
SKIP_COMMANDS = {"search", "doctor"}
def is_exempt(command: str, args: list[str]) -> bool:
"""Whether this invocation is outside the budget entirely."""
if command in SKIP_COMMANDS:
return True
subcommand = args[0] if args and not args[0].startswith("-") else ""
return (command, subcommand) in SKIP_COMMAND_PATHS
def _session_id() -> str:
return _shared_session_id()
def _session_id_source() -> str:
return _shared_session_id_source()
def _load_state() -> dict:
if not STATE_FILE.exists():
return {}
try:
return json.loads(STATE_FILE.read_text())
except (json.JSONDecodeError, OSError):
return {}
def prune_state(state: dict, now: float, ttl: float = SESSION_TTL_SECONDS) -> dict:
"""Drop sessions whose last recorded call is older than the TTL. Entries
written before `last_seen` existed are kept (they get a timestamp on their
next recorded call)."""
return {
session_id: entry
for session_id, entry in state.items()
if "last_seen" not in entry or now - entry["last_seen"] <= ttl
}
def _save_state(state: dict) -> None:
"""Write the state atomically: build the payload, write it to a sibling
temp file, then rename it over the real file. A crash or concurrent read
mid-write can never observe a truncated/partial JSON file this way -
os.replace() is atomic on POSIX."""
STATE_DIR.mkdir(parents=True, exist_ok=True)
payload = json.dumps(prune_state(state, time.time()), indent=2)
tmp_file = STATE_FILE.with_suffix(STATE_FILE.suffix + ".tmp")
tmp_file.write_text(payload)
os.replace(tmp_file, STATE_FILE)
@contextmanager
def _state_lock():
"""Exclusive cross-process lock guarding the budget state's load-modify-
save cycle. Without this, two `wikitool` calls racing in the same session
(e.g. two parallel subagents) can both load count=N, both compute N+1, and
both save - losing an increment and letting the session run past the gate
it exists to enforce. POSIX-only (fcntl); best-effort no-op if unavailable,
since the loop-breaker's identical-call check still degrades gracefully."""
STATE_DIR.mkdir(parents=True, exist_ok=True)
try:
import fcntl
except ImportError: # pragma: no cover - non-POSIX platform
yield
return
with open(LOCK_FILE, "w") as lock_fh:
fcntl.flock(lock_fh, fcntl.LOCK_EX)
try:
yield
finally:
fcntl.flock(lock_fh, fcntl.LOCK_UN)
def loop_breaker_message(call_signature: str, loop_window: int) -> str:
return (
f"Loop-Breaker: the last {loop_window} wikitool calls in this session were "
f"identical ('{call_signature}'). This usually means the agent is stuck retrying "
"the same failing operation instead of changing approach - per the tooling contract's "
"Tool Error Contracts, that is exactly the case for stopping and escalating rather than "
"retrying again. Stop, explain the situation to the user, and get explicit direction "
"before continuing. Only re-run with --override-budget once the user has confirmed "
"the repeat is intentional - never add it on the agent's own initiative."
)
def call_limit_message(count: int, call_limit: int) -> str:
return (
f"Iteration Budget Gate: this session has made {count} wikitool calls, exceeding the "
f"limit of {call_limit}. Per the tooling contract's 'Iteration and Cost Limits' section, a "
"single task should typically need roughly 5-15 calls (simple) or 20-35 (complex multi-tool "
"workflow like wiki-ingest/wiki-lint). This far past that band is a documented sign of poor "
"task decomposition or a stuck loop. Stop, summarize progress and the blocker to the "
"user, and get explicit direction before continuing. Only re-run with --override-budget "
"once the user has approved continuing this session - never add it unprompted."
)
def record_and_check(
command: str,
args: list[str],
override: bool,
call_limit: int = DEFAULT_CALL_LIMIT,
loop_window: int = DEFAULT_LOOP_WINDOW,
) -> bool:
"""Record this invocation against the session budget and enforce the gate.
Called once per process from main() before Typer dispatches to a
subcommand, so every wikitool command is covered uniformly.
A refused call is *not* recorded: it never ran, so counting it would keep
inflating the number quoted back to the user on every subsequent attempt.
The loop-breaker still trips on the next identical call, because the
history that made it identical is already stored.
Returns whether a slot was actually charged, so the caller knows whether
there is anything to hand back via `refund()`.
"""
if not command or is_exempt(command, args):
return False
with _state_lock():
session_id = _session_id()
state = _load_state()
entry = state.setdefault(session_id, {"count": 0, "recent": []})
recent = entry["recent"]
call_signature = f"{command} {' '.join(args)}".strip()
# Checked against history *before* this call is appended, so it answers
# "were the last `loop_window` calls already identical to this one?".
is_repeat_of_recent = (
len(recent) >= loop_window
and all(c == call_signature for c in recent[-loop_window:])
)
if not override:
if is_repeat_of_recent:
emit(
"wikitool",
"gate.refused",
{
"gate": "loop-breaker",
"command": command,
"args": args,
"call_signature": call_signature,
"loop_window": loop_window,
"count": entry["count"],
},
)
fail(loop_breaker_message(call_signature, loop_window))
if entry["count"] + 1 > call_limit:
emit(
"wikitool",
"gate.refused",
{
"gate": "iteration-budget",
"command": command,
"args": args,
"count": entry["count"] + 1,
"limit": call_limit,
},
)
fail(call_limit_message(entry["count"] + 1, call_limit))
entry["count"] += 1
recent.append(call_signature)
entry["recent"] = recent[-max(loop_window, 10):]
entry["last_seen"] = time.time()
_save_state(state)
return True
def refund() -> None:
"""Give the current session its last charged slot back.
Called when the command declined instead of acting: a rejected argument,
or a read-only check reporting findings (`_util.fail`, exit 1). The
tooling contract answers a rejected argument with "fix it and retry once",
so charging for the rejection makes the prescribed response cost two slots
for one operation - and the budget exists to bound iteration on the wiki,
which a call that changed nothing did not do.
The call stays in `recent`. Repeating the same broken invocation is a real
failure, and the loop-breaker is the instrument for it: it needs the
history, not the counter.
"""
with _state_lock():
state = _load_state()
entry = state.get(_session_id())
if not entry or entry.get("count", 0) <= 0:
return
entry["count"] -= 1
_save_state(state)
def status_command():
"""Show the current session's call count and recent command history."""
state = _load_state()
entry = state.get(_session_id())
typer.echo(f"Session: {_session_id()} (from {_session_id_source()})")
if not entry:
success("No recorded calls yet for this session.")
return
typer.echo(f"Calls so far: {entry['count']} (limit {DEFAULT_CALL_LIMIT})")
typer.echo("Recent calls:")
for c in entry["recent"]:
typer.echo(f" - {c}")
def reset_message() -> str:
return (
"`budget reset` clears the Iteration Budget Gate - the check that exists to stop a "
"session looping or sprawling unnoticed. Resetting it on the agent's own initiative "
"would make the gate advisory, which is exactly what it was built not to be. Stop, "
"summarize what the session has done so far and why it needs more calls, and only "
"re-run with --yes once the user has approved continuing."
)
def reset_command(
all_sessions: bool = typer.Option(
False, "--all", help="Clear every session's budget, not just the current one."
),
yes: bool = typer.Option(
False,
"--yes",
"-y",
help="Confirm clearing the budget. Only pass this after a human has approved "
"continuing the session - never set it automatically to work around the gate.",
),
):
"""Clear the current session's (or all sessions') recorded budget."""
if not yes:
fail(reset_message())
with _state_lock():
if all_sessions:
if STATE_FILE.exists():
STATE_FILE.unlink()
success("Cleared budget state for all sessions.")
return
state = _load_state()
if state.pop(_session_id(), None) is not None:
_save_state(state)
success(f"Cleared budget state for session {_session_id()}.")
app.command("status")(status_command)
app.command("reset")(reset_command)
+221
View File
@@ -0,0 +1,221 @@
"""`wikitool search` - find pages without reading `kb/index.md`.
This command exists to make retrieval cheap. Before it, the documented way to
find a page was to read the whole generated index; at a few hundred pages that
is tens of thousands of tokens spent to learn three filenames. A search returns
the same pointers for a fraction of it.
Two halves, deliberately kept separate:
- Text search is answered by a pluggable backend (`rg` today) - see
`chemenu/search/`.
- Frontmatter predicates (`--field`) are evaluated here, in-process, on the
structured YAML rather than on its rendering. With no text at all this is a
pure structured query, which is how "systems below 0.6 confidence, oldest
first" is asked without a second command.
Scope is `kb/` only. `instructions/` is discovered through
`wikitool instructions list`, because a procedure is found by what it is *for*
(its description), not by keywords in its body.
"""
from __future__ import annotations
import json
from pathlib import Path
import typer
from chemenu import config
from chemenu.commands._util import fail, today_iso
from chemenu.frontmatter_io import read_page
from chemenu.kb_scan import iter_kb_pages
from chemenu.page import Page
from chemenu.search import filters
from chemenu.search.base import page_key
from chemenu.search.filters import PredicateError
from chemenu.search.fuse import reciprocal_rank_fusion
from chemenu.search.registry import UnknownBackend, resolve
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing, build_hit
from chemenu.search.types import Predicate, SearchHit, SearchQuery
TITLE_WIDTH = 34
SUMMARY_WIDTH = 84
def load_pages_by_path(kb_dir: Path | None = None, root: Path | None = None) -> dict[str, Page]:
"""Every page under `kb/`, keyed by repo-relative path.
Path-keyed rather than title-keyed on purpose: `load_kb_pages()` drops one
of two pages sharing a stem, and search should still find both - a
duplicate title is a lint finding, not a reason to hide a page.
"""
kb_dir = kb_dir or config.KB_DIR
root = root or config.ROOT
pages: dict[str, Page] = {}
for path in iter_kb_pages(kb_dir):
frontmatter, body = read_page(path)
pages[page_key(path, root)] = Page(path=path, frontmatter=frontmatter, body=body)
return pages
def _sort_key(hit: SearchHit, field: str):
value = hit.as_dict().get(field)
if value is None:
# Missing values sort last in either direction rather than crashing on
# a None comparison.
return (1, "")
if isinstance(value, (int, float)):
return (0, value)
return (0, str(value).lower())
def sort_hits(hits: list[SearchHit], sort: str | None) -> list[SearchHit]:
"""Sort by a hit field. A leading `-` reverses, e.g. `--sort -confidence`."""
if not sort:
return hits
descending = sort.startswith("-")
field = sort.lstrip("-")
ordered = sorted(hits, key=lambda h: _sort_key(h, field), reverse=descending)
return ordered
def run_search(
query: SearchQuery,
pages: dict[str, Page],
backends: list,
kb_dir: Path | None = None,
) -> list[SearchHit]:
"""Answer a query. Pure: no I/O beyond whatever a backend does."""
filters.validate_fields(query.predicates, pages)
if query.text:
rankings = [backend.search(query, pages) for backend in backends]
hits = rankings[0] if len(rankings) == 1 else reciprocal_rank_fusion(rankings)
allowed = filters.apply_predicates(pages, query.predicates, kb_dir)
hits = [hit for hit in hits if hit.path in allowed]
else:
selected = filters.apply_predicates(pages, query.predicates, kb_dir)
hits = [
build_hit(page, key, [], query, backend="frontmatter", kb_dir=kb_dir)
for key, page in selected.items()
]
hits.sort(key=lambda h: h.title.lower())
hits = sort_hits(hits, query.sort)
return hits[: query.limit] if query.limit else hits
def _truncate(text: str, width: int) -> str:
text = " ".join(text.split())
return text if len(text) <= width else text[: width - 1] + "\u2026"
def render_table(hits: list[SearchHit], show_matches: bool) -> str:
if not hits:
return "No matches."
lines = []
for hit in hits:
kind = hit.kind or "?"
if hit.subtype:
kind = f"{kind}/{hit.subtype}"
lines.append(
f"{hit.score:6.1f} {_truncate(hit.title, TITLE_WIDTH):<{TITLE_WIDTH}} "
f"{kind:<18} {_truncate(hit.summary, SUMMARY_WIDTH)}"
)
if show_matches:
for match in hit.matches:
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
lines.append("")
lines.append(f"{len(hits)} result(s).")
return "\n".join(lines)
def search_command(
text: str = typer.Argument(
None,
help="Text to search for. Omit it to run a pure frontmatter query.",
),
field: list[str] = typer.Option(
None,
"--field",
"-f",
help="Frontmatter predicate, repeatable (AND). Forms: field=value, "
"field~substring, 'field>=value', 'field:*' (present), '!field' (absent).",
),
kind: str = typer.Option(None, "--kind", help="Shorthand for --field kind=<value>."),
subtype: str = typer.Option(None, "--subtype", help="Shorthand for --field subtype=<value>."),
collection: str = typer.Option(
None, "--collection", help="Shorthand for --field collection=<value>."
),
tag: str = typer.Option(None, "--tag", help="Shorthand for --field tags=<value>."),
regex: bool = typer.Option(
False, "--regex", help="Treat the query as a regex. Off by default: terms are literal."
),
limit: int = typer.Option(20, "--limit", help="Maximum number of results. 0 for no limit."),
sort: str = typer.Option(
None, "--sort", help="Sort by a result field; prefix with '-' to reverse, e.g. -confidence."
),
backend: str = typer.Option(
None,
"--backend",
help="Search backend(s), comma-separated. Default 'rg' (or $WIKITOOL_SEARCH_BACKEND).",
),
show_matches: bool = typer.Option(
False, "--matches", help="Print the matching lines under each result."
),
json_out: bool = typer.Option(False, "--json", help="Print the results as JSON."),
):
"""Search kb/ by text, by frontmatter, or by both."""
raw_predicates = list(field or [])
for value, name in ((kind, "kind"), (subtype, "subtype"), (collection, "collection")):
if value:
raw_predicates.append(f"{name}={value}")
if tag:
raw_predicates.append(f"tags={tag}")
if not text and not raw_predicates:
fail("Nothing to search for: give a query, or at least one --field predicate.")
try:
predicates: tuple[Predicate, ...] = tuple(
filters.parse_predicate(raw) for raw in raw_predicates
)
except PredicateError as exc:
fail(str(exc))
try:
backends = resolve(backend)
except UnknownBackend as exc:
fail(str(exc))
query = SearchQuery(
text=text,
predicates=predicates,
regex=regex,
limit=limit,
sort=sort,
)
pages = load_pages_by_path()
try:
hits = run_search(query, pages, backends)
except PredicateError as exc:
fail(str(exc))
except RipgrepMissing as exc:
fail(str(exc))
except RipgrepFailed as exc:
fail(str(exc))
if json_out:
payload = {
"generated": today_iso(),
"query": text,
"predicates": [p.render() for p in predicates],
"backend": ",".join(b.name for b in backends),
"count": len(hits),
"results": [hit.as_dict() for hit in hits],
}
typer.echo(json.dumps(payload, indent=2))
return
typer.echo(render_table(hits, show_matches))
+319
View File
@@ -0,0 +1,319 @@
"""`wikitool touch` - update the self-describing frontmatter fields of a page.
`modified:`, `summary:`, `provenance:` and `confidence_base:` describe the page
itself rather than its relationships, so they were the one part of frontmatter
the skills still told the LLM to edit by hand - a carve-out in the otherwise
absolute "never hand-write frontmatter" rule. Bumping a date and rewriting a
one-line summary are mechanical, so they belong here: the field name is chosen
from the type's own schema (`modified` for entity/concept, `date` for source),
and the result is schema-validated before it is written.
Only `modified:` is bumped automatically. A source's `date:` is the publication
date of the material itself, not a record of when we last edited the page, so it
changes only on an explicit `--date`.
"""
from __future__ import annotations
import datetime
from typing import Any, Dict, Optional
import typer
# Hard, non-optional dependency - see type_resolver.py's import comment.
from jsonschema import Draft202012Validator, FormatChecker
from chemenu import config
from chemenu.commands._util import (
check_raw_files_exist,
fail,
parse_set_fields,
rel_path,
success,
)
from chemenu.frontmatter_io import normalize_dates
from chemenu.frontmatter_io import write_page
from chemenu.kb_scan import load_kb_pages
from chemenu.type_resolver import resolver
# Ordered by preference: whichever the page's schema declares is the one that
# records "when was this page's content last confirmed?".
DATE_FIELDS = ("modified", "date")
# Fields `--set` refuses, each with the command that owns it instead. This is a
# denylist rather than an allowlist on purpose: an allowlist is a second copy of
# the schema, and the copy is the one that drifts - a field added to a type-spec
# would silently stay unwritable until someone remembered to widen the list.
# Everything the schema declares is settable unless there is a reason here.
UNSETTABLE = {
"type": (
"changing it changes the page's schema *and* the directory it belongs in - "
"see instructions/page-lifecycle.md"
),
"confidence": (
"derived, not authored: set `--confidence-base` and run "
"`wikitool confidence decay --apply` to recompute it"
),
"related": "page-reference field - use `wikitool xref add` / `xref remove`",
"sources": (
"page-reference field - written from the other side by "
"`wikitool xref link-source`, or cleared with `xref remove`"
),
"entities": (
"page-reference field - use `wikitool xref link-source --source <this page> "
"--entities <titles>`, which writes both directions; `xref remove` clears one"
),
"concepts": (
"page-reference field - use `wikitool xref link-source --source <this page> "
"--entities <titles>`, which writes both directions; `xref remove` clears one"
),
}
def _date_field(schema: Optional[Dict[str, Any]], frontmatter: Dict[str, Any]) -> Optional[str]:
properties = (schema or {}).get("properties", {})
for field in DATE_FIELDS:
if field in properties or field in frontmatter:
return field
return None
def _parse_date(text: str) -> datetime.date:
"""`--date` as a real date, or a refusal naming the expected shape."""
try:
return datetime.date.fromisoformat(text)
except ValueError:
fail(f"--date must be YYYY-MM-DD, got '{text}'.")
def validate_fields(
frontmatter: Dict[str, Any], schema: Optional[Dict[str, Any]], fields: set[str]
) -> Optional[str]:
"""Validate only the fields this command is writing.
Whole-document validation would refuse to bump `modified:` on a page that
is invalid for some unrelated, pre-existing reason - which is exactly the
page most in need of maintenance. Errors whose path points outside the
touched fields (missing required fields elsewhere, legacy extra keys) are
left for `wikitool lint` to report.
"""
if schema is None:
return None
validator = Draft202012Validator(schema, format_checker=FormatChecker())
messages = [
error.message
for error in validator.iter_errors(normalize_dates(frontmatter))
if error.path and error.path[0] in fields
]
return "; ".join(messages) if messages else None
def _settable_or_fail(field: str, schema: Optional[Dict[str, Any]], type_path: str) -> Dict[str, Any]:
"""Refuse a field this command must not write, and return its subschema.
Two refusals, deliberately worded differently. A field on `UNSETTABLE` is
writable in principle but belongs to another command, so the message names
that command. A field the schema does not declare is not a routing problem
but a typo or a wrong page type, so the message lists what this page
actually has - the value is knowing that `tag` should have been `tags`.
"""
if field in UNSETTABLE:
fail(f"`{field}` cannot be set with --set: {UNSETTABLE[field]}")
properties = (schema or {}).get("properties", {})
if field not in properties:
settable = sorted(set(properties) - set(UNSETTABLE))
fail(
f"Type {type_path} declares no field '{field}'.\n"
f" Settable fields for this page: {', '.join(settable) or '(none)'}"
)
return properties[field]
def _apply_set(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
if frontmatter.get(field) == value:
return None
before = frontmatter.get(field)
frontmatter[field] = value
return f"{field}: {before!r} -> {value!r}"
def _apply_add(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
"""Append list elements not already present, preserving order."""
if not isinstance(value, list):
fail(f"--add works on array fields only; '{field}' is not one. Use --set.")
current = list(frontmatter.get(field) or [])
added = [item for item in value if item not in current]
if not added:
return None
frontmatter[field] = current + added
return f"{field}: added {', '.join(repr(i) for i in added)}"
def _apply_remove(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
"""Drop list elements, reporting the ones that were not there.
Removing something absent succeeds rather than failing - `xref remove` is
idempotent for the same reason, and a repair command that refuses to run
twice is a repair command nobody dares script. But it is *reported*: a
silent no-op is how a mistyped element name looks exactly like a successful
removal.
"""
if not isinstance(value, list):
fail(f"--remove works on array fields only; '{field}' is not one. Use --set.")
current = list(frontmatter.get(field) or [])
present = [item for item in value if item in current]
absent = [item for item in value if item not in current]
if absent:
typer.echo(f" {field}: not present, nothing removed: {', '.join(repr(i) for i in absent)}")
if not present:
return None
frontmatter[field] = [item for item in current if item not in present]
return f"{field}: removed {', '.join(repr(i) for i in present)}"
def touch_command(
page_title: str = typer.Option(..., "--page", help="Exact page title, e.g. 'Docker Cheatsheet'"),
summary: Optional[str] = typer.Option(None, "--summary", help="Replace the page's 1-line summary"),
provenance: Optional[str] = typer.Option(
None, "--provenance", help="Replace the page's provenance marker (sourced|general|mixed)"
),
confidence_base: Optional[float] = typer.Option(
None,
"--confidence-base",
help="Re-assess the page's undecayed confidence (0.0-1.0). `confidence` itself is derived - "
"run `wikitool confidence decay --apply` afterwards to recompute it.",
),
date: Optional[str] = typer.Option(
None, "--date",
help="Date to record (YYYY-MM-DD). `modified:` defaults to today; a source's "
"`date:` is its publication date and changes only when given here.",
),
set_fields: Optional[list[str]] = typer.Option(
None,
"--set",
help="Replace a frontmatter field, repeatable: --set tags=a,b. Array values split on "
"commas (escape a literal one as \\,); repeating --set for one array field appends "
"within this call. Page-reference fields belong to `xref`, not here",
),
add_fields: Optional[list[str]] = typer.Option(
None,
"--add",
help="Append elements to an array field without naming the whole list: --add tags=x. "
"Elements already present are left alone",
),
remove_fields: Optional[list[str]] = typer.Option(
None,
"--remove",
help="Drop elements from an array field: --remove tags=x. Removing an absent element "
"succeeds and says so",
),
no_date: bool = typer.Option(
False, "--no-date", help="Only change the given fields; leave the modified/date field alone"
),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview the new frontmatter instead of writing"),
):
"""Bump a page's `modified:` date and optionally rewrite its other frontmatter fields.
`--summary`/`--provenance`/`--confidence-base` are shorthands for the three
fields worth their own flag; `--set`/`--add`/`--remove` reach every other
field the page's type declares. Before they existed, a field `new` wrote
once - `tags:`, `raw_files:` - could never be corrected: `touch` did not
know it, hand-editing frontmatter is what the tool exists to prevent, and
deleting the page to recreate it breaks every reference already pointing at
it. A mistyped `--set tags=` at creation was therefore permanent, and `new`
is not idempotent, so the window to get it right was exactly one command.
"""
pages = load_kb_pages(config.KB_DIR)
page = pages.get(page_title)
if page is None:
fail(f"No page titled '{page_title}' found under wiki/. Create it first with `wikitool new ...`.")
type_path = page.frontmatter.get("type")
if not type_path:
fail(f"Page '{page_title}' has no `type:` frontmatter - fix it before touching it.")
try:
schema = resolver.get_schema(type_path, page.path)
except ValueError as exc:
fail(str(exc))
frontmatter = dict(page.frontmatter)
changes: list[str] = []
touched: set[str] = set()
if not no_date:
field = _date_field(schema, frontmatter)
if field is None:
fail(f"Type {type_path} declares no modified/date field - pass --no-date to skip it.")
# `modified:` is ours to bump - it records when we last touched the page.
# `date:` is not: on a source it is the source material's own publication
# date, a fact about the world that today's date is simply wrong for.
# Auto-bumping it silently replaced a raw file's real date with the day
# the summary happened to be rewritten, and left the page contradicting
# the `**Datum:**` line in its own body. It is still writable, but only
# when the caller says so with an explicit `--date`.
if field == "date" and date is None:
touched.discard(field)
else:
# A `datetime.date`, not a string: that is what `yaml.safe_load`
# yields for every page already on disk, and writing anything else
# made an unchanged date compare unequal to itself - so `touch`
# reported a change on every run, and `dump_frontmatter` had to
# guess whether to quote what it was handed.
new_date = _parse_date(date) if date else datetime.date.today()
touched.add(field)
if frontmatter.get(field) != new_date:
frontmatter[field] = new_date
changes.append(f"{field}: {page.frontmatter.get(field)} -> {new_date.isoformat()}")
if summary is not None:
frontmatter["summary"] = summary
touched.add("summary")
changes.append("summary updated")
if provenance is not None:
frontmatter["provenance"] = provenance
touched.add("provenance")
changes.append(f"provenance: {page.frontmatter.get('provenance')} -> {provenance}")
if confidence_base is not None:
frontmatter["confidence_base"] = round(confidence_base, 2)
touched.add("confidence_base")
changes.append(
f"confidence_base: {page.frontmatter.get('confidence_base')} -> {round(confidence_base, 2)}"
)
# --set/--add/--remove last, so an explicit field always wins over the
# shorthand flags rather than depending on option order.
for flag, values, apply in (
("--set", set_fields, _apply_set),
("--add", add_fields, _apply_add),
("--remove", remove_fields, _apply_remove),
):
parsed = parse_set_fields(values, schema, flag=flag)
for field, value in parsed.items():
_settable_or_fail(field, schema, type_path)
change = apply(frontmatter, field, value)
touched.add(field)
if change:
changes.append(change)
# Filesystem check, not a data-shape one, so the schema cannot carry it -
# and `touch` writes this field now, so it owes the same check `new` does.
if "raw_files" in touched:
check_raw_files_exist(frontmatter.get("raw_files"))
error = validate_fields(frontmatter, schema, touched)
if error:
fail(f"Invalid value for type {type_path}: {error}")
if not changes:
success(f"'{page_title}' already up to date; nothing to change.")
return
for change in changes:
typer.echo(f" {change}")
if dry_run:
typer.echo("No files written (--dry-run).")
return
write_page(page.path, frontmatter, page.body)
success(f"Touched {rel_path(page.path)}")
+122
View File
@@ -0,0 +1,122 @@
"""`wikitool types ...` - discover and describe the wiki's type-spec contracts.
Lets an LLM (or human) find out what page types exist and what a given type
requires by asking wikitool, instead of reading raw type-spec markdown files
into context on every skill invocation. A type-spec's own frontmatter
(`name`, `description`, `schema`, `subtype_field`, `base_dir`) and its
declared `.schema.yaml` are the single source of truth; this command only
formats what `TypeResolver` already resolves - it does not duplicate or
re-derive any type knowledge.
"""
from __future__ import annotations
import json
from typing import Any, Dict
import typer
from chemenu.commands._util import fail
from chemenu.type_resolver import resolver
app = typer.Typer(help="Discover and describe Chemenu type-spec contracts.")
@app.command("list")
def list_types(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
"""List every type-spec under types/, with its name, schema, subtype
field (if any), base directory, and description."""
rows: list[Dict[str, Any]] = []
for type_path, frontmatter in resolver.list_type_specs():
rows.append({
"name": frontmatter.get("name"),
"type_path": type_path,
"schema": frontmatter.get("schema"),
"subtype_field": frontmatter.get("subtype_field"),
"root": frontmatter.get("root") or "kb",
"base_dir": frontmatter.get("base_dir"),
"description": frontmatter.get("description"),
})
if json_out:
typer.echo(json.dumps(rows, indent=2))
return
for row in rows:
typer.echo(f"{row['name']} ({row['type_path']})")
typer.echo(f" schema: {row['schema']}")
if row["subtype_field"]:
typer.echo(f" subtype_field: {row['subtype_field']}")
if row["base_dir"]:
typer.echo(f" base_dir: {row['root']}/{row['base_dir']}")
typer.echo(f" {row['description']}")
typer.echo("")
@app.command("describe")
def describe_type(
name: str = typer.Argument(..., help="Type name, e.g. 'entity' (see `types list`)"),
json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON"),
):
"""Print one type's full contract: frontmatter fields (required/optional,
with enums where declared), its subtype field if any, and its authoring
body - the same information an LLM would otherwise gather by reading the
raw type-spec and `.schema.yaml` files directly."""
type_path = resolver.find_type_by_name(name)
if type_path is None:
available = sorted(fm.get("name") for _, fm in resolver.list_type_specs())
fail(f"No type-spec named '{name}'. Available: {', '.join(available)}")
return # unreachable; keeps type-checkers happy about `type_path` below
type_spec = resolver.load_type_spec(type_path)
frontmatter = type_spec["frontmatter"]
body = type_spec["body"]
schema = resolver.get_schema(type_path)
fields: list[Dict[str, Any]] = []
if schema is not None:
required = set(schema.get("required", []))
for field_name, field_schema in schema.get("properties", {}).items():
fields.append({
"field": field_name,
"required": field_name in required,
"type": field_schema.get("type"),
"enum": field_schema.get("enum"),
})
if json_out:
typer.echo(json.dumps({
"name": frontmatter.get("name"),
"type_path": type_path,
"description": frontmatter.get("description"),
"schema": frontmatter.get("schema"),
"subtype_field": frontmatter.get("subtype_field"),
"base_dir": frontmatter.get("base_dir"),
"title_prefix": frontmatter.get("title_prefix"),
"fields": fields,
"body": body.strip(),
}, indent=2))
return
typer.echo(f"# {frontmatter.get('name')} ({type_path})")
typer.echo(frontmatter.get("description", ""))
typer.echo("")
if frontmatter.get("subtype_field"):
typer.echo(f"subtype_field: {frontmatter['subtype_field']}")
if frontmatter.get("base_dir"):
typer.echo(f"base_dir: {frontmatter.get('root') or 'kb'}/{frontmatter['base_dir']}")
if frontmatter.get("title_prefix"):
typer.echo(f"title_prefix: {frontmatter['title_prefix']!r}")
typer.echo("")
if not fields:
typer.echo("(no schema declared for this type)")
else:
typer.echo("## Frontmatter fields")
for field in fields:
marker = "required" if field["required"] else "optional"
extra = f", enum: {field['enum']}" if field["enum"] else ""
typer.echo(f"- `{field['field']}` ({marker}, {field['type']}{extra})")
typer.echo("")
typer.echo("## Authoring guidance")
typer.echo(body.strip())
+276
View File
@@ -0,0 +1,276 @@
"""`wikitool version` - report, bump, and check the stack's version.
Three jobs that all hang off one number (see `chemenu/version.py` for what
that number means):
- `version show` answers "which stack is this instance running", offline, from
`VERSION` plus the release stamp `dist export` writes.
- `version bump` moves it, and writes the changelog *heading* that has to
accompany the move - the same structure-by-tool/prose-by-author split as
`new`. `docs verify` then holds the two together.
- `version check` is the one command in `wikitool` that makes a network call.
It is deliberately its own command: nothing else reaches for it implicitly,
it needs no key, it times out, and a feed that cannot be reached is reported
as an error rather than silently answered as "up to date".
"""
from __future__ import annotations
import json as _json
from typing import Optional
import typer
from chemenu import config, version as version_mod
from chemenu.commands._util import console, fail, rel_path, success, today_iso
from chemenu.version import Version, VersionError
app = typer.Typer(
help="Report, bump, and check the stack version (see tools/CONTRACT.md).",
invoke_without_command=True,
)
@app.callback()
def version_callback(ctx: typer.Context) -> None:
"""Bare `wikitool version` is a convenience alias for `version show`."""
if ctx.invoked_subcommand is None:
show_command(json_out=False)
def _describe_origin(stamp: Optional[dict]) -> str:
if not stamp:
return "development tree (no release stamp)"
parts = []
exported = stamp.get("exported_at")
if exported:
parts.append(f"exported {exported}")
commit = str(stamp.get("source_commit") or "")
if commit:
parts.append(f"from commit {commit[:12]}")
repo = stamp.get("source_repo")
if repo:
parts.append(str(repo))
return "distribution: " + ", ".join(parts) if parts else "distribution"
@app.command("show")
def show_command(
json_out: bool = typer.Option(False, "--json", help="Print the version and stamp as JSON"),
):
"""Print this instance's stack version and where it came from. Read-only,
offline, and exempt from the Iteration Budget Gate."""
try:
current = version_mod.read_version()
stamp = version_mod.read_stamp()
except VersionError as exc:
fail(str(exc))
return
if json_out:
typer.echo(
_json.dumps(
{
"version": str(current),
"compat_key": list(current.compat_key),
"stamp": stamp,
"update_url": version_mod.update_url(stamp),
},
indent=2,
)
)
return
console.print(f"[bold]{current}[/bold] ({_describe_origin(stamp)})")
if stamp and stamp.get("release_url"):
console.print(f"release: {stamp['release_url']}")
@app.command("check")
def check_command(
url: Optional[str] = typer.Option(
None, "--url", help="Release feed to ask (default: the stamp's, else the built-in origin)"
),
timeout: float = typer.Option(10.0, "--timeout", help="Seconds to wait for the feed"),
json_out: bool = typer.Option(False, "--json", help="Print the result as JSON"),
):
"""Ask the origin's release feed whether a newer stack exists.
The only networked command in `wikitool`. Exits 1 if the feed cannot be
reached or does not answer with a release - an unreachable feed is not the
same answer as "up to date", and must never be reported as one."""
import os
try:
current = version_mod.read_version()
stamp = version_mod.read_stamp()
except VersionError as exc:
fail(str(exc))
return
feed = url or version_mod.update_url(stamp)
token = os.environ.get(version_mod.UPDATE_TOKEN_ENV, "").strip() or None
try:
latest, release_url, published = version_mod.fetch_latest_release(feed, token, timeout)
except VersionError as exc:
fail(str(exc))
return
status = version_mod.UpdateStatus(
local=current,
latest=latest,
state=version_mod.compare(current, latest),
release_url=release_url,
published_at=published,
)
if json_out:
typer.echo(
_json.dumps(
{
"local": str(status.local),
"latest": str(status.latest),
"state": status.state,
"requires_migration": status.state == "migration",
"release_url": status.release_url,
"published_at": status.published_at,
"feed": feed,
},
indent=2,
)
)
return
color = {"current": "green", "ahead": "yellow", "update": "cyan", "migration": "bold yellow"}
console.print(f"[{color[status.state]}]{status.headline}[/{color[status.state]}]")
if status.release_url:
console.print(f"release: {status.release_url}")
if status.state in ("update", "migration"):
console.print(
"Applying it is a separate, manual step - see INSTALL.md "
"§ 'Eine Instanz aktualisieren'."
)
@app.command("notes")
def notes_command(
version: Optional[str] = typer.Option(
None, "--version", help="Which entry to print (default: this tree's VERSION)"
),
):
"""Print one version's `CHANGES.md` entry, for use as release notes.
Mechanical extraction, so the release workflow never has to parse markdown
in shell."""
try:
wanted = Version.parse(version) if version else version_mod.read_version()
except VersionError as exc:
fail(str(exc))
return
changes = version_mod.changes_file()
if not changes.is_file():
fail(f"{version_mod.CHANGES_FILENAME} is missing - there are no release notes to print")
return
section = version_mod.changes_section(changes.read_text(encoding="utf-8"), wanted)
if section is None:
fail(
f"{version_mod.CHANGES_FILENAME} has no entry for {wanted} - "
f"run `wikitool version bump` before releasing, or write the entry"
)
return
typer.echo(section, nl=False)
@app.command("bump")
def bump_command(
major: bool = typer.Option(False, "--major", help="Bump MAJOR (resets MINOR and PATCH)"),
minor: bool = typer.Option(False, "--minor", help="Bump MINOR (resets PATCH)"),
patch: bool = typer.Option(False, "--patch", help="Bump PATCH"),
title: str = typer.Option(..., "--title", help="One-line title for the new CHANGES.md entry"),
no_migration: Optional[str] = typer.Option(
None,
"--no-migration",
help="Why this boundary-crossing bump needs no content migration (recorded in CHANGES.md)",
),
dry_run: bool = typer.Option(False, "--dry-run", help="Report the change without writing"),
):
"""Raise the stack version and open its `CHANGES.md` entry.
Writes `VERSION` and inserts the entry's heading, date and author - the
entry's body stays the author's to write, the same way `new` produces
frontmatter and leaves the prose. `docs verify` afterwards enforces that
the two agree, so a bump with no entry cannot reach a release.
A bump that crosses the compatibility boundary additionally requires a
migration document for the new version, or `--no-migration "<reason>"`.
An instance learning that it must migrate, with nothing telling it how, is
the gap this closes."""
selected = [name for name, chosen in (("major", major), ("minor", minor), ("patch", patch)) if chosen]
if len(selected) != 1:
fail("Pass exactly one of --major / --minor / --patch")
return
if not title.strip():
fail("--title must not be empty - it becomes the changelog entry's heading")
return
try:
current = version_mod.read_version()
new_version = current.bumped(selected[0])
except VersionError as exc:
fail(str(exc))
return
changes = version_mod.changes_file()
if not changes.is_file():
fail(f"{version_mod.CHANGES_FILENAME} is missing - a bump has nowhere to record itself")
return
text = changes.read_text(encoding="utf-8")
existing = version_mod.top_changes_version(text)
if existing is not None and existing >= new_version:
fail(
f"{version_mod.CHANGES_FILENAME} already documents {existing}, which is not older "
f"than {new_version} - bump past it, or fix the changelog"
)
return
author = config.default_author() or "unknown"
crossing = new_version.compat_key != current.compat_key
boundary = " (crosses a compatibility boundary - instances must migrate)" if crossing else ""
if crossing and not no_migration:
from chemenu import kb_state
if not any(m.target == new_version for m in kb_state.load_migrations()):
fail(
f"{current} -> {new_version} crosses the compatibility boundary, so every existing "
f"instance must migrate - but no migration document targets {new_version}.\n"
f"Write one under {rel_path(kb_state.migrations_dir())}/{new_version}-<slug>.md "
f"(see instructions/migrate-corpus.md), or, if no content actually has to change, "
f're-run with --no-migration "<reason>".'
)
return
if no_migration and not crossing:
fail(
f"--no-migration only applies to a bump that crosses the compatibility boundary; "
f"{current} -> {new_version} does not."
)
return
if dry_run:
success(f"Dry run: {current} -> {new_version}{boundary}. Nothing written.")
return
version_mod.write_version(new_version)
changes.write_text(
version_mod.insert_changes_entry(
text, new_version, today_iso(), title.strip(), author,
no_migration_reason=no_migration.strip() if no_migration else None,
),
encoding="utf-8",
)
success(
f"{current} -> {new_version}{boundary}. Wrote {version_mod.VERSION_FILENAME} and opened "
f"the {version_mod.CHANGES_FILENAME} entry - write its body before publishing."
)
+244
View File
@@ -0,0 +1,244 @@
"""`wikitool work` - scaffold and inspect workshop runs under `work/`.
A workshop is the tracked, transient scratch directory for a task that does not
fit in one session (see work/CONTRACT.md). The only mechanical part of it is the
run key: it is derived from the input path, it is the directory name, and a
collision means the same tree is already being ingested. Doing that by hand is
how a second identifier and a `-2` suffix creep in, so it lives here instead.
"""
from __future__ import annotations
import re
import shutil
from typing import Optional
import typer
from chemenu import config
from chemenu.commands._util import fail, rel_path, success, today_iso
app = typer.Typer(help="Workshop runs under work/ (see work/CONTRACT.md).")
RUN_KEY_PREFIX = "ingest-"
# Everything outside this set is folded to a single hyphen, so a run key is
# always a safe directory name and always reproducible from the same input.
_UNSAFE_RE = re.compile(r"[^a-z0-9]+")
REQUIRED_FILES = ("README.md", "plan.md")
def derive_run_key(input_path: str) -> str:
"""Run key for an input path under `raw/`.
Derived from the path *below* `raw/` with separators flattened, never from
the basename: `raw/documents/handbook` and `raw/articles/handbook` share a
basename but are different sources.
"""
relative = input_path.strip().strip("/")
if relative == "raw":
return ""
if relative.startswith("raw/"):
relative = relative[len("raw/"):]
slug = _UNSAFE_RE.sub("-", relative.lower()).strip("-")
if not slug:
return ""
return f"{RUN_KEY_PREFIX}{slug}"
def normalize_run_key(key: str) -> str:
"""Run key for a run with no raw input, given explicitly by the caller.
Not every multi-session task is an ingest. A migration or a sweep across
`kb/` has no input tree to derive a key from, and the alternative - opening
no workshop at all - costs the run its `plan.md`, which is what makes taking
a new session id per unit legitimate rather than a way around a refusal
(instructions/gates.md).
The `ingest-` prefix stays reserved for derived keys, so a directory name
always says which kind of run made it.
"""
slug = _UNSAFE_RE.sub("-", key.strip().lower()).strip("-")
if not slug:
return ""
if slug.startswith(RUN_KEY_PREFIX):
return ""
return slug
def readme_template(run_key: str, input_path: Optional[str]) -> str:
input_line = f"`{input_path}`" if input_path else "none - this run is not an ingest"
return f"""# Workshop: {run_key}
- **Run key:** `{run_key}` (this directory's name - there is no other identifier)
- **Input:** {input_line}
- **Started:** {today_iso()}
- **Session id form:** `WIKITOOL_SESSION_ID="{run_key}/u<N>"`, one per unit
## Goal
TODO: what this run must produce.
## Closes when
TODO: the condition that ends the run - normally "every unit in plan.md is published and
its `## Not Extracted` section is filled".
## Checklist
TODO: one line per unit from plan.md, e.g.
- [ ] u1 <unit> - extract / promote / publish
## Open decisions
- None yet. Record blockers as `DECISION NEEDED: <question>` and stop at them.
"""
def plan_template(run_key: str, input_path: Optional[str]) -> str:
if input_path is None:
return f"""# Plan: {run_key}
Cut the work into units. One unit is one session id and one `publish`, so it has to fit inside
the 60-call iteration budget on its own - count the `wikitool` calls the unit needs before
committing to its size.
| # | Unit | Job | Done when |
|---|------|-----|-----------|
| u1 | TODO | TODO | TODO |
## Deliberately excluded from this run
- TODO: what this run is not touching, and why.
"""
return f"""# Plan: {run_key}
Input tree: `{input_path}`
Cut the tree into units. One unit does one job and becomes one source page. A unit whose
`raw_files` list would pass roughly 15 entries is still too coarse.
| # | Unit (input subtree) | Job | Planned source page | Why this cut |
|---|----------------------|-----|---------------------|--------------|
| u1 | TODO | TODO | `Source - TODO` | TODO |
## Deliberately excluded from this run
- TODO: parts of the tree that are not being ingested at all, and why.
"""
@app.command("new")
def new_command(
input_path: Optional[str] = typer.Option(
None,
"--input",
help="Path to the raw source tree or file this run covers, e.g. raw/documents/handbook",
),
key: Optional[str] = typer.Option(
None,
"--key",
help="Explicit run key for a run with no raw input (a migration, a sweep across kb/). "
"Mutually exclusive with --input; may not start with 'ingest-'.",
),
again: bool = typer.Option(
False,
"--again",
help="This is a deliberate re-ingest of a tree already processed before: append today's "
"date to the run key instead of refusing the collision.",
),
dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be created, write nothing"),
):
"""Scaffold `work/<runkey>/` for one workshop run."""
if (input_path is None) == (key is None):
fail(
"Pass exactly one of --input (an ingest of raw material, key derived from the path) "
"or --key (a run with no raw input, key given explicitly)."
)
if key is not None:
run_key = normalize_run_key(key)
if not run_key:
fail(
f"`{key}` is not a usable run key - it must contain letters or digits and must not "
f"start with `{RUN_KEY_PREFIX}`, which is reserved for keys derived from a raw path."
)
else:
source = (config.ROOT / input_path).resolve()
try:
source.relative_to(config.RAW_DIR)
except ValueError:
fail(
f"`{input_path}` is not under raw/. A run key is derived from the input path below "
"raw/, so an --input workshop can only be opened for raw material. A run with no raw "
"input takes --key instead."
)
if not source.exists():
fail(f"`{input_path}` does not exist - a workshop is opened for material that is already in raw/")
run_key = derive_run_key(input_path)
if not run_key:
fail(f"`{input_path}` yields an empty run key - point --input at a subtree of raw/, not at raw/ itself")
if again:
run_key = f"{run_key}-{today_iso()}"
target = config.WORK_DIR / run_key
if target.exists():
if again:
fail(
f"`{rel_path(target)}` already exists - a second re-ingest of the same tree on the "
"same day. Resume that run or close it first."
)
fail(
f"`{rel_path(target)}` already exists, which means this tree is already being ingested. "
"Resume that run, or - if the tree itself has changed since - re-run with --again to "
"open a dated second pass. Never work around this with a numbered suffix."
)
if dry_run:
success(f"Would create {rel_path(target)}/ with {', '.join(REQUIRED_FILES)}")
return
target.mkdir(parents=True)
(target / "README.md").write_text(readme_template(run_key, input_path), encoding="utf-8")
(target / "plan.md").write_text(plan_template(run_key, input_path), encoding="utf-8")
typer.echo(f"Run key: {run_key}")
typer.echo(f"Workshop: {rel_path(target)}/")
typer.echo(f"Next: fill in plan.md, then export WIKITOOL_SESSION_ID=\"{run_key}/u1\"")
success(f"Created workshop {run_key}")
@app.command("close")
def close_command(
run_key: str = typer.Option(..., "--run-key", help="The workshop directory name"),
yes: bool = typer.Option(
False,
"--yes",
"-y",
help="Confirm deletion. Required: closing discards the only copy of the run's working "
"notes, so the durable conclusions must already be in kb/.",
),
dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be deleted, delete nothing"),
):
"""Delete a finished workshop. Its conclusions must already be in `kb/`."""
target = config.WORK_DIR / run_key
if not target.is_dir():
fail(f"No workshop `{run_key}` under work/ - `ls work/` shows the open ones")
files = sorted(p for p in target.rglob("*") if p.is_file())
if dry_run or not yes:
listing = "\n".join(f"- {rel_path(p)}" for p in files)
message = (
f"Closing `{run_key}` deletes {len(files)} file(s):\n{listing}\n"
"Nothing here is recoverable from the rest of the repo. Confirm the durable "
"conclusions are already in kb/, then re-run with --yes."
)
if dry_run:
typer.echo(message)
return
fail(message)
shutil.rmtree(target)
success(f"Closed workshop {run_key} ({len(files)} file(s) deleted). Log it with `log append`.")
+339
View File
@@ -0,0 +1,339 @@
"""Bidirectional cross-reference management between wiki pages.
`xref add` keeps two pages' frontmatter `related:` lists AND their body
"## Relationships" sections in sync in one operation, instead of the 3-5
separate manual edits this used to take per pair of pages. It is idempotent:
re-running it never duplicates a link.
"""
from __future__ import annotations
import re
from pathlib import Path
import typer
from chemenu import config, sections
from chemenu.commands._util import fail, parse_list, success
from chemenu.commands.page_ops import strip_frontmatter_ref
from chemenu.frontmatter_io import write_page
from chemenu.page import Page
from chemenu.kb_scan import load_kb_pages
app = typer.Typer(help="Manage bidirectional cross-references between wiki pages.")
def _find_page(pages: dict[str, Page], name: str) -> Page:
if name not in pages:
fail(f"No page titled '{name}' found under wiki/. Create it first with `wikitool new ...`.")
return pages[name]
def _declared_ref_fields(page: Page) -> list[str]:
from chemenu.commands.page_ops import page_ref_fields
return page_ref_fields(page)
def _require_related_field(page: Page, title: str) -> None:
"""Refuse to write `related:` on a type that does not declare it.
`xref add` used to write the field unconditionally. On a source page -
whose type declares `page_ref_fields: [entities, concepts]` - that produced
frontmatter the schema rejects (`additionalProperties: false`), which
`xref remove` then could not clear, because it only swept declared fields.
One command created a state another could not undo.
"""
declared = _declared_ref_fields(page)
if "related" in declared:
return
fail(
f"'{title}' is a {page.frontmatter.get('type')} page, whose type does not declare a "
f"`related:` field, so `xref add` has nothing to write there.\n"
f" Reference fields this type declares: {', '.join(declared) or '(none)'}\n"
f" For a source page, `wikitool xref link-source --source \"{title}\" "
f"--entities <titles>` is the command that fills them."
)
def _back_reference_field(source: Page, target: Page) -> str | None:
"""Which of `source`'s declared ref fields `target` belongs in.
Derived from the target's collection rather than a hardcoded type-to-field
map: a page under `kb/entities/` belongs in `entities:`, one under
`kb/concepts/` in `concepts:`. The collection directory *is* the field
name, so a new collection needs no code change here - it needs a type that
declares the matching field.
"""
try:
collection = target.path.relative_to(config.KB_DIR).parts[0]
except (ValueError, IndexError):
return None
return collection if collection in _declared_ref_fields(source) else None
def add_related(frontmatter: dict, other_title: str) -> bool:
"""Add other_title to frontmatter['related'] if not already present.
Returns True if a change was made."""
related = frontmatter.setdefault("related", [])
if other_title in related:
return False
related.append(other_title)
return True
def _section_bounds(body: str, heading: str) -> tuple[int, int] | None:
match = sections.heading_re(heading).search(body)
if not match:
return None
start = match.end()
next_heading = re.search(r"^## ", body[start:], re.MULTILINE)
end = start + next_heading.start() if next_heading else len(body)
return start, end
def add_bullet_to_section(body: str, heading: str, bullet: str, dedup_link: str) -> str:
"""Insert `bullet` into the `## {heading}` section of body, unless a
wikilink to dedup_link already appears there. Creates the section
(before the See Also section if present, else at the end) if missing.
`heading` is a canonical name from `sections`; an existing section is found
under its aliases too, so a page that has not been translated yet is still
appended to rather than given a duplicate section. A section this creates
always carries the canonical name."""
bounds = _section_bounds(body, heading)
if bounds is None:
section = f"## {heading}\n\n{bullet}\n\n"
see_also = sections.heading_re(sections.SEE_ALSO).search(body)
if heading != sections.SEE_ALSO and see_also:
return body[: see_also.start()] + section + body[see_also.start() :]
return body.rstrip("\n") + "\n\n" + section.rstrip("\n") + "\n"
start, end = bounds
section_text = body[start:end]
if f"[[{dedup_link}]]" in section_text:
return body
trimmed = section_text.rstrip("\n")
new_section = trimmed + "\n" + bullet + "\n\n"
return body[:start] + new_section + body[end:]
def add_relationship_bullet(body: str, label: str, other_title: str) -> str:
bullet = f"- **{label}:** [[{other_title}]]"
return add_bullet_to_section(body, sections.RELATIONSHIPS, bullet, other_title)
def add_see_also_bullet(body: str, other_title: str) -> str:
return add_bullet_to_section(body, sections.SEE_ALSO, f"- [[{other_title}]]", other_title)
@app.command("add")
def xref_add(
a: str = typer.Option(..., "--a", help="Exact title of page A"),
b: str = typer.Option(..., "--b", help="Exact title of page B"),
rel_a: str = typer.Option("related to", "--rel-a", help="Relationship label on A pointing to B"),
rel_b: str = typer.Option("related to", "--rel-b", help="Relationship label on B pointing to A"),
see_also: bool = typer.Option(True, "--see-also/--no-see-also", help="Also add reciprocal 'See Also' bullets"),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes to both pages instead of writing"),
):
pages = load_kb_pages(config.KB_DIR)
page_a = _find_page(pages, a)
page_b = _find_page(pages, b)
# Both refusals before either write, so a rejected pair leaves no half-link.
_require_related_field(page_a, a)
_require_related_field(page_b, b)
related_changed_a = add_related(page_a.frontmatter, b)
related_changed_b = add_related(page_b.frontmatter, a)
body_a = add_relationship_bullet(page_a.body, rel_a, b)
body_b = add_relationship_bullet(page_b.body, rel_b, a)
if see_also:
body_a = add_see_also_bullet(body_a, b)
body_b = add_see_also_bullet(body_b, a)
changed_a = related_changed_a or body_a != page_a.body
changed_b = related_changed_b or body_b != page_b.body
if dry_run:
state_a = "would update" if changed_a else "already up to date"
state_b = "would update" if changed_b else "already up to date"
typer.echo(f"[dry-run] '{a}': {state_a} (related / Relationships / See Also)")
typer.echo(f"[dry-run] '{b}': {state_b} (related / Relationships / See Also)")
typer.echo("No files written (--dry-run).")
return
try:
write_page(page_a.path, page_a.frontmatter, body_a)
except OSError as exc:
fail(f"Failed to write '{a}': {exc}. '{b}' was not touched - fix the write failure and retry once.")
try:
write_page(page_b.path, page_b.frontmatter, body_b)
except OSError as exc:
fail(
f"'{a}' was updated but writing '{b}' failed: {exc}. The link is now one-directional - "
f"fix the write failure, then re-run `xref add --a \"{a}\" --b \"{b}\"` (idempotent, safe to retry)."
)
success(f"Linked '{a}' <-> '{b}' ({rel_a} / {rel_b})")
def remove_related(frontmatter: dict, other_title: str) -> bool:
"""Drop other_title from frontmatter['related'] if present. Returns True if
a change was made."""
related = frontmatter.get("related")
if not related or other_title not in related:
return False
frontmatter["related"] = [title for title in related if title != other_title]
return True
def remove_link_bullets(body: str, other_title: str) -> str:
"""Remove the whole-line Relationships/See Also bullets `xref add` writes -
`- **label:** [[Other]]` and `- [[Other]]`.
Deliberately narrow, matching `xref add`'s own output: a bullet carrying
prose alongside the link is left for the author to edit.
"""
escaped = re.escape(other_title)
pattern = re.compile(
rf"^[ \t]*-[ \t]+(?:\*\*[^*\n]+:\*\*[ \t]+)?\[\[{escaped}\]\][ \t]*\n?",
re.MULTILINE,
)
return pattern.sub("", body)
@app.command("remove")
def xref_remove(
a: str = typer.Option(..., "--a", help="Exact title of page A (must exist)"),
b: str = typer.Option(..., "--b", help="Title to unlink from A; need not still exist as a page"),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes instead of writing"),
):
"""Remove a cross-reference: the inverse of `xref add`.
Clears `--b` from *every* page-ref frontmatter field the type declares
(`related:`, `sources:`, `entities:`, `concepts:`), not just `related:`,
so it is equally the inverse of `xref link-source`.
`--b` deliberately does not have to exist. Clearing a reference left
behind by a hand-deleted or hand-renamed page is the main reason this
command exists, and in that case the target is exactly what is missing.
Idempotent: removing a link that is already gone is a no-op.
"""
pages = load_kb_pages(config.KB_DIR)
page_a = _find_page(pages, a)
page_b = pages.get(b)
body_a = remove_link_bullets(page_a.body, b)
changed_a = strip_frontmatter_ref(page_a, b) or body_a != page_a.body
changed_b = False
body_b = ""
if page_b is not None:
body_b = remove_link_bullets(page_b.body, a)
changed_b = strip_frontmatter_ref(page_b, a) or body_b != page_b.body
if dry_run:
typer.echo(f"[dry-run] '{a}': {'would update' if changed_a else 'no reference to remove'}")
if page_b is None:
typer.echo(f"[dry-run] '{b}': not a page - only '{a}' would be updated")
else:
typer.echo(f"[dry-run] '{b}': {'would update' if changed_b else 'no reference to remove'}")
typer.echo("No files written (--dry-run).")
return
if changed_a:
write_page(page_a.path, page_a.frontmatter, body_a)
if changed_b and page_b is not None:
write_page(page_b.path, page_b.frontmatter, body_b)
if not changed_a and not changed_b:
success(f"No link between '{a}' and '{b}' to remove; nothing changed.")
return
if page_b is None:
success(f"Removed '{a}' -> '{b}' ('{b}' is not a page, so only '{a}' was updated).")
return
success(f"Unlinked '{a}' <-> '{b}'")
@app.command("link-source")
def xref_link_source(
source: str = typer.Option(..., "--source", help="Exact source page title, e.g. 'Source - Docker Cheatsheet'"),
entities: str = typer.Option(..., "--entities", help="Comma-separated entity/concept titles the source mentions"),
dry_run: bool = typer.Option(False, "--dry-run", help="Preview which pages would be linked instead of writing"),
):
pages = load_kb_pages(config.KB_DIR)
source_page = _find_page(pages, source)
names = parse_list(entities)
linked: list[str] = []
skipped: list[str] = []
failed: list[str] = []
unrouted: list[str] = []
source_changed = False
for name in names:
page = pages.get(name)
if page is None:
skipped.append(name)
continue
sources = page.frontmatter.setdefault("sources", [])
if source not in sources:
sources.append(source)
body = add_see_also_bullet(page.body, source)
# The way back. Until this existed the command wrote only the targets,
# so a source page's own `entities:`/`concepts:` stayed as `new` left
# them - and an ingest that creates its concept pages *after* the
# source page (which it must, since their titles come out of the
# extraction) left them empty with no command able to fill them.
field = _back_reference_field(source_page, page)
if field is None:
unrouted.append(name)
else:
entries = source_page.frontmatter.setdefault(field, [])
if name not in entries:
entries.append(name)
source_changed = True
if not dry_run:
try:
write_page(page.path, page.frontmatter, body)
except OSError as exc:
failed.append(f"{name} ({exc})")
continue
linked.append(name)
if source_changed and not dry_run:
try:
write_page(source_page.path, source_page.frontmatter, source_page.body)
except OSError as exc:
fail(
f"Targets were updated but writing '{source}' failed: {exc}. Its reference "
f"arrays are now behind - fix the write failure and re-run (idempotent)."
)
if unrouted:
typer.echo(
f"Not recorded on '{source}' (no matching reference field for their collection): "
f"{', '.join(unrouted)}"
)
if linked:
verb = "Would link" if dry_run else "Linked"
typer.echo(f"{verb} source '{source}' to: {', '.join(linked)}")
if skipped:
typer.echo(f"Skipped (page not found): {', '.join(skipped)}")
if failed:
typer.echo(f"Failed to write (fix and re-run for just these names): {', '.join(failed)}")
if dry_run:
typer.echo("No files written (--dry-run).")
return
if skipped or failed:
parts = []
if skipped:
parts.append(f"page(s) not found: {', '.join(skipped)}")
if failed:
parts.append(f"page(s) failed to write: {', '.join(failed)}")
fail(f"Linked {len(linked)}/{len(names)} page(s); " + "; ".join(parts))
success(f"Linked source '{source}' to {len(names)} page(s)")