Chemenu 2.1.0 - deterministischer Wissenskompiler
Chemenu kompiliert Rohnotizen zu einem verlinkten, quellengebundenen Wiki: raw/ -> types/ + tools/ -> kb/ -> reports/. Was mechanisch ist, macht tools/wikitool; was Urteil braucht, macht ein Agent unter Contracts, deren Grenzen in Code durchgesetzt sind statt im Prompt. Dieser Commit ist der Startpunkt der oeffentlichen Historie. Die vorherige Entwicklung fand in einer privaten Instanz statt und ist nicht Teil dieses Repositorys; ihre Erzaehlung steht vollstaendig in CHANGES.md, das mit 44 Eintraegen von 0.1.0 bis 2.1.0 erhalten geblieben ist. Der mitgelieferte Korpus ist ein Testbett und eine Demo: 170 Seiten ueber den Stack selbst - Gates, Lint, Versionierung, Suche, das Wiki-Muster. Er dokumentiert das Werkzeug mit den eigenen Mitteln des Werkzeugs. Lizenz: AGPL-3.0 fuer den Stack (tools/, types/), CC-BY-4.0 fuer die Inhalte. Die Grenze zwischen beiden ist der Dateiplan, den dist export berechnet - siehe NOTICE.
This commit is contained in:
commit
18ae28f918
368 files changed
+50628
No files matched your search
Whitespace-only changes.
@@ -0,0 +1,192 @@
|
||||
"""Shared helpers for wikitool subcommands."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
import typer
|
||||
from rich.console import Console
|
||||
|
||||
console = Console()
|
||||
|
||||
# A third outcome alongside success (0) and validation error (1): the command
|
||||
# is refusing until a *human* has seen its output and cleared it. It exists as
|
||||
# its own code so the caller - an agent, a harness hook, a CI job, a trajectory
|
||||
# scorer - can tell "stop and ask the user" apart from "your input was wrong,
|
||||
# fix it and retry". Nothing about *why* clearance is needed lives in the agent
|
||||
# instructions: the command's own output carries the reason, the evidence, and
|
||||
# the exact re-run line.
|
||||
EXIT_NEEDS_CLEARANCE = 42
|
||||
|
||||
|
||||
def success(msg: str) -> None:
|
||||
console.print(f"[green]OK[/green] {msg}")
|
||||
|
||||
|
||||
# Set by `fail()`, read once per process by the CLI entry point. Exit 1 raised
|
||||
# through `fail()` means the command declined and did the thing it was asked
|
||||
# for: the argument was rejected, or a read-only check reported findings.
|
||||
# Neither is an iteration step on the wiki, so the Iteration Budget Gate gives
|
||||
# the slot back (see run_budget.refund). A command that has already done its
|
||||
# work and then reports a non-zero result - `lint --fail-on-error` writes its
|
||||
# report first - raises `typer.Exit(1)` directly and stays counted.
|
||||
_declined = False
|
||||
|
||||
|
||||
def declined() -> bool:
|
||||
"""Whether this process left through `fail()`."""
|
||||
return _declined
|
||||
|
||||
|
||||
def fail(msg: str) -> None:
|
||||
global _declined
|
||||
_declined = True
|
||||
console.print(f"[bold red]ERROR[/bold red] {msg}")
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
def needs_clearance(msg: str) -> None:
|
||||
"""Refuse with EXIT_NEEDS_CLEARANCE. The message is written to be shown to
|
||||
a human verbatim - it is the whole user-facing artifact of this gate."""
|
||||
console.print(f"[bold yellow]NEEDS USER CLEARANCE[/bold yellow] {msg}")
|
||||
raise typer.Exit(code=EXIT_NEEDS_CLEARANCE)
|
||||
|
||||
|
||||
# A comma preceded by a backslash is a literal comma, not a separator.
|
||||
_UNESCAPED_COMMA = re.compile(r"(?<!\\),")
|
||||
|
||||
|
||||
def parse_list(value: str | None) -> list[str]:
|
||||
"""Split a comma-separated CLI value into list elements.
|
||||
|
||||
`\\,` is an escaped literal comma: it survives the split and lands inside
|
||||
the element. Without it a list format simply cannot express an element
|
||||
that contains a comma - and shell quoting is no help, because the quotes
|
||||
are gone long before this sees the string. Paths and page titles carry
|
||||
commas often enough for that to matter: it once cost a `raw/` file its
|
||||
original name, which `raw/CONTRACT.md` forbids.
|
||||
"""
|
||||
if not value:
|
||||
return []
|
||||
parts = (part.replace("\\,", ",").strip() for part in _UNESCAPED_COMMA.split(value))
|
||||
return [part for part in parts if part]
|
||||
|
||||
|
||||
def coerce_set_value(raw_value: str, field_schema: Optional[Dict[str, Any]]) -> Any:
|
||||
"""Coerce a `--set field=value` string to the type its schema declares.
|
||||
|
||||
Arrays are comma-split (see `parse_list` for the escape), numbers are
|
||||
parsed as float/int, booleans as true/false; everything else stays a
|
||||
string. Unknown fields (no schema entry) pass through as strings and are
|
||||
then caught by schema validation's `additionalProperties: false`.
|
||||
"""
|
||||
declared = (field_schema or {}).get("type")
|
||||
if declared == "array":
|
||||
return parse_list(raw_value)
|
||||
if declared == "number":
|
||||
try:
|
||||
return float(raw_value)
|
||||
except ValueError:
|
||||
return raw_value
|
||||
if declared == "integer":
|
||||
try:
|
||||
return int(raw_value)
|
||||
except ValueError:
|
||||
return raw_value
|
||||
if declared == "boolean":
|
||||
if raw_value.lower() in ("true", "false"):
|
||||
return raw_value.lower() == "true"
|
||||
return raw_value
|
||||
|
||||
|
||||
def parse_set_fields(
|
||||
set_fields: Optional[list[str]], schema: Optional[Dict[str, Any]], flag: str = "--set"
|
||||
) -> Dict[str, Any]:
|
||||
"""Parse repeated `<flag> field=value` pairs into a frontmatter dict,
|
||||
coercing each value by the field's declared schema type.
|
||||
|
||||
Repeating the flag for an *array* field appends rather than replaces, so
|
||||
`--set raw_files=a --set raw_files=b` yields both. That is the form that
|
||||
needs no separator at all, and therefore the one to reach for when an
|
||||
element contains a comma; `\\,` inside a single value does the same job
|
||||
for a one-liner. Repeating a scalar field still means "last one wins" -
|
||||
there is nothing to append to.
|
||||
|
||||
Note that this is per *invocation*. What a parsed value then means for a
|
||||
page already on disk is the caller's decision: `new` writes it as the
|
||||
page's initial value, while `touch` replaces, extends or subtracts
|
||||
depending on which flag it came from.
|
||||
"""
|
||||
explicit: Dict[str, Any] = {}
|
||||
properties = (schema or {}).get("properties", {})
|
||||
for pair in set_fields or []:
|
||||
if "=" not in pair:
|
||||
fail(f"{flag} expects field=value, got: {pair}")
|
||||
field_name, raw_value = pair.split("=", 1)
|
||||
field_name = field_name.strip()
|
||||
if not field_name:
|
||||
fail(f"{flag} expects field=value, got: {pair}")
|
||||
value = coerce_set_value(raw_value, properties.get(field_name))
|
||||
previous = explicit.get(field_name)
|
||||
if isinstance(value, list) and isinstance(previous, list):
|
||||
previous.extend(value)
|
||||
else:
|
||||
explicit[field_name] = value
|
||||
return explicit
|
||||
|
||||
|
||||
def check_raw_files_exist(raw_files: Any) -> None:
|
||||
"""Verify every `raw_files:` entry is an existing file.
|
||||
|
||||
This is the one validation that genuinely cannot live in the schema:
|
||||
it is filesystem I/O, not a data-shape constraint. Cardinality
|
||||
(`minItems: 1`) is already enforced by the schema itself, so only
|
||||
existence and file-vs-directory are checked here.
|
||||
|
||||
Shared by `new` and `touch` - both write the field, and a page pointing at
|
||||
a raw file that is not there is the same defect whichever wrote it.
|
||||
"""
|
||||
from chemenu import config
|
||||
|
||||
for raw_path in raw_files or []:
|
||||
full_path = config.ROOT / raw_path
|
||||
if not full_path.exists():
|
||||
fail(
|
||||
f"raw_files path does not exist: {raw_path}\n"
|
||||
" This is one element after splitting the value on commas. If the real "
|
||||
"filename contains a comma, escape it as `\\,` or pass one `--set "
|
||||
"raw_files=<path>` per file - never rename the raw file to fit the flag."
|
||||
)
|
||||
if full_path.is_dir():
|
||||
fail(f"raw_files must be a file, not a directory: {raw_path}")
|
||||
|
||||
|
||||
def today_iso() -> str:
|
||||
return date.today().isoformat()
|
||||
|
||||
|
||||
def rel_path(path: Path) -> str:
|
||||
"""Format a path relative to the repo root for display, falling back to
|
||||
the raw path if it lies outside the root (e.g. in tests)."""
|
||||
from chemenu import config
|
||||
|
||||
try:
|
||||
return str(Path(path).relative_to(config.ROOT))
|
||||
except ValueError:
|
||||
return str(path)
|
||||
|
||||
|
||||
def check_collision(name: str) -> None:
|
||||
"""Fail if any page under wiki/ already has `name` as its filename stem.
|
||||
|
||||
The stem *is* the page title and wikilinks resolve by title alone, so two
|
||||
files sharing a stem in different directories are indistinguishable to
|
||||
every link in the wiki. Shared by `new` and `rename`.
|
||||
"""
|
||||
from chemenu import config
|
||||
|
||||
for path in config.KB_DIR.rglob("*.md"):
|
||||
if path.stem == name:
|
||||
fail(f"A page titled '{name}' already exists at {rel_path(path)}")
|
||||
@@ -0,0 +1,220 @@
|
||||
"""`wikitool cite ...` - real GFM footnote citations.
|
||||
|
||||
A citation marker is `[^cite-id]` in a page's prose, resolved by a
|
||||
`[^cite-id]: [[Source - X]]` (or `[[Source - X|file.md]]`) definition line in
|
||||
the page's trailing Footnotes block (see chemenu.provenance for the
|
||||
regexes and cite_id() derivation). AGENTS.md invariant 1 forbids hand-writing
|
||||
generated structure, and a cite-id is exactly that - an author must never
|
||||
compute or paste one by hand. `cite add` is the only way to get one onto a
|
||||
page; `cite sync` is the only way to reconcile a page's block after prose
|
||||
edits changed which ids are actually referenced.
|
||||
|
||||
None of this writes the inline `[^cite-id]` reference into prose: where a
|
||||
citation belongs in a sentence is an editorial call, same as the prose itself
|
||||
(see tools/CONTRACT.md's design notes). `cite add` prints the marker to paste
|
||||
in; the LLM places it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, rel_path, success
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.page import Page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.provenance import (
|
||||
CITE_REF_RE,
|
||||
cite_id,
|
||||
cite_block_heading,
|
||||
render_page_body,
|
||||
split_cite_block,
|
||||
unique_cite_id,
|
||||
)
|
||||
|
||||
app = typer.Typer(help="Manage [^cite-id] footnote citations and their Footnotes definition blocks.")
|
||||
|
||||
|
||||
def _find_page(pages: dict[str, Page], title: str) -> Page:
|
||||
if title not in pages:
|
||||
fail(f"No page titled '{title}' found under wiki/.")
|
||||
return pages[title]
|
||||
|
||||
|
||||
@app.command("id")
|
||||
def cite_id_command(
|
||||
title: str = typer.Option(..., "--title", help="Source page title, e.g. 'Source - Docker Cheatsheet'"),
|
||||
file: Optional[str] = typer.Option(None, "--file", help="Qualifier for a multi-file source, e.g. 'storage-model.md'"),
|
||||
):
|
||||
"""Print the deterministic id cite_id() would derive for (--title, --file).
|
||||
|
||||
Read-only preview - does not check the id is actually free on any given
|
||||
page (two pages, or two distinct pairs on one page, can share this base
|
||||
id; `cite add`/`cite sync` are what apply the real -2/-3 suffixing).
|
||||
"""
|
||||
typer.echo(cite_id(title, file))
|
||||
|
||||
|
||||
def upsert_citation(page: Page, source_title: str, qualifier: Optional[str]) -> tuple[str, str, bool]:
|
||||
"""Ensure `page` has a Footnotes definition for (source_title, qualifier)
|
||||
and that source_title is in its frontmatter `sources:`. Returns
|
||||
(cite_id_to_use, new_body, changed) - reuses an existing definition for
|
||||
the same pair instead of minting a duplicate id."""
|
||||
head, definitions = split_cite_block(page.body)
|
||||
|
||||
existing_id = next(
|
||||
(cid for cid, pair in definitions.items() if pair == (source_title, qualifier)),
|
||||
None,
|
||||
)
|
||||
if existing_id is not None:
|
||||
marker_id = existing_id
|
||||
block_changed = False
|
||||
else:
|
||||
marker_id = unique_cite_id(set(definitions), source_title, qualifier)
|
||||
definitions[marker_id] = (source_title, qualifier)
|
||||
block_changed = True
|
||||
|
||||
sources = page.frontmatter.setdefault("sources", [])
|
||||
sources_changed = source_title not in sources
|
||||
if sources_changed:
|
||||
sources.append(source_title)
|
||||
|
||||
new_body = render_page_body(head, definitions, cite_block_heading(page.body))
|
||||
changed = block_changed or sources_changed or new_body != page.body
|
||||
return marker_id, new_body, changed
|
||||
|
||||
|
||||
@app.command("add")
|
||||
def cite_add(
|
||||
page_title: str = typer.Option(..., "--page", help="Exact title of the page to add a citation on"),
|
||||
source: str = typer.Option(..., "--source", help="Exact title of the source page being cited, e.g. 'Source - X'"),
|
||||
file: Optional[str] = typer.Option(None, "--file", help="Qualifier for a multi-file source, e.g. 'storage-model.md'"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview instead of writing"),
|
||||
):
|
||||
"""Upsert a Footnotes definition for `--source` (reusing it if the page
|
||||
already cites the same source/file pair) and ensure `--source` is in the
|
||||
page's frontmatter `sources:`. Prints the `[^cite-id]` marker to paste
|
||||
into the prose - placing it is still the caller's job.
|
||||
"""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
page = _find_page(pages, page_title)
|
||||
if source not in pages:
|
||||
fail(f"No page titled '{source}' found under wiki/ - citing a page that doesn't exist would be a dangling reference.")
|
||||
|
||||
marker_id, new_body, changed = upsert_citation(page, source, file)
|
||||
marker = f"[^{marker_id}]"
|
||||
|
||||
if dry_run:
|
||||
state = "would update" if changed else "already up to date"
|
||||
typer.echo(f"[dry-run] '{page_title}': {state}")
|
||||
typer.echo(f"marker: {marker}")
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
if changed:
|
||||
write_page(page.path, page.frontmatter, new_body)
|
||||
typer.echo(f"marker: {marker}")
|
||||
success(
|
||||
f"{'Updated' if changed else 'Already up to date:'} '{page_title}' cites '{source}'"
|
||||
+ (f" ({file})" if file else "")
|
||||
+ f". Paste {marker} at the point in the prose the fact appears."
|
||||
)
|
||||
|
||||
|
||||
def sync_page(page: Page) -> tuple[str, bool, list[str], list[str]]:
|
||||
"""Reconcile one page's Footnotes block against its actual `[^id]`
|
||||
references: prune definitions nothing references any more, and re-render
|
||||
the block in first-reference order. Never mints or recomputes an id from
|
||||
a title - a reference with no definition is reported, not guessed at.
|
||||
|
||||
Returns (new_body, changed, pruned_ids, undefined_ref_ids).
|
||||
"""
|
||||
head, definitions = split_cite_block(page.body)
|
||||
referenced_ids = [m.group(1) for m in CITE_REF_RE.finditer(head)]
|
||||
referenced_set = set(referenced_ids)
|
||||
|
||||
if not definitions and not referenced_ids:
|
||||
# No citation content at all - leave the page's whitespace exactly as
|
||||
# it is. Without this, re-rendering an empty block still normalizes
|
||||
# trailing newlines, which would make `cite sync --all` rewrite every
|
||||
# page in the wiki instead of just the ones it actually has work to do.
|
||||
return page.body, False, [], []
|
||||
|
||||
pruned = [cid for cid in definitions if cid not in referenced_set]
|
||||
undefined = sorted({cid for cid in referenced_ids if cid not in definitions})
|
||||
|
||||
ordered: dict[str, tuple[str, Optional[str]]] = {}
|
||||
seen: set[str] = set()
|
||||
for cid in referenced_ids:
|
||||
if cid in definitions and cid not in seen:
|
||||
ordered[cid] = definitions[cid]
|
||||
seen.add(cid)
|
||||
|
||||
new_body = render_page_body(head, ordered, cite_block_heading(page.body))
|
||||
changed = new_body != page.body
|
||||
return new_body, changed, pruned, undefined
|
||||
|
||||
|
||||
@app.command("sync")
|
||||
def cite_sync(
|
||||
page_title: Optional[str] = typer.Option(None, "--page", help="Sync just this page"),
|
||||
all_pages: bool = typer.Option(False, "--all", help="Sync every page under wiki/"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Report what would change instead of writing"),
|
||||
):
|
||||
"""Prune orphan Footnotes definitions and re-render each page's block in
|
||||
first-reference order. Reports any `[^id]` reference left with no
|
||||
definition - that is an editorial gap (a citation whose `cite add` never
|
||||
ran, or a hand-typed id), not something this command can fix."""
|
||||
if bool(page_title) == bool(all_pages):
|
||||
fail("Provide exactly one of --page or --all")
|
||||
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
targets = [_find_page(pages, page_title)] if page_title else sorted(pages.values(), key=lambda p: p.path)
|
||||
|
||||
touched: list[str] = []
|
||||
undefined_report: dict[str, list[str]] = {}
|
||||
failed: list[str] = []
|
||||
|
||||
for page in targets:
|
||||
new_body, changed, pruned, undefined = sync_page(page)
|
||||
title = page.path.stem
|
||||
if undefined:
|
||||
undefined_report[title] = undefined
|
||||
if not changed:
|
||||
continue
|
||||
touched.append(title)
|
||||
if dry_run:
|
||||
continue
|
||||
try:
|
||||
write_page(page.path, page.frontmatter, new_body)
|
||||
except OSError as exc:
|
||||
failed.append(f"{title} ({exc})")
|
||||
|
||||
if failed:
|
||||
fail(
|
||||
f"Synced {len(touched) - len(failed)}/{len(touched)} page(s) before a write failed: "
|
||||
f"{', '.join(failed)}. Safe to retry - each page's re-render is idempotent."
|
||||
)
|
||||
|
||||
verb = "Would update" if dry_run else "Updated"
|
||||
if touched:
|
||||
typer.echo(f"{verb} {len(touched)} page(s):")
|
||||
for title in touched:
|
||||
typer.echo(f" - {title}")
|
||||
else:
|
||||
typer.echo("No pages needed a Footnotes block change.")
|
||||
|
||||
if undefined_report:
|
||||
typer.echo("")
|
||||
typer.echo("Undefined [^id] reference(s) - run `cite add` for these, or fix the typo:")
|
||||
for title, ids in undefined_report.items():
|
||||
typer.echo(f" - {title}: {', '.join(ids)}")
|
||||
|
||||
if dry_run:
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
if not touched and not undefined_report:
|
||||
success("Every Footnotes block already matches its page's references.")
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Apply the confidence decay formula defined in the wiki contract's
|
||||
"Confidence Scoring" section: confidence decays at 1% per month since last
|
||||
confirmation (the page's `modified` / `date` / `created` field), floored at 0.2.
|
||||
|
||||
This is pure arithmetic - previously left to the LLM's judgment even though
|
||||
the contract specifies it exactly. Dry-run by default; `--apply` writes changes.
|
||||
|
||||
`confidence` is a *derived* field: it is always recomputed as
|
||||
`confidence_base * (1 - 0.01 * months)`, never from its own previous value.
|
||||
Keeping the undecayed anchor in `confidence_base` is what makes repeated runs
|
||||
idempotent - decaying the stored `confidence` in place (the pre-2026-08-13
|
||||
behavior) compounded on every run, because the elapsed-months factor kept
|
||||
growing while the multiplicand had already shrunk.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import success
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
|
||||
app = typer.Typer(help="Apply confidence decay per the wiki contract's Confidence Scoring formula.")
|
||||
|
||||
DECAY_RATE_PER_MONTH = 0.01
|
||||
FLOOR = 0.2
|
||||
DAYS_PER_MONTH = 30.44
|
||||
|
||||
|
||||
def _parse_date(value) -> Optional[datetime.date]:
|
||||
if isinstance(value, datetime.datetime):
|
||||
return value.date()
|
||||
if isinstance(value, datetime.date):
|
||||
return value
|
||||
if isinstance(value, str):
|
||||
try:
|
||||
return datetime.date.fromisoformat(value)
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def compute_decay(confidence: float, last_confirmed: datetime.date, today: datetime.date) -> float:
|
||||
months = max(0.0, (today - last_confirmed).days / DAYS_PER_MONTH)
|
||||
decayed = confidence * (1 - DECAY_RATE_PER_MONTH * months)
|
||||
return round(max(FLOOR, decayed), 2)
|
||||
|
||||
|
||||
def _last_confirmed(frontmatter: dict) -> Optional[datetime.date]:
|
||||
return (
|
||||
_parse_date(frontmatter.get("modified"))
|
||||
or _parse_date(frontmatter.get("date"))
|
||||
or _parse_date(frontmatter.get("created"))
|
||||
)
|
||||
|
||||
|
||||
@app.command("init-base")
|
||||
def confidence_init_base(
|
||||
apply: bool = typer.Option(False, "--apply", help="Write changes; default is dry-run (preview only)"),
|
||||
):
|
||||
"""Backfill `confidence_base` from the current `confidence` on pages that
|
||||
don't have one yet.
|
||||
|
||||
Needed once, when a wiki predates the derived-`confidence` model. Pages
|
||||
already carrying a base are left untouched, so this is safe to re-run.
|
||||
"""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
changes = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
confidence = page.frontmatter.get("confidence")
|
||||
if confidence is None or page.frontmatter.get("confidence_base") is not None:
|
||||
continue
|
||||
changes.append((title, page, float(confidence)))
|
||||
|
||||
if not changes:
|
||||
success("Every page with a confidence already has a confidence_base.")
|
||||
return
|
||||
|
||||
for title, page, base in changes:
|
||||
typer.echo(f"{title}: confidence_base <- {base:.2f}")
|
||||
if apply:
|
||||
_set_after(page.frontmatter, "confidence", "confidence_base", round(base, 2))
|
||||
write_page(page.path, page.frontmatter, page.body)
|
||||
|
||||
if apply:
|
||||
success(f"Set confidence_base on {len(changes)} page(s).")
|
||||
else:
|
||||
typer.echo(f"\n{len(changes)} page(s) would change. Re-run with --apply to write.")
|
||||
|
||||
|
||||
def _set_after(frontmatter: dict, after_key: str, key: str, value) -> None:
|
||||
"""Insert `key` immediately after `after_key`, preserving frontmatter order
|
||||
(write_page serializes in dict insertion order, and the schemas list
|
||||
confidence_base right after confidence)."""
|
||||
if key in frontmatter or after_key not in frontmatter:
|
||||
frontmatter[key] = value
|
||||
return
|
||||
items = list(frontmatter.items())
|
||||
frontmatter.clear()
|
||||
for existing_key, existing_value in items:
|
||||
frontmatter[existing_key] = existing_value
|
||||
if existing_key == after_key:
|
||||
frontmatter[key] = value
|
||||
|
||||
|
||||
@app.command("decay")
|
||||
def confidence_decay(
|
||||
apply: bool = typer.Option(False, "--apply", help="Write changes; default is dry-run (preview only)"),
|
||||
):
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
today = datetime.date.today()
|
||||
changes = []
|
||||
missing_base = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
confidence = page.frontmatter.get("confidence")
|
||||
if confidence is None:
|
||||
continue
|
||||
base = page.frontmatter.get("confidence_base")
|
||||
if base is None:
|
||||
missing_base.append(title)
|
||||
continue
|
||||
last_confirmed = _last_confirmed(page.frontmatter)
|
||||
if last_confirmed is None:
|
||||
continue
|
||||
new_confidence = compute_decay(float(base), last_confirmed, today)
|
||||
if abs(new_confidence - round(float(confidence), 2)) >= 0.01:
|
||||
changes.append((title, page, float(confidence), new_confidence))
|
||||
|
||||
if missing_base:
|
||||
typer.echo(
|
||||
f"Skipped {len(missing_base)} page(s) with a confidence but no confidence_base "
|
||||
"- run `wikitool confidence init-base --apply` first:"
|
||||
)
|
||||
for title in missing_base[:10]:
|
||||
typer.echo(f" - {title}")
|
||||
if len(missing_base) > 10:
|
||||
typer.echo(f" ... and {len(missing_base) - 10} more")
|
||||
typer.echo("")
|
||||
|
||||
if not changes:
|
||||
success("No confidence values need decaying.")
|
||||
return
|
||||
|
||||
for title, page, old, new in changes:
|
||||
typer.echo(f"{title}: {old:.2f} -> {new:.2f}")
|
||||
if apply:
|
||||
page.frontmatter["confidence"] = new
|
||||
write_page(page.path, page.frontmatter, page.body)
|
||||
|
||||
if apply:
|
||||
success(f"Updated confidence on {len(changes)} page(s).")
|
||||
else:
|
||||
typer.echo(f"\n{len(changes)} page(s) would change. Re-run with --apply to write.")
|
||||
@@ -0,0 +1,442 @@
|
||||
"""`wikitool dist export` - build a distributable, contentless copy of this
|
||||
repo's machinery.
|
||||
|
||||
`export` copies the pipeline's schema/compiler/control-plane layers (types/,
|
||||
tools/, instructions/, the stage contracts, every kb/*/COLLECTION.md) into an
|
||||
empty target, with no kb/ pages, no raw/ content, and no git history - see
|
||||
instructions/setup-instance.md for what happens after. It never calls git.
|
||||
|
||||
Three independent exclusion mechanisms feed the plan, for three different
|
||||
shapes of "does not belong in someone else's instance":
|
||||
|
||||
- Every copied text file passes through `strip_markers()`, which removes any
|
||||
region between `<!-- dist:strip-start -->` and `<!-- dist:strip-end -->`,
|
||||
markers included - for dev-only *content inside* a file that is otherwise
|
||||
shipped (e.g. a routing line in AGENTS.md).
|
||||
- `instructions/dev/` is pruned from the copy wholesale - for dev-only
|
||||
*whole files* (procedures and the skill that switches an agent into
|
||||
tool-development mode). One-way: nothing reconstructs it in a distributed
|
||||
instance, on purpose - see instructions/dev/ itself for the current
|
||||
contents and AGENTS.md's routing line for what a dev instance sees instead.
|
||||
- Build output under `tools/` is dropped, by directory (`TOOLS_EXCLUDE_DIRS`)
|
||||
where it has one, and by filename (`_is_coverage_output`) where it does not.
|
||||
Not dev-only but *derived*: recomputable, and measured against this repo's
|
||||
own test run rather than the receiving instance's.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import stat
|
||||
from pathlib import Path
|
||||
from typing import Callable, NamedTuple, Optional, Union
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, kb_collections, kb_state, version as version_mod
|
||||
from chemenu.commands._util import fail, rel_path, success, today_iso
|
||||
|
||||
app = typer.Typer(help="Build a distributable copy of the wiki machinery.")
|
||||
|
||||
MARKER_START = "<!-- dist:strip-start -->"
|
||||
MARKER_END = "<!-- dist:strip-end -->"
|
||||
|
||||
DIST_TEMPLATES_DIR = Path(__file__).resolve().parent.parent / "dist_templates"
|
||||
|
||||
# Root files copied verbatim (after marker-stripping). INSTALL.md is optional
|
||||
# here: it does not exist until the distribution docs land, and `export`
|
||||
# must not fail just because a later stage of the same repo hasn't shipped
|
||||
# yet.
|
||||
#
|
||||
# The personalization *templates* ship; the filled `USER.md`/`SOUL.md` never
|
||||
# do. This allowlist is what makes that split automatic - a file is copied
|
||||
# because it is named here, so an instance's own personalization is excluded
|
||||
# by construction rather than by a rule someone has to remember.
|
||||
#
|
||||
# `ENVIRONMENT.md.template` rides the same split for the same reason: a
|
||||
# distribution can describe what the file is for, but never what a particular
|
||||
# checkout's harness, MCP servers and remotes are. The filled `ENVIRONMENT.md`
|
||||
# is additionally gitignored, so it is excluded twice over.
|
||||
#
|
||||
# `CLAUDE.md` is harness glue, not a second control plane: Claude Code loads it
|
||||
# and does not load `AGENTS.md`, so it ships for the same reason
|
||||
# `.claude/settings.json` does - a distributed instance running that harness
|
||||
# would otherwise start every session without the control plane.
|
||||
ROOT_FILES = (
|
||||
"AGENTS.md", "CLAUDE.md", "README.md", "EVALS.md", "INSTALL.md", ".gitignore", "VERSION",
|
||||
*config.LICENSE_FILES,
|
||||
*config.PERSONALIZATION_TEMPLATES,
|
||||
config.ENVIRONMENT_TEMPLATE,
|
||||
)
|
||||
|
||||
# The one part of ROOT_FILES that may not be quietly skipped. Every other entry
|
||||
# copies only `if source.is_file()`, which is right for `INSTALL.md` (it did not
|
||||
# exist until the distribution docs landed) and wrong for a licence: an export
|
||||
# that silently omits it hands the receiving instance the AGPL-covered `tools/`
|
||||
# tree with no licence text, which is a violation the moment that instance is
|
||||
# pushed anywhere public. Missing means the export is broken, not minimal.
|
||||
REQUIRED_ROOT_FILES = config.LICENSE_FILES
|
||||
|
||||
# Harness-specific session-tracing config: generic machinery (feeds
|
||||
# tools/chemenu/telemetry/ and tools/trace_ingest.py via EVALS.md), not
|
||||
# personal state - unlike `.obsidian/`/`.vscode/`, which are never copied.
|
||||
HOOK_DIRS = (".github/hooks", ".vibe")
|
||||
|
||||
# tools/ subpaths never copied - build/venv/cache artifacts, not machinery.
|
||||
# `htmlcov/` is coverage.py's HTML report: derived output, and a large tree of
|
||||
# it, measured against the source repo's own test run. `.coveragerc` beside it
|
||||
# *does* ship, the same way `pytest.ini` does - it is configuration, not output.
|
||||
TOOLS_EXCLUDE_DIRS = {".venv", "__pycache__", ".pytest_cache", ".wikitool_session", "htmlcov"}
|
||||
|
||||
# The rest of coverage's output lands beside the code rather than in a directory
|
||||
# of its own - `.coverage`, `coverage.xml`, and `.coverage.<host>.<pid>` under a
|
||||
# parallel run - so a directory exclusion cannot reach it. Same argument as
|
||||
# `reports/`: derived, recomputable, and about the source repo rather than about
|
||||
# the instance that would receive it.
|
||||
COVERAGE_OUTPUT_NAMES = frozenset({".coverage", "coverage.xml"})
|
||||
|
||||
|
||||
def _is_coverage_output(filename: str) -> bool:
|
||||
return filename in COVERAGE_OUTPUT_NAMES or filename.startswith(".coverage.")
|
||||
|
||||
|
||||
# instructions/dev/ holds stack-development-only procedures and the skill
|
||||
# that switches an agent into tool-development mode - never shipped to a
|
||||
# distributed instance. One-way: there is no `enable-dev`-style command that
|
||||
# reconstructs it afterwards, unlike the marker-block content below.
|
||||
INSTRUCTIONS_EXCLUDE_DIRS = {"dev"}
|
||||
|
||||
# Fixed by raw/CONTRACT.md's routing table, unlike kb/'s areas (which are
|
||||
# organic - see kb/CONTRACT.md - so `export` does not manufacture them).
|
||||
RAW_SUBDIRS = ("articles", "documents", "notes", "assets")
|
||||
|
||||
# Stage contracts that are not collections and carry no pages: copied as a
|
||||
# single file each, nothing else from their directory.
|
||||
CONTRACT_ONLY_STAGES = ("raw/CONTRACT.md", "reports/CONTRACT.md", "work/CONTRACT.md")
|
||||
|
||||
# Single tracked files copied out of an otherwise-untouched, partially-ignored
|
||||
# directory. `.claude/` holds the harness's own session-tracing config
|
||||
# (`settings.json`, tracked) alongside generated skill copies and personal
|
||||
# untracked state (`.claude/skills/`, `.claude/settings.local.json`) - neither
|
||||
# of which belongs in a distribution. Adding `.claude` to HOOK_DIRS would copy
|
||||
# the whole directory, skills included; a single-file entry avoids that
|
||||
# without needing an exclude set HOOK_DIRS doesn't otherwise carry.
|
||||
SINGLE_FILES = (".claude/settings.json",)
|
||||
|
||||
Content = Union[str, bytes]
|
||||
|
||||
|
||||
class PlannedFile(NamedTuple):
|
||||
content: Content
|
||||
executable: bool = False
|
||||
|
||||
|
||||
_MARKER_TOKEN_RE = re.compile(re.escape(MARKER_START) + "|" + re.escape(MARKER_END))
|
||||
# The leading/trailing `\n?` consume the blank line on each side of the
|
||||
# block - the convention is that a marker block always sits as its own
|
||||
# paragraph. Without eating both, a strip leaves two blank lines where the
|
||||
# clean file only ever had one.
|
||||
_MARKER_BLOCK_RE = re.compile(
|
||||
r"\n?" + re.escape(MARKER_START) + r".*?" + re.escape(MARKER_END) + r"\n?", re.DOTALL
|
||||
)
|
||||
|
||||
|
||||
def _validate_markers(text: str, label: str) -> None:
|
||||
"""A marker file must be a sequence of well-formed, non-nested
|
||||
start/end pairs. Malformed markers would make `strip_markers` remove
|
||||
either too little or too much, silently - this fails loudly instead."""
|
||||
depth = 0
|
||||
for match in _MARKER_TOKEN_RE.finditer(text):
|
||||
if match.group() == MARKER_START:
|
||||
if depth != 0:
|
||||
fail(f"{label}: nested dist:strip-start markers are not supported")
|
||||
depth = 1
|
||||
else:
|
||||
if depth != 1:
|
||||
fail(f"{label}: dist:strip-end without a matching dist:strip-start")
|
||||
depth = 0
|
||||
if depth != 0:
|
||||
fail(f"{label}: dist:strip-start without a matching dist:strip-end")
|
||||
|
||||
|
||||
def strip_markers(text: str) -> str:
|
||||
"""Remove every marked region, markers included. Generic by design: it
|
||||
does not matter what is inside, or how many regions a file has."""
|
||||
return _MARKER_BLOCK_RE.sub("", text)
|
||||
|
||||
|
||||
def _is_executable(path: Path) -> bool:
|
||||
return bool(path.stat().st_mode & stat.S_IXUSR)
|
||||
|
||||
|
||||
def _read_planned_file(path: Path, label: str) -> PlannedFile:
|
||||
executable = _is_executable(path)
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return PlannedFile(path.read_bytes(), executable)
|
||||
# Marker syntax is an HTML/Markdown comment convention, scoped to .md
|
||||
# files on purpose: applying it to every text file would let the marker
|
||||
# strings themselves - inline here as Python string literals - match as
|
||||
# a region in this file's own source when tools/ gets copied, and eat
|
||||
# the code between them.
|
||||
if path.suffix != ".md":
|
||||
return PlannedFile(text, executable)
|
||||
_validate_markers(text, label)
|
||||
return PlannedFile(strip_markers(text), executable)
|
||||
|
||||
|
||||
def _copy_tree(
|
||||
source_root: Path,
|
||||
dest_prefix: str,
|
||||
exclude_dirs: frozenset[str],
|
||||
exclude_file: Optional[Callable[[str], bool]] = None,
|
||||
) -> dict[str, PlannedFile]:
|
||||
"""Every file under source_root, marker-stripped, keyed by its
|
||||
destination-relative path. Excluded directories are pruned during the
|
||||
walk rather than filtered after, so a large `.venv/` is never read.
|
||||
`exclude_file` drops individual files by name, for output that lands
|
||||
beside the code instead of in a directory a prune could catch."""
|
||||
files: dict[str, PlannedFile] = {}
|
||||
if not source_root.is_dir():
|
||||
return files
|
||||
for dirpath, dirnames, filenames in os.walk(source_root):
|
||||
dirnames[:] = sorted(d for d in dirnames if d not in exclude_dirs)
|
||||
for filename in sorted(filenames):
|
||||
if exclude_file is not None and exclude_file(filename):
|
||||
continue
|
||||
path = Path(dirpath) / filename
|
||||
relative = path.relative_to(source_root).as_posix()
|
||||
dest_rel = f"{dest_prefix}/{relative}"
|
||||
files[dest_rel] = _read_planned_file(path, dest_rel)
|
||||
return files
|
||||
|
||||
|
||||
def _digest(content: Content) -> str:
|
||||
data = content if isinstance(content, bytes) else content.encode("utf-8")
|
||||
return "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
|
||||
|
||||
def build_stamp(plan: dict[str, PlannedFile], origin: "Origin") -> str:
|
||||
"""The release stamp written into every export.
|
||||
|
||||
Two jobs. The version and origin fields are what `version check` compares
|
||||
against a release feed - without them an instance cannot tell which stack
|
||||
it is running. The per-file digests are for the update *after* detection:
|
||||
they record what the machinery looked like when it was installed, which is
|
||||
the only way a later upgrade can tell a file the instance edited from one
|
||||
it merely received. Nothing reads them today; writing them now is what
|
||||
keeps that upgrade from needing a format change.
|
||||
"""
|
||||
stamp = {
|
||||
"schema": version_mod.STAMP_SCHEMA,
|
||||
"version": str(version_mod.read_version()),
|
||||
"exported_at": today_iso(),
|
||||
"source_repo": origin.source_repo,
|
||||
"source_commit": origin.source_commit,
|
||||
"release_url": origin.release_url,
|
||||
"update_url": origin.update_url or version_mod.DEFAULT_UPDATE_URL,
|
||||
"files": {relative: _digest(planned.content) for relative, planned in sorted(plan.items())},
|
||||
}
|
||||
return json.dumps(stamp, indent=2, sort_keys=False) + "\n"
|
||||
|
||||
|
||||
class Origin(NamedTuple):
|
||||
"""Where this export came from. Supplied by the caller (the release
|
||||
workflow knows the commit and the release URL); `dist export` itself never
|
||||
calls git, so it cannot discover any of it."""
|
||||
|
||||
source_repo: Optional[str] = None
|
||||
source_commit: Optional[str] = None
|
||||
release_url: Optional[str] = None
|
||||
update_url: Optional[str] = None
|
||||
|
||||
|
||||
def build_plan(origin: Optional[Origin] = None) -> dict[str, PlannedFile]:
|
||||
"""Every (destination-relative path -> planned file) the export writes."""
|
||||
plan: dict[str, PlannedFile] = {}
|
||||
|
||||
missing_licences = [
|
||||
name for name in REQUIRED_ROOT_FILES if not (config.ROOT / name).is_file()
|
||||
]
|
||||
if missing_licences:
|
||||
fail(
|
||||
"export would ship code without its licence: "
|
||||
+ ", ".join(missing_licences)
|
||||
+ " missing from the source tree. Restore them before exporting - a "
|
||||
"distribution carrying tools/ without LICENSE is a copyleft violation "
|
||||
"the moment the receiving instance is published."
|
||||
)
|
||||
|
||||
for name in ROOT_FILES:
|
||||
source = config.ROOT / name
|
||||
if source.is_file():
|
||||
plan[name] = _read_planned_file(source, name)
|
||||
|
||||
plan.update(_copy_tree(config.INSTRUCTIONS_DIR, "instructions", frozenset(INSTRUCTIONS_EXCLUDE_DIRS)))
|
||||
plan.update(_copy_tree(config.TYPES_DIR, "types", frozenset()))
|
||||
plan.update(_copy_tree(
|
||||
config.ROOT / "tools", "tools", frozenset(TOOLS_EXCLUDE_DIRS), _is_coverage_output
|
||||
))
|
||||
for hook_dir in HOOK_DIRS:
|
||||
plan.update(_copy_tree(config.ROOT / hook_dir, hook_dir, frozenset()))
|
||||
|
||||
kb_contract = config.KB_DIR / "CONTRACT.md"
|
||||
if kb_contract.is_file():
|
||||
plan["kb/CONTRACT.md"] = _read_planned_file(kb_contract, "kb/CONTRACT.md")
|
||||
for collection in kb_collections.iter_kb_collections():
|
||||
rel = f"kb/{collection.name}/COLLECTION.md"
|
||||
plan[rel] = _read_planned_file(collection / "COLLECTION.md", rel)
|
||||
|
||||
for relative in CONTRACT_ONLY_STAGES:
|
||||
source = config.ROOT / relative
|
||||
if source.is_file():
|
||||
plan[relative] = _read_planned_file(source, relative)
|
||||
|
||||
for relative in SINGLE_FILES:
|
||||
source = config.ROOT / relative
|
||||
if source.is_file():
|
||||
plan[relative] = _read_planned_file(source, relative)
|
||||
|
||||
for sub in RAW_SUBDIRS:
|
||||
plan[f"raw/{sub}/.gitkeep"] = PlannedFile("")
|
||||
|
||||
plan["kb/log.md"] = PlannedFile((DIST_TEMPLATES_DIR / "log.md").read_text(encoding="utf-8"))
|
||||
plan["CHANGES.md"] = PlannedFile((DIST_TEMPLATES_DIR / "CHANGES.md").read_text(encoding="utf-8"))
|
||||
|
||||
# A fresh instance's content is empty, so it is trivially in the shape this
|
||||
# machinery expects - which is exactly what makes the initial declaration
|
||||
# safe to write here rather than leaving it to `migrate baseline`. Only an
|
||||
# instance predating this file has to answer that question by hand.
|
||||
plan[kb_state.KB_STATE_FILENAME] = PlannedFile(
|
||||
kb_state.render_kb_state(version_mod.read_version(), [])
|
||||
)
|
||||
|
||||
# Last, so it can digest everything above it. It is the one file in the
|
||||
# export that describes the export rather than being copied into it.
|
||||
plan[version_mod.RELEASE_STAMP_FILENAME] = PlannedFile(
|
||||
build_stamp(plan, origin or Origin())
|
||||
)
|
||||
|
||||
return plan
|
||||
|
||||
|
||||
# Content that must never appear in a plan, expressed structurally rather than
|
||||
# by matching text. Three allowlists feed `build_plan`, and each one holds only
|
||||
# because someone remembered the rule when they edited it - nothing re-checks
|
||||
# the result. This does.
|
||||
#
|
||||
# The checks are deliberately structural: a filled personalization file, a kb
|
||||
# page, a raw source, a dev-only instruction. A text-pattern scan (hostnames,
|
||||
# IP literals) was considered and rejected - the project's own host legitimately
|
||||
# appears in INSTALL.md and version.py, so such a scan would either whitelist
|
||||
# the very string it is looking for or cry wolf on every export.
|
||||
_CONTENT_PREFIXES = ("kb/", "raw/")
|
||||
_CONTENT_ALLOWED_NAMES = ("CONTRACT.md", "COLLECTION.md", "log.md", ".gitkeep")
|
||||
|
||||
|
||||
def find_leaks(plan: dict[str, PlannedFile]) -> list[str]:
|
||||
"""Planned paths that carry one instance's own data instead of machinery."""
|
||||
leaks: list[str] = []
|
||||
for relative in sorted(plan):
|
||||
name = relative.rsplit("/", 1)[-1]
|
||||
if name in config.PERSONALIZATION_FILES or name == config.ENVIRONMENT_FILE:
|
||||
leaks.append(f"{relative} (one instance's own personalization)")
|
||||
elif relative.startswith("instructions/dev/"):
|
||||
leaks.append(f"{relative} (stack-development only)")
|
||||
elif relative.startswith(_CONTENT_PREFIXES) and name not in _CONTENT_ALLOWED_NAMES:
|
||||
leaks.append(f"{relative} (wiki content, not machinery)")
|
||||
return leaks
|
||||
|
||||
|
||||
def _write_plan(target: Path, plan: dict[str, PlannedFile]) -> None:
|
||||
for relative, planned in plan.items():
|
||||
dest = target / relative
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
if isinstance(planned.content, bytes):
|
||||
dest.write_bytes(planned.content)
|
||||
else:
|
||||
dest.write_text(planned.content, encoding="utf-8")
|
||||
if planned.executable:
|
||||
dest.chmod(dest.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
|
||||
|
||||
|
||||
@app.command("export")
|
||||
def export_command(
|
||||
target: Path = typer.Argument(
|
||||
..., help="Directory to write the distribution into. Must not exist, or must be empty."
|
||||
),
|
||||
dry_run: bool = typer.Option(
|
||||
False, "--dry-run", help="List what would be written, without writing anything."
|
||||
),
|
||||
source_repo: Optional[str] = typer.Option(
|
||||
None, "--source-repo", help="Repository this export was built from (recorded in the stamp)"
|
||||
),
|
||||
source_commit: Optional[str] = typer.Option(
|
||||
None, "--source-commit", help="Commit this export was built from (recorded in the stamp)"
|
||||
),
|
||||
release_url: Optional[str] = typer.Option(
|
||||
None, "--release-url", help="Release page this export ships as (recorded in the stamp)"
|
||||
),
|
||||
update_url: Optional[str] = typer.Option(
|
||||
None, "--update-url", help="Release feed `version check` should ask (recorded in the stamp)"
|
||||
),
|
||||
):
|
||||
"""Export a contentless, distributable copy of this repo's machinery:
|
||||
AGENTS.md/README.md (dev-instance-only marker blocks removed),
|
||||
instructions/ (no instructions/dev/), types/, tools/ (no venv/caches),
|
||||
the .github/hooks/+.vibe session-tracing config plus .claude/settings.json,
|
||||
every kb/*/COLLECTION.md (no pages, no areas), empty
|
||||
raw/{articles,documents,notes,assets}/, VERSION, the USER.md/SOUL.md
|
||||
personalization templates (never the filled files), and a
|
||||
.wikitool-release.json stamp. The --source-*/--release-url/--update-url
|
||||
options only fill fields in that stamp: `export` never calls git and cannot
|
||||
discover them. See instructions/setup-instance.md for what comes next."""
|
||||
run_export(
|
||||
target,
|
||||
dry_run=dry_run,
|
||||
origin=Origin(
|
||||
source_repo=source_repo,
|
||||
source_commit=source_commit,
|
||||
release_url=release_url,
|
||||
update_url=update_url,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def run_export(target: Path, dry_run: bool = False, origin: Optional[Origin] = None) -> None:
|
||||
"""The export itself, free of Typer's option objects so it can be called
|
||||
directly - by the command above, and by the tests."""
|
||||
target = target.resolve()
|
||||
if target.exists():
|
||||
if not target.is_dir():
|
||||
fail(f"{target} exists and is not a directory.")
|
||||
if any(target.iterdir()):
|
||||
fail(f"{target} is not empty. `dist export` refuses to write into a non-empty directory.")
|
||||
|
||||
try:
|
||||
plan = build_plan(origin)
|
||||
except version_mod.VersionError as exc:
|
||||
fail(f"{exc} - a distribution must carry the version it ships.")
|
||||
return
|
||||
|
||||
leaks = find_leaks(plan)
|
||||
if leaks:
|
||||
fail(
|
||||
"export would carry this instance's own data, not just machinery:\n "
|
||||
+ "\n ".join(leaks)
|
||||
+ "\nThis is an allowlist bug in dist_cmd.py, not something to work "
|
||||
"around - fix the allowlist rather than deleting files from the target."
|
||||
)
|
||||
return
|
||||
|
||||
if dry_run:
|
||||
for relative in sorted(plan):
|
||||
typer.echo(f"write {relative}")
|
||||
success(f"Dry run: would write {len(plan)} file(s) to {target}. Nothing written.")
|
||||
return
|
||||
|
||||
_write_plan(target, plan)
|
||||
success(f"Exported {len(plan)} file(s) to {rel_path(target)}.")
|
||||
@@ -0,0 +1,499 @@
|
||||
"""`wikitool docs verify` - machine-check the documentation copies that can be
|
||||
re-derived from the code and the repo layout.
|
||||
|
||||
The wiki's own rule is that a derived copy of recomputable truth must be
|
||||
checked or absent. Three such copies survive on purpose because they earn
|
||||
their keep as reading material:
|
||||
|
||||
1. `tools/CONTRACT.md`'s command table (re-derivable from the Typer app)
|
||||
2. the collection and stage contracts (their existence and placement, not
|
||||
their content)
|
||||
3. the absence of pre-type-system `type: entity` frontmatter in the
|
||||
contract docs - the exact drift that left a stale comparison template
|
||||
sitting in AGENTS.md for months after the type migration
|
||||
|
||||
A fourth check has a different shape: `.gitignore` is not documentation, but
|
||||
it is the one file that can silently un-publish content. A pattern excluding a
|
||||
file under `raw/` or `kb/` is a data-loss bug - `sources coverage` reads the
|
||||
filesystem and reports the file as covered, while `publish` (`git add -A`)
|
||||
never commits it, so a fresh clone has a broken `raw_files:` reference. The
|
||||
same check runs in reverse over `reports/`, where a *missing* ignore rule would
|
||||
start committing derived output.
|
||||
|
||||
A fifth has the same shape as the fourth: `VERSION` is not documentation
|
||||
either, but it is the one number a release stamps into every distributed
|
||||
instance, and a version raised without a changelog entry ships release notes
|
||||
that describe the previous release.
|
||||
|
||||
Everything here is a hard oracle: a set comparison or a regex, no judgment.
|
||||
Content quality of the contracts themselves stays with the LLM.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, kb_collections, version as version_mod
|
||||
from chemenu.commands._util import fail, rel_path, success
|
||||
|
||||
app = typer.Typer(help="Verify documentation that mirrors the code or repo layout.")
|
||||
|
||||
# Contracts that are not COLLECTION.md files, because their directories are not
|
||||
# collections. Each is the authoring contract for one stage or layer.
|
||||
STAGE_CONTRACTS = (
|
||||
"raw/CONTRACT.md",
|
||||
"kb/CONTRACT.md",
|
||||
"types/type-spec.md",
|
||||
"reports/CONTRACT.md",
|
||||
"work/CONTRACT.md",
|
||||
"tools/CONTRACT.md",
|
||||
"instructions/CONTRACT.md",
|
||||
)
|
||||
|
||||
# Directories whose contents are the repository's reason to exist, and which
|
||||
# therefore may never be excluded by an ignore rule. `work/` is here because a
|
||||
# workshop is the only record of a multi-session run: unlike `reports/`, losing
|
||||
# it loses judgment that nothing can recompute.
|
||||
CONTENT_DIRS = ("raw", "kb", "work")
|
||||
|
||||
# Paths that must never be ignored. They deliberately do not have to exist:
|
||||
# `git check-ignore --no-index` answers about the *pattern set*, not the
|
||||
# filesystem, so these catch a trap before a real file ever falls into it.
|
||||
# Every entry corresponds to a pattern that was genuinely swallowing content
|
||||
# before the 2026-08-13 `.gitignore` rewrite.
|
||||
IGNORE_CANARIES = (
|
||||
"raw/notes/template.md", # was caught by `*temp*`
|
||||
"raw/notes/temperature-sensors.md", # was caught by `*temp*`
|
||||
"raw/notes/scratch.md", # was caught by `*scratch*`
|
||||
"raw/assets/build.log", # was caught by `*.log`
|
||||
"raw/documents/go.mod", # was caught by `go.mod`
|
||||
"raw/assets/bin/tool.txt", # was caught by `bin/`
|
||||
"raw/assets/diagram.orig", # was caught by `*.orig`
|
||||
"kb/concepts/Template Method.md", # was caught by `*temp*`
|
||||
"kb/entities/tools/core.md", # was caught by `core`
|
||||
"kb/entities/tools/tags.md", # was caught by `tags`
|
||||
# `work/` is tracked on purpose: unlike `reports/`, a workshop holds
|
||||
# judgment in progress that nothing can recompute, so an ignore rule
|
||||
# reaching it would silently discard a multi-session run's only record.
|
||||
"work/ingest-documents-example/extract-00-architecture.md",
|
||||
)
|
||||
|
||||
# The mirror image of IGNORE_CANARIES. `reports/` holds derived output that must
|
||||
# stay *out* of git, so an ignore rule going missing there is as much a bug as an
|
||||
# ignore rule appearing over content - it would start committing a second,
|
||||
# drifting copy of something `wikitool lint` recomputes on demand. The contract
|
||||
# is the one file that must survive the rule.
|
||||
#
|
||||
# The skill directories are here for a different reason: they are copies of
|
||||
# `instructions/<name>/SKILL.md`, published by `wikitool instructions sync`.
|
||||
# Committing them would create exactly the drifting second copy this repo
|
||||
# refuses to keep anywhere else.
|
||||
#
|
||||
# `ENVIRONMENT.md` is a third reason again: it is per-checkout, so committing
|
||||
# one working copy's harness, MCP servers and remotes would hand every other
|
||||
# clone a file that is confidently wrong rather than honestly absent. Its
|
||||
# `.template` sits in REQUIRED_TRACKED_PATHS below, because the obvious
|
||||
# careless pattern (`ENVIRONMENT.md*`) would swallow both.
|
||||
#
|
||||
# The coverage paths are the reports/ argument applied to `pytest --cov`
|
||||
# output: derived, recomputable, and in the way of `publish`'s `git add -A`.
|
||||
REQUIRED_IGNORE_CANARIES = (
|
||||
"reports/Lint Report 2026-01-01.md",
|
||||
".agents/skills/wiki-query/SKILL.md",
|
||||
".claude/skills/wiki-query/SKILL.md",
|
||||
"ENVIRONMENT.md",
|
||||
"tools/coverage.xml",
|
||||
"tools/htmlcov/index.html",
|
||||
)
|
||||
REQUIRED_TRACKED_PATHS = (
|
||||
"reports/CONTRACT.md",
|
||||
"instructions/CONTRACT.md",
|
||||
"instructions/wiki-query/SKILL.md",
|
||||
"ENVIRONMENT.md.template",
|
||||
)
|
||||
|
||||
CLI_README = config.ROOT / "tools" / "CONTRACT.md"
|
||||
|
||||
# The root README is the "absent" half of the checked-or-absent rule: it used to
|
||||
# carry its own copy of the command table, which drifted because nothing
|
||||
# compared it to anything. It now points at tools/CONTRACT.md instead, and this
|
||||
# check keeps it that way.
|
||||
ROOT_README = config.ROOT / "README.md"
|
||||
|
||||
# `README.md` is for humans, `CONTRACT.md` is the agent-facing contract, and a
|
||||
# stage may carry both. The split only holds while the README stays prose: the
|
||||
# first thing that drifted last time was a second copy of the command table, and
|
||||
# tools/README.md is exactly the file it drifted in. INSTALL.md is here for the
|
||||
# same reason: it is human-facing prose about installing an instance, and the
|
||||
# command reference lives exactly once, in tools/CONTRACT.md.
|
||||
STAGE_READMES = ("tools/README.md", "INSTALL.md")
|
||||
|
||||
# Docs that must not re-introduce the pre-migration bare-enum `type:` form.
|
||||
# The per-collection contracts are appended at call time, since which ones exist
|
||||
# is a filesystem question rather than a constant.
|
||||
TYPE_GUARD_DOCS = ("AGENTS.md", "README.md", *STAGE_CONTRACTS)
|
||||
|
||||
LEGACY_TYPE_RE = re.compile(r"^type:\s*(entity|concept|source|comparison)\s*$", re.MULTILINE)
|
||||
|
||||
# First backticked cell of a markdown table row, e.g. "| `xref add --a ...` | ... |"
|
||||
TABLE_CELL_RE = re.compile(r"^\|\s*`([^`]+)`", re.MULTILINE)
|
||||
|
||||
|
||||
def registered_commands() -> set[str]:
|
||||
"""Every command path the CLI exposes, e.g. {'new', 'xref add', ...}.
|
||||
|
||||
Imported lazily: `chemenu.cli` imports this module, so a top-level
|
||||
import would be circular.
|
||||
"""
|
||||
from chemenu import cli
|
||||
|
||||
paths: set[str] = set()
|
||||
for command in cli.app.registered_commands:
|
||||
name = command.name or (command.callback.__name__.replace("_", "-") if command.callback else None)
|
||||
if name:
|
||||
paths.add(name)
|
||||
for group in cli.app.registered_groups:
|
||||
group_name = group.name
|
||||
sub_app = group.typer_instance
|
||||
if not group_name or sub_app is None:
|
||||
continue
|
||||
for command in sub_app.registered_commands:
|
||||
name = command.name or (command.callback.__name__.replace("_", "-") if command.callback else None)
|
||||
if name:
|
||||
paths.add(f"{group_name} {name}")
|
||||
return paths
|
||||
|
||||
|
||||
def top_level_names() -> set[str]:
|
||||
return {path.split(" ", 1)[0] for path in registered_commands()}
|
||||
|
||||
|
||||
def documented_commands(readme_text: str) -> list[str]:
|
||||
return [match.group(1).strip() for match in TABLE_CELL_RE.finditer(readme_text)]
|
||||
|
||||
|
||||
def check_cli_readme() -> list[str]:
|
||||
"""Every registered command must appear in tools/CONTRACT.md's command
|
||||
table, and every command documented there must exist.
|
||||
|
||||
The reverse check matches a documented cell against the full registered
|
||||
command path (e.g. `xref add`, `confidence init-base`), not just its first
|
||||
token - checking only the top-level word would let a typo'd or invented
|
||||
subcommand (`xref frobnicate`) sit undetected next to a real command group
|
||||
(`xref`) forever.
|
||||
"""
|
||||
if not CLI_README.exists():
|
||||
return [f"{CLI_README.relative_to(config.ROOT)} is missing"]
|
||||
|
||||
text = CLI_README.read_text(encoding="utf-8")
|
||||
cells = documented_commands(text)
|
||||
issues = []
|
||||
|
||||
registered = sorted(registered_commands())
|
||||
for command_path in registered:
|
||||
if not any(cell == command_path or cell.startswith(command_path + " ") for cell in cells):
|
||||
issues.append(f"command `{command_path}` is not documented in tools/CONTRACT.md")
|
||||
|
||||
for cell in cells:
|
||||
if not any(cell == cp or cell.startswith(cp + " ") for cp in registered):
|
||||
first_token = cell.split(" ", 1)[0]
|
||||
issues.append(f"tools/CONTRACT.md documents `{cell}`, but `{first_token}` is not a wikitool command")
|
||||
|
||||
return issues
|
||||
|
||||
|
||||
def check_collection_contracts() -> list[str]:
|
||||
"""The three structural rules that define what a collection is.
|
||||
|
||||
Collections are discovered by contract presence rather than listed here, so
|
||||
`mkdir kb/<name>` + a COLLECTION.md is all it takes to add one. That only
|
||||
works if the inverse is also checked: a directory under kb/ *without* a
|
||||
contract is an unclaimed subtree whose pages obey no local rules, and a
|
||||
contract outside kb/ quietly widens "collection" back out to "any directory".
|
||||
"""
|
||||
issues = []
|
||||
|
||||
collections = {path.name for path in kb_collections.iter_kb_collections()}
|
||||
if config.KB_DIR.is_dir():
|
||||
for child in sorted(config.KB_DIR.iterdir()):
|
||||
if child.is_dir() and child.name not in collections:
|
||||
issues.append(
|
||||
f"kb/{child.name}/ has no COLLECTION.md - every directory under kb/ is a "
|
||||
"collection and needs its own authoring contract"
|
||||
)
|
||||
|
||||
for stray in kb_collections.stray_collection_contracts():
|
||||
relative = stray.relative_to(config.ROOT)
|
||||
if kb_collections.kb_collection_of(stray.parent) is not None:
|
||||
issues.append(
|
||||
f"{relative} is nested inside a collection - a subdirectory is an area and "
|
||||
"inherits the enclosing contract"
|
||||
)
|
||||
else:
|
||||
issues.append(
|
||||
f"{relative} is outside kb/ - only kb/ holds collections; other directories "
|
||||
"carry a CONTRACT.md instead"
|
||||
)
|
||||
|
||||
for relative_path in STAGE_CONTRACTS:
|
||||
if not (config.ROOT / relative_path).exists():
|
||||
issues.append(f"{relative_path} is missing - it is the authoring contract for its stage")
|
||||
|
||||
return issues
|
||||
|
||||
|
||||
def check_legacy_type_blocks() -> list[str]:
|
||||
issues = []
|
||||
guarded = [
|
||||
*TYPE_GUARD_DOCS,
|
||||
*(
|
||||
str((path / "COLLECTION.md").relative_to(config.ROOT))
|
||||
for path in kb_collections.iter_kb_collections()
|
||||
),
|
||||
]
|
||||
for relative_path in guarded:
|
||||
path = config.ROOT / relative_path
|
||||
if not path.exists():
|
||||
continue
|
||||
for match in LEGACY_TYPE_RE.finditer(path.read_text(encoding="utf-8")):
|
||||
line_number = path.read_text(encoding="utf-8")[: match.start()].count("\n") + 1
|
||||
issues.append(
|
||||
f"{relative_path}:{line_number} uses the pre-migration `type: {match.group(1)}` form "
|
||||
f"- pages reference types by path (`types/{match.group(1)}.md`)"
|
||||
)
|
||||
return issues
|
||||
|
||||
|
||||
def command_table_free_readmes() -> list[Path]:
|
||||
"""Every README that must not carry a copy of the command table.
|
||||
|
||||
Built at call time rather than at import, so a test can point ROOT_README at
|
||||
a fixture.
|
||||
"""
|
||||
return [ROOT_README, *(config.ROOT / relative for relative in STAGE_READMES)]
|
||||
|
||||
|
||||
def check_readmes_have_no_command_table() -> list[str]:
|
||||
"""No README may re-list wikitool commands in a table.
|
||||
|
||||
A derived copy of recomputable truth is either checked or absent. The
|
||||
command table is checked in tools/CONTRACT.md, so a second copy in a README
|
||||
has to be absent - otherwise it drifts silently, which is exactly what it
|
||||
did.
|
||||
"""
|
||||
known_top_level = top_level_names()
|
||||
issues = []
|
||||
for readme in command_table_free_readmes():
|
||||
if not readme.exists():
|
||||
continue
|
||||
offenders = sorted(
|
||||
{
|
||||
cell
|
||||
for cell in documented_commands(readme.read_text(encoding="utf-8"))
|
||||
if cell.split(" ", 1)[0] in known_top_level
|
||||
}
|
||||
)
|
||||
issues += [
|
||||
f"{rel_path(readme)} has a table row for `{cell}` - the command reference lives in "
|
||||
"tools/CONTRACT.md, which `docs verify` checks; link to it instead of copying it"
|
||||
for cell in offenders
|
||||
]
|
||||
return issues
|
||||
|
||||
|
||||
def _git(args: list[str], stdin: Optional[str] = None) -> Optional[subprocess.CompletedProcess]:
|
||||
"""Run a git command in the repo root, or return None if git is unavailable
|
||||
or this is not a checkout. Returning None (rather than raising) keeps
|
||||
`docs verify` usable in a source tree without git, where the ignore rules
|
||||
are unknowable rather than wrong."""
|
||||
try:
|
||||
return subprocess.run(
|
||||
["git", *args], cwd=config.ROOT, capture_output=True, text=True, input=stdin
|
||||
)
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
|
||||
def _check_ignore(paths: tuple[str, ...]) -> Optional[list[str]]:
|
||||
"""The subset of `paths` the repo's ignore rules would exclude, or None if
|
||||
git cannot answer.
|
||||
|
||||
`--no-index` makes this a pure question about the pattern set: it does not
|
||||
matter whether the path exists or is tracked, only whether a rule would
|
||||
swallow it. That is what turns a latent trap into a failing check.
|
||||
|
||||
None and `[]` have to stay distinguishable. For the forward canaries an
|
||||
unknowable answer and an empty answer both mean "no finding", but the
|
||||
reverse canaries assert that a path *is* ignored - so collapsing None into
|
||||
`[]` would turn a missing git binary into a fabricated failure.
|
||||
"""
|
||||
result = _git(["check-ignore", "--no-index", "-z", "--stdin"], stdin="\0".join(paths))
|
||||
if result is None or result.returncode not in (0, 1):
|
||||
return None
|
||||
return [path for path in result.stdout.split("\0") if path]
|
||||
|
||||
|
||||
def ignored_canaries(canaries: tuple[str, ...] = IGNORE_CANARIES) -> list[str]:
|
||||
"""The subset of `canaries` the ignore rules would exclude; empty if
|
||||
unknowable."""
|
||||
return _check_ignore(canaries) or []
|
||||
|
||||
|
||||
def ignored_content_files() -> list[str]:
|
||||
"""Files that actually exist under a CONTENT_DIRS directory but are ignored,
|
||||
and so would never be committed by `wikitool publish`."""
|
||||
result = _git(
|
||||
["ls-files", "--others", "--ignored", "--exclude-standard", "-z", "--", *CONTENT_DIRS]
|
||||
)
|
||||
if result is None or result.returncode != 0:
|
||||
return []
|
||||
return [path for path in result.stdout.split("\0") if path]
|
||||
|
||||
|
||||
def check_ignored_content() -> list[str]:
|
||||
"""No file under `raw/`, `kb/` or `work/` may be excluded by an ignore rule,
|
||||
and everything under `reports/` except its README must be."""
|
||||
issues = [
|
||||
f"`{path}` exists but is gitignored - `wikitool publish` will never commit it"
|
||||
for path in ignored_content_files()
|
||||
]
|
||||
issues += [
|
||||
f"an ignore rule would swallow `{path}` - anchor the pattern in .gitignore "
|
||||
"(see its header note) so content cannot be silently un-published"
|
||||
for path in ignored_canaries()
|
||||
]
|
||||
|
||||
still_ignored = _check_ignore(REQUIRED_IGNORE_CANARIES)
|
||||
if still_ignored is not None:
|
||||
issues += [
|
||||
f"`{path}` is NOT ignored - generated reports must stay out of git, or they become "
|
||||
"a second copy of what `wikitool lint` recomputes on demand"
|
||||
for path in REQUIRED_IGNORE_CANARIES
|
||||
if path not in still_ignored
|
||||
]
|
||||
|
||||
wrongly_ignored = _check_ignore(REQUIRED_TRACKED_PATHS)
|
||||
if wrongly_ignored is not None:
|
||||
issues += [
|
||||
f"`{path}` is ignored - it must survive the reports/ ignore rule"
|
||||
for path in REQUIRED_TRACKED_PATHS
|
||||
if path in wrongly_ignored
|
||||
]
|
||||
return issues
|
||||
|
||||
|
||||
def check_version_changelog() -> list[str]:
|
||||
"""`VERSION` must parse, and the newest versioned `CHANGES.md` entry must
|
||||
name it.
|
||||
|
||||
This is the check that makes `version bump` more than a convenience: a
|
||||
version raised with nothing written about it would ship a release whose
|
||||
notes describe the previous one. A changelog with *no* versioned entry at
|
||||
all is fine - that is a fresh distribution, and this repo's own pre-
|
||||
versioning history, neither of which claims to describe the current
|
||||
version.
|
||||
"""
|
||||
version_path = config.ROOT / version_mod.VERSION_FILENAME
|
||||
if not version_path.is_file():
|
||||
return [
|
||||
f"{version_mod.VERSION_FILENAME} is missing - the stack has no version for "
|
||||
"`dist export` to stamp or `version check` to compare"
|
||||
]
|
||||
try:
|
||||
declared = version_mod.Version.parse(version_path.read_text(encoding="utf-8"))
|
||||
except version_mod.VersionError as exc:
|
||||
return [f"{version_mod.VERSION_FILENAME}: {exc}"]
|
||||
|
||||
changes_path = config.ROOT / version_mod.CHANGES_FILENAME
|
||||
if not changes_path.is_file():
|
||||
return [f"{version_mod.CHANGES_FILENAME} is missing - a version has nowhere to be explained"]
|
||||
|
||||
documented = version_mod.top_changes_version(changes_path.read_text(encoding="utf-8"))
|
||||
if documented is not None and documented != declared:
|
||||
return [
|
||||
f"{version_mod.VERSION_FILENAME} says {declared}, but the newest versioned "
|
||||
f"{version_mod.CHANGES_FILENAME} entry is {documented} - run "
|
||||
"`wikitool version bump` (which writes both), or fix whichever is wrong"
|
||||
]
|
||||
return []
|
||||
|
||||
|
||||
def _second_changes_version(text: str) -> Optional["version_mod.Version"]:
|
||||
"""The version named by the second-newest versioned entry, or None."""
|
||||
seen = [
|
||||
version_mod.Version.parse(match.group(1))
|
||||
for match in version_mod._CHANGES_ENTRY_RE.finditer(text)
|
||||
]
|
||||
return seen[1] if len(seen) > 1 else None
|
||||
|
||||
|
||||
def check_migration_for_boundary() -> list[str]:
|
||||
"""A version that crosses the compatibility boundary must say how to cross it.
|
||||
|
||||
`version check` tells an instance that it must migrate. Without this, that
|
||||
is where the trail ends - the instance knows it is behind and nothing tells
|
||||
it what to do. So a boundary-crossing version needs either a migration
|
||||
document targeting it, or an explicit statement in its changelog entry that
|
||||
no content has to change.
|
||||
|
||||
Only the newest entry is checked. Older boundaries were either satisfied
|
||||
when they were written or cannot be fixed retroactively, and re-reporting
|
||||
them forever would make the check noise.
|
||||
"""
|
||||
from chemenu import kb_state
|
||||
|
||||
changes_path = config.ROOT / version_mod.CHANGES_FILENAME
|
||||
version_path = config.ROOT / version_mod.VERSION_FILENAME
|
||||
if not changes_path.is_file() or not version_path.is_file():
|
||||
return [] # already reported by check_version_changelog
|
||||
|
||||
text = changes_path.read_text(encoding="utf-8")
|
||||
current = version_mod.top_changes_version(text)
|
||||
previous = _second_changes_version(text)
|
||||
if current is None or previous is None:
|
||||
return [] # the first versioned entry has no predecessor to cross from
|
||||
if current.compat_key == previous.compat_key:
|
||||
return []
|
||||
|
||||
if version_mod.MIGRATION_NONE_MARKER in (version_mod.changes_section(text, current) or ""):
|
||||
return []
|
||||
if any(m.target == current for m in kb_state.load_migrations()):
|
||||
return []
|
||||
|
||||
return [
|
||||
f"{current} crosses the compatibility boundary from {previous}, so every existing "
|
||||
f"instance must migrate - but no document under "
|
||||
f"{rel_path(kb_state.migrations_dir())}/ targets it, and its {version_mod.CHANGES_FILENAME} "
|
||||
f"entry does not carry `{version_mod.MIGRATION_NONE_MARKER}`. Write the migration "
|
||||
"(instructions/migrate-corpus.md), or record why none is needed"
|
||||
]
|
||||
|
||||
|
||||
@app.command("verify")
|
||||
def verify():
|
||||
"""Check the CLI/README command tables, contract presence, type-form drift, ignore rules, and version/changelog agreement."""
|
||||
issues = (
|
||||
check_cli_readme()
|
||||
+ check_readmes_have_no_command_table()
|
||||
+ check_collection_contracts()
|
||||
+ check_legacy_type_blocks()
|
||||
+ check_ignored_content()
|
||||
+ check_version_changelog()
|
||||
+ check_migration_for_boundary()
|
||||
)
|
||||
|
||||
if issues:
|
||||
fail("Documentation issues found:\n" + "\n".join(f"- {i}" for i in issues))
|
||||
|
||||
success(
|
||||
f"Docs verified: {len(registered_commands())} command(s) documented, "
|
||||
f"{len(kb_collections.iter_kb_collections())} collection(s) and "
|
||||
f"{len(STAGE_CONTRACTS)} stage contract(s) present, no legacy type blocks, "
|
||||
f"{len(IGNORE_CANARIES)} ignore canaries clear, "
|
||||
f"{version_mod.CHANGES_FILENAME} documents version "
|
||||
f"{(config.ROOT / version_mod.VERSION_FILENAME).read_text(encoding='utf-8').strip()}."
|
||||
)
|
||||
@@ -0,0 +1,387 @@
|
||||
"""`wikitool doctor` - one deterministic health check for a wiki instance.
|
||||
|
||||
Read-only, never writes. Exists to back `instructions/setup-instance.md` (and
|
||||
any other instance-setup procedure) with a single command instead of ten
|
||||
individual checks spelled out in prose - the same reasoning that keeps
|
||||
mechanical work in code everywhere else in this repo. Each check reports
|
||||
`OK`, `WARN`, or `FAIL` plus, on anything but `OK`, the command to fix it.
|
||||
Only a `FAIL` makes the overall exit code non-zero: a fresh instance with no
|
||||
remote yet, or no `WIKITOOL_SESSION_ID` set, is a valid state, not a fault.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json as _json
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
from rich.console import Console
|
||||
|
||||
from chemenu import config, kb_collections, version as version_mod
|
||||
from chemenu.commands import instructions_cmd
|
||||
from chemenu.commands._util import rel_path
|
||||
from chemenu.session import ENV_VAR as SESSION_ENV_VAR
|
||||
|
||||
console = Console()
|
||||
|
||||
|
||||
@dataclass
|
||||
class Check:
|
||||
name: str
|
||||
status: str # "OK" | "WARN" | "FAIL"
|
||||
detail: str
|
||||
fix: Optional[str] = None
|
||||
|
||||
|
||||
def _git(args: list[str]) -> Optional[subprocess.CompletedProcess]:
|
||||
try:
|
||||
return subprocess.run(
|
||||
["git", *args], cwd=config.ROOT, capture_output=True, text=True, timeout=5
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return None
|
||||
|
||||
|
||||
def check_python() -> Check:
|
||||
version = sys.version_info
|
||||
if version < (3, 11):
|
||||
return Check(
|
||||
"python", "FAIL", f"Python {version.major}.{version.minor} found, need >= 3.11",
|
||||
"Install Python 3.11+ and recreate tools/.venv",
|
||||
)
|
||||
return Check("python", "OK", f"Python {version.major}.{version.minor}.{version.micro}")
|
||||
|
||||
|
||||
def check_ripgrep() -> Check:
|
||||
if shutil.which("rg"):
|
||||
return Check("ripgrep", "OK", "rg found on PATH")
|
||||
return Check(
|
||||
"ripgrep", "FAIL", "rg not found on PATH - `search` and `sources coverage` need it",
|
||||
"Install ripgrep (e.g. `apt install ripgrep` / `brew install ripgrep`)",
|
||||
)
|
||||
|
||||
|
||||
def check_author() -> Check:
|
||||
author = config.default_author()
|
||||
if author is None:
|
||||
return Check(
|
||||
"author", "FAIL", "Neither $WIKI_AUTHOR nor `git config user.name` resolves",
|
||||
"Run `git config user.name \"<Your Name>\"`, or export WIKI_AUTHOR",
|
||||
)
|
||||
import os
|
||||
|
||||
source = "WIKI_AUTHOR" if os.environ.get("WIKI_AUTHOR", "").strip() else "git config user.name"
|
||||
return Check("author", "OK", f"'{author}' (from {source})")
|
||||
|
||||
|
||||
def check_git_repo() -> list[Check]:
|
||||
checks: list[Check] = []
|
||||
inside = _git(["rev-parse", "--is-inside-work-tree"])
|
||||
if inside is None or inside.returncode != 0 or inside.stdout.strip() != "true":
|
||||
checks.append(
|
||||
Check(
|
||||
"git-repo", "FAIL", "Not inside a git working tree",
|
||||
"Run `git init -b main`",
|
||||
)
|
||||
)
|
||||
return checks
|
||||
checks.append(Check("git-repo", "OK", "Inside a git working tree"))
|
||||
|
||||
name = _git(["config", "user.name"])
|
||||
email = _git(["config", "user.email"])
|
||||
if not name or not name.stdout.strip():
|
||||
checks.append(
|
||||
Check("git-identity", "FAIL", "`git config user.name` is not set",
|
||||
"Run `git config user.name \"<Your Name>\"`")
|
||||
)
|
||||
elif not email or not email.stdout.strip():
|
||||
checks.append(
|
||||
Check("git-identity", "FAIL", "`git config user.email` is not set",
|
||||
"Run `git config user.email \"<you@example.com>\"`")
|
||||
)
|
||||
else:
|
||||
checks.append(Check("git-identity", "OK", f"{name.stdout.strip()} <{email.stdout.strip()}>"))
|
||||
|
||||
branch = _git(["rev-parse", "--abbrev-ref", "HEAD"])
|
||||
branch_name = branch.stdout.strip() if branch and branch.returncode == 0 else ""
|
||||
if not branch_name or branch_name == "HEAD":
|
||||
checks.append(
|
||||
Check("git-branch", "WARN", "No commit yet, or detached HEAD",
|
||||
"Make the first commit via `publish` once ready")
|
||||
)
|
||||
else:
|
||||
checks.append(Check("git-branch", "OK", f"On branch '{branch_name}'"))
|
||||
|
||||
remote = _git(["remote", "get-url", "origin"])
|
||||
if remote and remote.returncode == 0 and remote.stdout.strip():
|
||||
checks.append(Check("git-remote", "OK", remote.stdout.strip()))
|
||||
else:
|
||||
checks.append(
|
||||
Check(
|
||||
"git-remote", "WARN", "No 'origin' remote configured",
|
||||
"A local-only instance is valid - `git remote add origin <url>` if you want one. "
|
||||
"Every `publish` needs --no-push until then",
|
||||
)
|
||||
)
|
||||
return checks
|
||||
|
||||
|
||||
def check_skills() -> Check:
|
||||
sources = instructions_cmd.skill_dirs()
|
||||
if not sources:
|
||||
return Check("skills", "FAIL", "No skills found under instructions/", None)
|
||||
target_dirs = instructions_cmd.target_dirs()
|
||||
missing = 0
|
||||
drifted: list[str] = []
|
||||
for target_root in target_dirs:
|
||||
for source in sources:
|
||||
difference = instructions_cmd.drift(source, target_root / source.name)
|
||||
if difference == "missing":
|
||||
missing += 1
|
||||
elif difference:
|
||||
drifted.append(f"{rel_path(target_root / source.name)}: {difference}")
|
||||
expected = len(sources) * len(target_dirs)
|
||||
if missing == expected and not drifted:
|
||||
return Check(
|
||||
"skills", "FAIL", "No skills published yet",
|
||||
"Run `tools/wikitool instructions sync`",
|
||||
)
|
||||
if drifted:
|
||||
return Check(
|
||||
"skills", "FAIL", f"{len(drifted)} published copy/copies drifted from source",
|
||||
"Run `tools/wikitool instructions sync`",
|
||||
)
|
||||
return Check("skills", "OK", f"{expected} published copy/copies match their source")
|
||||
|
||||
|
||||
def check_structure() -> Check:
|
||||
missing = []
|
||||
for relative_path in (
|
||||
"kb/CONTRACT.md", "raw/CONTRACT.md", "reports/CONTRACT.md",
|
||||
"work/CONTRACT.md", "instructions/CONTRACT.md", "types/type-spec.md",
|
||||
):
|
||||
if not (config.ROOT / relative_path).exists():
|
||||
missing.append(relative_path)
|
||||
collections = kb_collections.iter_kb_collections()
|
||||
if not collections:
|
||||
missing.append("kb/*/COLLECTION.md")
|
||||
if missing:
|
||||
return Check(
|
||||
"structure", "FAIL", f"Missing: {', '.join(missing)}",
|
||||
"Re-run `dist export`, or restore the missing contract(s) from the source repo",
|
||||
)
|
||||
return Check(
|
||||
"structure", "OK", f"{len(collections)} collection(s), all stage contracts present"
|
||||
)
|
||||
|
||||
|
||||
def check_personalization() -> Check:
|
||||
"""Whether this instance knows who it works for, and how it sounds.
|
||||
|
||||
`USER.md` and `SOUL.md` are read every session, so an instance without
|
||||
them runs a generic agent against a wiki built for one person - which is
|
||||
a fault, not a preference, hence `FAIL` rather than `WARN`. They are also
|
||||
the one pair of required files a distribution cannot ship filled: their
|
||||
content is personal, so `dist export` carries the templates and the
|
||||
Personalization step of `setup-instance.md` writes the real ones. That
|
||||
makes a still-templated file the second failure mode worth naming
|
||||
separately - it looks present and answers nothing.
|
||||
"""
|
||||
missing: list[str] = []
|
||||
unfilled: list[str] = []
|
||||
for name in config.PERSONALIZATION_FILES:
|
||||
path = config.ROOT / name
|
||||
if not path.is_file():
|
||||
missing.append(name)
|
||||
elif config.TEMPLATE_SENTINEL in path.read_text(encoding="utf-8"):
|
||||
unfilled.append(name)
|
||||
|
||||
fix = (
|
||||
"Run the Personalization step of instructions/setup-instance.md - it interviews you "
|
||||
f"along {' and '.join(config.PERSONALIZATION_TEMPLATES)} and writes your answers verbatim"
|
||||
)
|
||||
if missing:
|
||||
return Check("personalization", "FAIL", f"Missing: {', '.join(missing)}", fix)
|
||||
if unfilled:
|
||||
return Check(
|
||||
"personalization", "FAIL",
|
||||
f"Still the unfilled template: {', '.join(unfilled)}", fix,
|
||||
)
|
||||
return Check("personalization", "OK", f"{', '.join(config.PERSONALIZATION_FILES)} present and filled")
|
||||
|
||||
|
||||
def check_environment() -> Check:
|
||||
"""Whether this checkout records the environment it works through.
|
||||
|
||||
`ENVIRONMENT.md` names the harness, the published skills, the MCP
|
||||
servers, the connectors and the git remotes this working copy actually
|
||||
uses. Missing it costs a session some questions, not correctness, so this
|
||||
check never FAILs - the whole point of the file is that it is optional,
|
||||
and a FAIL would make it mandatory by the back door.
|
||||
|
||||
The one thing worth reporting is the failure mode the personalization
|
||||
check already knows: a template renamed but not filled in. That file is
|
||||
present, is loaded into every session, and answers nothing - worse than
|
||||
absence, because absence is honest.
|
||||
"""
|
||||
path = config.ROOT / config.ENVIRONMENT_FILE
|
||||
if not path.is_file():
|
||||
return Check(
|
||||
"environment", "OK", f"{config.ENVIRONMENT_FILE} absent (optional)",
|
||||
)
|
||||
if config.TEMPLATE_SENTINEL in path.read_text(encoding="utf-8"):
|
||||
return Check(
|
||||
"environment", "WARN",
|
||||
f"{config.ENVIRONMENT_FILE} is still the unfilled template",
|
||||
f"Fill it in along {config.ENVIRONMENT_TEMPLATE}'s sections and drop the "
|
||||
f"`{config.TEMPLATE_SENTINEL}` line, or delete the file - it is optional",
|
||||
)
|
||||
return Check("environment", "OK", f"{config.ENVIRONMENT_FILE} present and filled")
|
||||
|
||||
|
||||
def check_generated_files() -> Check:
|
||||
missing = [
|
||||
rel_path(path)
|
||||
for path in (config.INDEX_FILE, config.LOG_FILE, config.PROVENANCE_FILE)
|
||||
if not path.exists()
|
||||
]
|
||||
if missing:
|
||||
return Check(
|
||||
"generated-files", "FAIL", f"Missing: {', '.join(missing)}",
|
||||
"Run `index rebuild` and `sources rebuild-index`",
|
||||
)
|
||||
return Check("generated-files", "OK", "kb/index.md, kb/log.md, kb/provenance.md present")
|
||||
|
||||
|
||||
def check_session_id() -> Check:
|
||||
import os
|
||||
|
||||
if os.environ.get(SESSION_ENV_VAR, "").strip():
|
||||
return Check("session-id", "OK", f"{SESSION_ENV_VAR}={os.environ[SESSION_ENV_VAR]}")
|
||||
return Check(
|
||||
"session-id", "WARN", f"{SESSION_ENV_VAR} is not set - budget falls back to the parent PID",
|
||||
"See instructions/session-setup.md",
|
||||
)
|
||||
|
||||
|
||||
def check_stack_version() -> Check:
|
||||
"""Which stack this instance runs, and where it came from.
|
||||
|
||||
A missing `VERSION` is a WARN, not a FAIL: instances exported before the
|
||||
stack was versioned are still perfectly functional - they just cannot
|
||||
answer `version check`. A malformed one is a FAIL, because then something
|
||||
edited a generated fact by hand and every comparison built on it is wrong.
|
||||
"""
|
||||
try:
|
||||
current = version_mod.read_version()
|
||||
except version_mod.VersionError as exc:
|
||||
if not version_mod.version_file().is_file():
|
||||
return Check(
|
||||
"stack-version", "WARN", "No VERSION file - this instance predates stack versioning",
|
||||
"Re-export from a current origin, or write the version this instance corresponds to",
|
||||
)
|
||||
return Check("stack-version", "FAIL", str(exc), f"Fix {version_mod.VERSION_FILENAME} by hand - it holds one semantic version, nothing else")
|
||||
|
||||
try:
|
||||
stamp = version_mod.read_stamp()
|
||||
except version_mod.VersionError as exc:
|
||||
return Check(
|
||||
"stack-version", "FAIL", str(exc),
|
||||
f"Delete {version_mod.RELEASE_STAMP_FILENAME} or restore it from the release it came from",
|
||||
)
|
||||
|
||||
origin = "development tree" if stamp is None else f"distribution, exported {stamp.get('exported_at', 'unknown')}"
|
||||
return Check("stack-version", "OK", f"{current} ({origin})")
|
||||
|
||||
|
||||
def check_kb_version() -> Check:
|
||||
"""Whether the content is in the shape this machinery expects.
|
||||
|
||||
A `WARN` when the content lags: that is the normal, transient state in the
|
||||
middle of an upgrade, not a fault - and `migrate status` names the chain
|
||||
that closes it. A missing declaration is also a `WARN` (an instance from
|
||||
before the file existed still works), an unreadable one a `FAIL`.
|
||||
"""
|
||||
from chemenu import kb_state
|
||||
|
||||
try:
|
||||
stack = version_mod.read_version()
|
||||
except version_mod.VersionError:
|
||||
return Check(
|
||||
"kb-version", "WARN", "No stack version to compare the content against",
|
||||
"See the stack-version check above",
|
||||
)
|
||||
try:
|
||||
kb_version = kb_state.read_kb_version()
|
||||
except version_mod.VersionError as exc:
|
||||
return Check(
|
||||
"kb-version", "FAIL", str(exc),
|
||||
f"Restore or delete {kb_state.KB_STATE_FILENAME}, then "
|
||||
"`tools/wikitool migrate baseline <version>`",
|
||||
)
|
||||
|
||||
if kb_version is None:
|
||||
return Check(
|
||||
"kb-version", "WARN",
|
||||
f"{kb_state.KB_STATE_FILENAME} is missing - the content's shape is undeclared",
|
||||
f"Run `tools/wikitool migrate baseline {stack}` if this instance's content has "
|
||||
"never lagged behind its machinery",
|
||||
)
|
||||
if kb_version < stack:
|
||||
pending = kb_state.chain(kb_state.load_migrations(), kb_version, stack)
|
||||
if pending:
|
||||
return Check(
|
||||
"kb-version", "WARN",
|
||||
f"Content is at {kb_version}, machinery at {stack} - "
|
||||
f"{len(pending)} migration(s) outstanding",
|
||||
"Run `tools/wikitool migrate status`",
|
||||
)
|
||||
return Check("kb-version", "OK", f"{kb_version} (nothing outstanding up to {stack})")
|
||||
return Check("kb-version", "OK", f"{kb_version}")
|
||||
|
||||
|
||||
def run_doctor() -> list[Check]:
|
||||
checks: list[Check] = [
|
||||
check_python(),
|
||||
check_ripgrep(),
|
||||
check_author(),
|
||||
check_stack_version(),
|
||||
check_kb_version(),
|
||||
*check_git_repo(),
|
||||
check_skills(),
|
||||
check_structure(),
|
||||
check_personalization(),
|
||||
check_environment(),
|
||||
check_generated_files(),
|
||||
check_session_id(),
|
||||
]
|
||||
return checks
|
||||
|
||||
|
||||
def doctor_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the checks as JSON"),
|
||||
):
|
||||
"""Check that this instance is correctly configured: dependencies, author,
|
||||
git identity/remote, published skills, structure, personalization,
|
||||
generated files, and session scoping. Read-only. Exits 1 only if a check
|
||||
FAILs."""
|
||||
checks = run_doctor()
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps([c.__dict__ for c in checks], indent=2))
|
||||
else:
|
||||
for check in checks:
|
||||
color = {"OK": "green", "WARN": "yellow", "FAIL": "bold red"}[check.status]
|
||||
line = f"[{color}]{check.status}[/{color}] {check.name}: {check.detail}"
|
||||
if check.fix and check.status != "OK":
|
||||
line += f"\n fix: {check.fix}"
|
||||
typer.echo(line) if False else None
|
||||
from rich.console import Console
|
||||
|
||||
Console().print(line)
|
||||
|
||||
if any(check.status == "FAIL" for check in checks):
|
||||
raise typer.Exit(code=1)
|
||||
@@ -0,0 +1,96 @@
|
||||
"""`wikitool eval` - score what a session did against what it left behind.
|
||||
|
||||
Read-only over `kb/`: the command runs lint's checks in-process and reads a
|
||||
trace. It writes only into `reports/evals/`, which is gitignored like the rest of
|
||||
that stage.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, rel_path, success
|
||||
from chemenu.evals import scorecard
|
||||
from chemenu.session import session_id as current_session_id
|
||||
from chemenu.telemetry import reader
|
||||
from chemenu.telemetry.writer import trace_root
|
||||
|
||||
app = typer.Typer(help="Score a traced session (see EVALS.md).")
|
||||
|
||||
EVALS_DIR = config.REPORTS_DIR / "evals"
|
||||
|
||||
|
||||
@app.command("sessions")
|
||||
def sessions_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the list as JSON"),
|
||||
):
|
||||
"""List sessions that have a trace, most recent first."""
|
||||
found = reader.sessions()
|
||||
if json_out:
|
||||
typer.echo(json.dumps(found, indent=2))
|
||||
return
|
||||
if not found:
|
||||
success(f"No traces yet under {rel_path(trace_root())}.")
|
||||
return
|
||||
for name in found:
|
||||
records = reader.read_trace(name)
|
||||
first = records[0]["ts"][:19] if records else "-"
|
||||
typer.echo(f"{name:40} {len(records):5d} event(s) since {first}")
|
||||
|
||||
|
||||
@app.command("score")
|
||||
def score_command(
|
||||
session: Optional[str] = typer.Option(
|
||||
None, "--session",
|
||||
help="Session to score. Defaults to this shell's session, the same id the "
|
||||
"budget gate uses.",
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the scorecard as JSON"),
|
||||
markdown_out: Optional[Path] = typer.Option(
|
||||
None, "--markdown",
|
||||
help="Write a markdown scorecard here, conventionally under reports/evals/.",
|
||||
),
|
||||
save: bool = typer.Option(
|
||||
False, "--save",
|
||||
help="Write the scorecard to reports/evals/<date>/<session>.{json,md}.",
|
||||
),
|
||||
fail_on_error: bool = typer.Option(
|
||||
False, "--fail-on-error",
|
||||
help="Exit non-zero when the tree has hard errors or an invariant was violated.",
|
||||
),
|
||||
):
|
||||
"""Score one session: structural state (L1) plus trajectory rules (L2)."""
|
||||
target = session or current_session_id()
|
||||
records = reader.read_trace(target)
|
||||
if not records:
|
||||
fail(
|
||||
f"No trace for session '{target}'. `wikitool eval sessions` lists the ones "
|
||||
"that exist; a session records nothing when WIKI_TRACE=0."
|
||||
)
|
||||
|
||||
card = scorecard.score(target, records)
|
||||
|
||||
if save:
|
||||
directory = EVALS_DIR / date.today().isoformat()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
stem = target.replace("/", "__")
|
||||
(directory / f"{stem}.json").write_text(json.dumps(card, indent=2), encoding="utf-8")
|
||||
(directory / f"{stem}.md").write_text(
|
||||
scorecard.render_markdown(card) + "\n", encoding="utf-8"
|
||||
)
|
||||
success(f"Wrote {rel_path(directory / stem)}.json/.md")
|
||||
if markdown_out:
|
||||
markdown_out.write_text(scorecard.render_markdown(card) + "\n", encoding="utf-8")
|
||||
success(f"Wrote {rel_path(markdown_out)}")
|
||||
if json_out:
|
||||
typer.echo(json.dumps(card, indent=2))
|
||||
if not json_out and not markdown_out and not save:
|
||||
typer.echo(scorecard.render_markdown(card))
|
||||
|
||||
if fail_on_error and scorecard.failed(card):
|
||||
raise typer.Exit(code=1)
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,325 @@
|
||||
"""Deterministically regenerate the wiki's catalog from every page's frontmatter.
|
||||
|
||||
This replaces manual statistics counting and manual sorted-row insertion, which
|
||||
was a repeated source of errors (miscounts, wrong alphabetical position) when
|
||||
done by hand.
|
||||
|
||||
The catalog is **sharded**, not one file. `kb/index.md` is a map: statistics,
|
||||
one row per collection and per area, and a link to the shard that lists those
|
||||
pages. The tables themselves live in a generated `INDEX.md` inside each
|
||||
collection, and an area that grows past `SHARD_THRESHOLD` rows gets its own.
|
||||
|
||||
Why: a single flat catalog has to be read in full to answer any question about
|
||||
it, so its cost grows with the wiki while the answer being looked for does not.
|
||||
At a few hundred pages that is tens of thousands of tokens spent to learn three
|
||||
filenames. The wiki's own `Index Scaling` page sets the threshold used here.
|
||||
The map stays small enough to browse; `wikitool search` answers everything else.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import rel_path, success
|
||||
from chemenu.kb_collections import iter_kb_collections
|
||||
from chemenu.page import Page
|
||||
from chemenu.kb_scan import GENERATED_INDEX, load_kb_pages
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
app = typer.Typer(help="Manage the generated wiki catalog (kb/index.md + per-collection INDEX.md).")
|
||||
|
||||
TABLE_HEADER = "| Page | Type | Summary | Last Modified |"
|
||||
TABLE_SEP = "|------|------|---------|----------------|"
|
||||
|
||||
SUMMARY_HEADINGS = ("Description", "Definition", "Summary")
|
||||
|
||||
# Rows per area before it is split into its own shard. From the wiki's own
|
||||
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
|
||||
# number so growth is handled by arithmetic rather than by a judgment call.
|
||||
SHARD_THRESHOLD = 50
|
||||
|
||||
# Display title for pages sitting directly in a collection root rather than in
|
||||
# an area subdirectory.
|
||||
UNGROUPED_TITLE = "All"
|
||||
|
||||
DO_NOT_EDIT = "<!-- Generated by `wikitool index rebuild`. Do not hand-edit. -->"
|
||||
|
||||
|
||||
def _summary(page: Page) -> str:
|
||||
fm_summary = page.frontmatter.get("summary")
|
||||
if fm_summary:
|
||||
return str(fm_summary).strip()
|
||||
for heading in SUMMARY_HEADINGS:
|
||||
match = re.search(rf"^## {heading}\s*\n+(.+)", page.body, re.MULTILINE)
|
||||
if match:
|
||||
line = match.group(1).strip().splitlines()[0].strip()
|
||||
if line and not line.upper().startswith("TODO"):
|
||||
return line[:117] + "..." if len(line) > 120 else line
|
||||
return "TODO: add summary"
|
||||
|
||||
|
||||
def _last_modified(page: Page) -> str:
|
||||
for key in ("modified", "date", "created"):
|
||||
value = page.frontmatter.get(key)
|
||||
if value:
|
||||
return str(value)
|
||||
return date.fromtimestamp(page.path.stat().st_mtime).isoformat()
|
||||
|
||||
|
||||
def _type_label(page: Page) -> str:
|
||||
return page.subtype or page.kind or "unknown"
|
||||
|
||||
|
||||
def _table(pages: list[Page]) -> list[str]:
|
||||
lines = [TABLE_HEADER, TABLE_SEP]
|
||||
for page in sorted(pages, key=lambda p: p.title.lower()):
|
||||
lines.append(
|
||||
f"| [[{page.title}]] | {_type_label(page)} | {_summary(page)} | {_last_modified(page)} |"
|
||||
)
|
||||
return lines
|
||||
|
||||
|
||||
def _anchor(title: str) -> str:
|
||||
"""GitHub-style heading anchor, so the map can deep-link into a shard."""
|
||||
slug = re.sub(r"[^a-z0-9\s-]", "", title.lower())
|
||||
return re.sub(r"\s+", "-", slug.strip())
|
||||
|
||||
|
||||
@dataclass
|
||||
class Area:
|
||||
"""One grouping inside a collection: a subdirectory, or the collection root
|
||||
for pages that sit directly in it."""
|
||||
|
||||
name: str
|
||||
title: str
|
||||
pages: list[Page] = field(default_factory=list)
|
||||
own_shard: bool = False
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return len(self.pages)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Collection:
|
||||
name: str
|
||||
areas: list[Area] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return sum(area.count for area in self.areas)
|
||||
|
||||
|
||||
def _area_titles() -> dict[str, str]:
|
||||
"""Display titles for entity areas, taken from the entity type-spec's own
|
||||
`layout:` rather than a hardcoded map - so a new subtype names its own
|
||||
section by adding a type-spec, with no code change."""
|
||||
layout = resolver.get_layout(resolver.find_type_by_name("entity")) or {}
|
||||
return {spec.get("dir", key): spec.get("title", key.title()) for key, spec in layout.items()}
|
||||
|
||||
|
||||
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
|
||||
"""Group pages by their physical location: collection directory, then area
|
||||
subdirectory.
|
||||
|
||||
Location rather than `kind` because a shard lives in the directory it
|
||||
describes, and the two agree by construction: a type-spec's `base_dir:` is
|
||||
what put the page there.
|
||||
"""
|
||||
titles = _area_titles()
|
||||
grouped: dict[str, dict[str, Area]] = {}
|
||||
|
||||
# Seed from the collections that exist on disk, not only from the ones that
|
||||
# happen to hold pages: an empty collection is a real (if unfilled) part of
|
||||
# the wiki, and dropping it from the map would hide it from every reader.
|
||||
for collection_dir in iter_kb_collections(kb_dir):
|
||||
grouped.setdefault(collection_dir.name, {})
|
||||
|
||||
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
|
||||
try:
|
||||
parts = page.path.relative_to(kb_dir).parts
|
||||
except ValueError: # pragma: no cover - pages always live under kb_dir
|
||||
continue
|
||||
if len(parts) < 2:
|
||||
collection_name, area_name = "(kb root)", ""
|
||||
else:
|
||||
collection_name = parts[0]
|
||||
area_name = parts[1] if len(parts) > 2 else ""
|
||||
areas = grouped.setdefault(collection_name, {})
|
||||
area = areas.get(area_name)
|
||||
if area is None:
|
||||
title = titles.get(area_name, area_name.title()) if area_name else UNGROUPED_TITLE
|
||||
area = Area(name=area_name, title=title)
|
||||
areas[area_name] = area
|
||||
area.pages.append(page)
|
||||
|
||||
collections = []
|
||||
for name in sorted(grouped):
|
||||
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
|
||||
for area in ordered:
|
||||
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
|
||||
collections.append(Collection(name=name, areas=ordered))
|
||||
return collections
|
||||
|
||||
|
||||
def build_area_shard(area: Area) -> str:
|
||||
lines = [DO_NOT_EDIT, "", f"# {area.title}", "", f"{area.count} page(s).", ""]
|
||||
lines.extend(_table(area.pages))
|
||||
lines.append("")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def build_collection_shard(collection: Collection) -> str:
|
||||
lines = [DO_NOT_EDIT, "", f"# kb/{collection.name}/ - Index", ""]
|
||||
lines.append(f"{collection.count} page(s). Regenerated by `wikitool index rebuild`.")
|
||||
lines.append("")
|
||||
for area in collection.areas:
|
||||
lines.append(f"## {area.title}")
|
||||
lines.append("")
|
||||
if area.own_shard:
|
||||
lines.append(
|
||||
f"{area.count} page(s) - listed in "
|
||||
f"[{area.name}/{GENERATED_INDEX}]({area.name}/{GENERATED_INDEX})."
|
||||
)
|
||||
else:
|
||||
lines.extend(_table(area.pages))
|
||||
lines.append("")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def build_index_map(collections: list[Collection]) -> str:
|
||||
"""The root catalog: counts and pointers, no page rows.
|
||||
|
||||
Deliberately carries no summaries. A summary is what makes a hit worth
|
||||
opening, and that judgment belongs where the hit is produced - `search` and
|
||||
the shards - not in a file every reader pays for in full.
|
||||
"""
|
||||
totals = {c.name: c.count for c in collections}
|
||||
total = sum(totals.values())
|
||||
|
||||
lines = [
|
||||
DO_NOT_EDIT,
|
||||
"",
|
||||
"# Wiki Index",
|
||||
"",
|
||||
"A map of the wiki, not a catalog of it: counts and pointers only.",
|
||||
"",
|
||||
"To *find* a page, search instead of reading this file:",
|
||||
"",
|
||||
'- `tools/wikitool search "<text>"` - ranked text search, with summaries',
|
||||
"- `tools/wikitool search --field entity_type=system --field 'confidence<0.6'`"
|
||||
" - structured query over frontmatter",
|
||||
"",
|
||||
"The page tables live in a generated `INDEX.md` inside each collection, linked below.",
|
||||
"",
|
||||
"## Statistics",
|
||||
"",
|
||||
f"- **Total Pages:** {total}",
|
||||
]
|
||||
for name in sorted(totals):
|
||||
lines.append(f"- **{name.title()}:** {totals[name]}")
|
||||
lines.append(f"- **Last Updated:** {date.today().isoformat()}")
|
||||
lines.append("")
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
lines.append("## Collections")
|
||||
lines.append("")
|
||||
lines.append("| Collection | Pages | Index |")
|
||||
lines.append("|------------|------:|-------|")
|
||||
for collection in collections:
|
||||
target = f"{collection.name}/{GENERATED_INDEX}"
|
||||
lines.append(f"| `{collection.name}/` | {collection.count} | [{target}]({target}) |")
|
||||
lines.append("")
|
||||
|
||||
for collection in collections:
|
||||
listed = [area for area in collection.areas if area.name]
|
||||
if not listed:
|
||||
continue
|
||||
lines.append(f"### {collection.name}/")
|
||||
lines.append("")
|
||||
lines.append("| Area | Pages | Index |")
|
||||
lines.append("|------|------:|-------|")
|
||||
for area in listed:
|
||||
if area.own_shard:
|
||||
target = f"{collection.name}/{area.name}/{GENERATED_INDEX}"
|
||||
else:
|
||||
target = f"{collection.name}/{GENERATED_INDEX}#{_anchor(area.title)}"
|
||||
lines.append(f"| {area.title} | {area.count} | [{target}]({target}) |")
|
||||
lines.append("")
|
||||
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
lines.append("## Notes")
|
||||
lines.append("")
|
||||
lines.append(
|
||||
"This map and every `INDEX.md` under `kb/` are generated by "
|
||||
"`wikitool index rebuild`. Do not hand-edit them."
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("To add a new page, run `wikitool new ...`, then `wikitool index rebuild`.")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def plan_index(kb_dir: Path) -> dict[Path, str]:
|
||||
"""Every file the catalog consists of, as {path: content}.
|
||||
|
||||
Returning the whole plan instead of writing as it goes is what makes the
|
||||
stale-shard sweep possible: anything named `INDEX.md` that is not in the
|
||||
plan is a leftover from a collection or area that no longer exists.
|
||||
"""
|
||||
collections = group_pages(kb_dir, load_kb_pages(kb_dir))
|
||||
plan: dict[Path, str] = {kb_dir / "index.md": build_index_map(collections)}
|
||||
for collection in collections:
|
||||
collection_dir = kb_dir / collection.name
|
||||
if not collection_dir.is_dir():
|
||||
continue
|
||||
plan[collection_dir / GENERATED_INDEX] = build_collection_shard(collection)
|
||||
for area in collection.areas:
|
||||
if area.own_shard:
|
||||
plan[collection_dir / area.name / GENERATED_INDEX] = build_area_shard(area)
|
||||
return plan
|
||||
|
||||
|
||||
def stale_shards(kb_dir: Path, plan: dict[Path, str]) -> list[Path]:
|
||||
"""Generated shards on disk that the current plan does not produce."""
|
||||
return sorted(p for p in kb_dir.rglob(GENERATED_INDEX) if p not in plan)
|
||||
|
||||
|
||||
def build_index(kb_dir: Path) -> str:
|
||||
"""The root map. Kept as a named function because callers (and tests) ask
|
||||
for "the index" meaning the entry point, not the whole plan."""
|
||||
return build_index_map(group_pages(kb_dir, load_kb_pages(kb_dir)))
|
||||
|
||||
|
||||
@app.command("rebuild")
|
||||
def index_rebuild(
|
||||
dry_run: bool = typer.Option(
|
||||
False, "--dry-run", help="Print what would be written instead of writing it"
|
||||
),
|
||||
):
|
||||
plan = plan_index(config.KB_DIR)
|
||||
stale = stale_shards(config.KB_DIR, plan)
|
||||
|
||||
if dry_run:
|
||||
for path in sorted(plan):
|
||||
typer.echo(f"--- {rel_path(path)}")
|
||||
# nl=False: the content already ends in a newline.
|
||||
typer.echo(plan[path], nl=False)
|
||||
for path in stale:
|
||||
typer.echo(f"--- would remove stale shard: {rel_path(path)}")
|
||||
return
|
||||
|
||||
for path, content in plan.items():
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
for path in stale:
|
||||
path.unlink()
|
||||
|
||||
shards = len(plan) - 1
|
||||
removed = f", removed {len(stale)} stale" if stale else ""
|
||||
success(f"Rebuilt {rel_path(config.INDEX_FILE)} and {shards} shard(s){removed}")
|
||||
@@ -0,0 +1,500 @@
|
||||
"""`wikitool instructions sync|verify|list` - the instruction layer.
|
||||
|
||||
`instructions/` is the single source for everything an agent is told to do. It
|
||||
holds two forms, told apart structurally rather than by any flag:
|
||||
|
||||
- `instructions/<name>.md` - an instruction, reached by link or on request
|
||||
- `instructions/<name>/SKILL.md` - a skill, published into the harness directories
|
||||
|
||||
Publication is by **copy**, into `.agents/skills/` (read natively by GitHub
|
||||
Copilot, Codex CLI and Mistral Vibe) and `.claude/skills/` (Claude Code reads
|
||||
nothing else). This reverses an earlier design that used relative symlinks. The
|
||||
symlink argument was that a link cannot go stale; the counter-arguments that
|
||||
won are that symlinks are unreliable on Windows checkouts and do not survive
|
||||
being archived or copied, and that this repo is meant to stay reproducible
|
||||
elsewhere. The price of a copy is drift, so `verify` checks every copy byte for
|
||||
byte against its source - the copy is derived truth, and derived truth is either
|
||||
checked or absent.
|
||||
|
||||
Both target directories are gitignored. A fresh clone has no skills until `sync`
|
||||
runs; `instructions/bootstrap.md` is the procedure, and `verify` says so rather
|
||||
than reporting an error when *every* copy is missing, because that is the
|
||||
expected state of a clean checkout rather than a fault.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import filecmp
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
import yaml
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands import dist_cmd
|
||||
from chemenu.commands._util import fail, rel_path, success
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
app = typer.Typer(help="Manage instructions/ and publish its skills into the harness directories.")
|
||||
|
||||
SKILL_FILE = "SKILL.md"
|
||||
CONTRACT_FILE = "CONTRACT.md"
|
||||
INSTRUCTION_TYPE = "types/instruction.md"
|
||||
|
||||
# instructions/dev/ is a second, purpose-scoped root nested one level in:
|
||||
# stack-development-only instructions and (nested one level further) the
|
||||
# skill that switches a session into that mode. `dist export` prunes it
|
||||
# wholesale (see dist_cmd.INSTRUCTIONS_EXCLUDE_DIRS) - discovery below treats
|
||||
# it as a second location to scan, not as ordinary recursion.
|
||||
DEV_SUBDIR = "dev"
|
||||
|
||||
# Root files an agent actually loads, and from which a link therefore *reaches*
|
||||
# it. CLAUDE.md sits alongside AGENTS.md rather than being folded into one name:
|
||||
# AGENTS.md is read natively by every other harness (Codex CLI, GitHub Copilot
|
||||
# CLI, Mistral Vibe), while CLAUDE.md is read only by Claude Code, which does not
|
||||
# load AGENTS.md on its own (see CLAUDE.md itself, and AGENTS.md's file-naming
|
||||
# table). A link that belongs in only one harness's auto-loaded file still has to
|
||||
# count - see automatic_load_paths() below for the matching half of this split.
|
||||
AGENT_ROOT_FILES = ("AGENTS.md", "CLAUDE.md")
|
||||
|
||||
# README.md and CHANGES.md are deliberately NOT above. They describe the stack
|
||||
# to humans: AGENTS.md's file-naming table defines README.md as "never by an
|
||||
# agent as instruction", and CHANGES.md is a changelog. An instruction whose only
|
||||
# mention is in one of them deploys to no one, so counting either as a reference
|
||||
# would be a false green by construction - `verify` would go on reporting the
|
||||
# layer healthy while the instruction had become unreachable.
|
||||
#
|
||||
# The dev-boundary check asks a different question - what would *dangle* in a
|
||||
# distributed instance - so it scans README.md too, because `dist export` copies
|
||||
# it verbatim (dist_cmd.ROOT_FILES). CHANGES.md stays out even there; see
|
||||
# dev_only_forbidden_references for why.
|
||||
SHIPPED_DOC_ROOT_FILES = ("README.md",)
|
||||
|
||||
|
||||
def skill_dirs(instructions_dir: Path | None = None) -> list[Path]:
|
||||
"""Directories under instructions/ (including instructions/dev/) that contain a SKILL.md."""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
if not root.is_dir():
|
||||
return []
|
||||
candidates = list(root.iterdir())
|
||||
dev_root = root / DEV_SUBDIR
|
||||
if dev_root.is_dir():
|
||||
candidates += list(dev_root.iterdir())
|
||||
return sorted(p for p in candidates if p.is_dir() and (p / SKILL_FILE).exists())
|
||||
|
||||
|
||||
def instruction_files(instructions_dir: Path | None = None) -> list[Path]:
|
||||
"""Flat instruction files - everything except the layer's own contract.
|
||||
Includes instructions/dev/*.md, the stack-development-only subset
|
||||
`dist export` excludes wholesale."""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
if not root.is_dir():
|
||||
return []
|
||||
files = list(root.glob("*.md"))
|
||||
dev_root = root / DEV_SUBDIR
|
||||
if dev_root.is_dir():
|
||||
files += list(dev_root.glob("*.md"))
|
||||
return sorted(p for p in files if p.name != CONTRACT_FILE)
|
||||
|
||||
|
||||
def is_dev_only(path: Path, instructions_dir: Path | None = None) -> bool:
|
||||
"""Whether `path` sits under instructions/dev/."""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
try:
|
||||
path.relative_to(root / DEV_SUBDIR)
|
||||
except ValueError:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def target_dirs() -> list[Path]:
|
||||
return [config.AGENTS_SKILLS_DIR, config.CLAUDE_SKILLS_DIR]
|
||||
|
||||
|
||||
def _read_frontmatter(path: Path) -> tuple[dict | None, str]:
|
||||
"""Return (frontmatter, error). Exactly one is meaningful."""
|
||||
text = path.read_text(encoding="utf-8")
|
||||
if not text.startswith("---"):
|
||||
return None, "has no YAML frontmatter"
|
||||
end = text.find("---", 3)
|
||||
if end == -1:
|
||||
return None, "frontmatter block is not closed with `---`"
|
||||
try:
|
||||
frontmatter = yaml.safe_load(text[3:end]) or {}
|
||||
except yaml.YAMLError as exc:
|
||||
return None, f"frontmatter is not valid YAML ({exc})"
|
||||
if not isinstance(frontmatter, dict):
|
||||
return None, "frontmatter is not a mapping"
|
||||
return frontmatter, ""
|
||||
|
||||
|
||||
def _copy_skill(source: Path, target: Path, force: bool) -> None:
|
||||
if target.is_symlink():
|
||||
# Left over from the symlink-based mirror this replaced.
|
||||
target.unlink()
|
||||
elif target.exists():
|
||||
if not _looks_like_a_published_skill(target) and not force:
|
||||
fail(
|
||||
f"{rel_path(target)} is not a published skill (no SKILL.md). Syncing would "
|
||||
"delete it and everything in it. Check whether it holds anything worth "
|
||||
"keeping, then re-run with --force."
|
||||
)
|
||||
shutil.rmtree(target)
|
||||
shutil.copytree(source, target)
|
||||
|
||||
|
||||
def _looks_like_a_published_skill(target: Path) -> bool:
|
||||
"""Whether this target directory is a skill slot sync owns.
|
||||
|
||||
Deliberately a *structural* test, not a comparison against the source. An
|
||||
edited copy also differs from its source, and re-running `sync` is the
|
||||
documented fix for exactly that - so refusing on difference would refuse
|
||||
the repair. What sync must not silently delete is a directory that was
|
||||
never a published skill at all.
|
||||
"""
|
||||
return (target / SKILL_FILE).exists()
|
||||
|
||||
|
||||
def drift(source: Path, target: Path) -> str | None:
|
||||
"""Describe how a published copy differs from its source, or None."""
|
||||
if not target.exists():
|
||||
return "missing"
|
||||
if target.is_symlink():
|
||||
return "is a symlink, not a copy"
|
||||
comparison = filecmp.dircmp(str(source), str(target))
|
||||
if comparison.diff_files:
|
||||
return f"differs in {', '.join(sorted(comparison.diff_files))}"
|
||||
if comparison.left_only:
|
||||
return f"missing {', '.join(sorted(comparison.left_only))}"
|
||||
if comparison.right_only:
|
||||
return f"has extra {', '.join(sorted(comparison.right_only))}"
|
||||
return None
|
||||
|
||||
|
||||
def _reference_haystacks(
|
||||
root: Path, *, include_dev: bool, root_files: tuple[str, ...]
|
||||
) -> list[Path]:
|
||||
"""Every file that could mention an instruction: the given `root_files`,
|
||||
every CONTRACT.md, every COLLECTION.md, every skill's SKILL.md, every flat
|
||||
instruction. `include_dev=False` restricts the skill/instruction portion to
|
||||
files outside instructions/dev/ - what `dev_only_forbidden_references`
|
||||
needs, since it specifically asks about mentions from outside that
|
||||
boundary.
|
||||
|
||||
`root_files` is the axis the two callers actually differ on, and it is
|
||||
passed explicitly rather than defaulted because getting it wrong is silent
|
||||
in both directions: too wide, and a human-only document keeps a dead
|
||||
instruction looking alive; too narrow, and a reference that would dangle in
|
||||
a distributed instance goes unreported. See AGENT_ROOT_FILES and
|
||||
SHIPPED_DOC_ROOT_FILES."""
|
||||
haystacks: list[Path] = []
|
||||
for name in root_files:
|
||||
candidate = config.ROOT / name
|
||||
if candidate.exists():
|
||||
haystacks.append(candidate)
|
||||
haystacks += sorted(config.ROOT.rglob(CONTRACT_FILE))
|
||||
haystacks += sorted(config.KB_DIR.rglob("COLLECTION.md")) if config.KB_DIR.is_dir() else []
|
||||
skills = skill_dirs(root)
|
||||
instructions = instruction_files(root)
|
||||
if not include_dev:
|
||||
skills = [d for d in skills if not is_dev_only(d, root)]
|
||||
instructions = [p for p in instructions if not is_dev_only(p, root)]
|
||||
haystacks += [d / SKILL_FILE for d in skills]
|
||||
haystacks += instructions
|
||||
return haystacks
|
||||
|
||||
|
||||
def referenced_names(instructions_dir: Path | None = None) -> set[str]:
|
||||
"""Every instruction filename referenced from somewhere that loads it.
|
||||
|
||||
Scans only what an agent can actually reach: AGENTS.md/CLAUDE.md, the
|
||||
contracts and collection files, the skills, and the other instructions.
|
||||
README.md and CHANGES.md are deliberately not in the haystack - a mention
|
||||
there documents an instruction to a human without deploying it to anyone.
|
||||
"""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
haystacks = _reference_haystacks(root, include_dev=True, root_files=AGENT_ROOT_FILES)
|
||||
|
||||
referenced: set[str] = set()
|
||||
for path in haystacks:
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except OSError: # pragma: no cover - unreadable file
|
||||
continue
|
||||
for candidate in instruction_files(root):
|
||||
if candidate == path:
|
||||
continue # a file referencing itself is not a reference
|
||||
if candidate.name in text:
|
||||
referenced.add(candidate.name)
|
||||
return referenced
|
||||
|
||||
|
||||
def automatic_load_paths(instructions_dir: Path | None = None) -> list[Path]:
|
||||
"""Where an agent encounters a link *without* asking for it by name:
|
||||
AGENTS.md (loaded every session by every other harness) and CLAUDE.md
|
||||
(loaded every session, but only by Claude Code, which does not load
|
||||
AGENTS.md on its own), plus every published skill's SKILL.md (loaded
|
||||
once by the harness, then followed as live procedure).
|
||||
|
||||
Deliberately narrower than `referenced_names()`'s haystack: a mention in
|
||||
a CONTRACT.md, a COLLECTION.md, or another instruction's "see also" is
|
||||
documentation a reader opts into, not something that runs on its own -
|
||||
this is what `manual: true` (see instructions/CONTRACT.md) checks against.
|
||||
"""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
agents_md = config.ROOT / "AGENTS.md"
|
||||
claude_md = config.ROOT / "CLAUDE.md"
|
||||
haystacks: list[Path] = [p for p in (agents_md, claude_md) if p.exists()]
|
||||
haystacks += [d / SKILL_FILE for d in skill_dirs(root)]
|
||||
return haystacks
|
||||
|
||||
|
||||
def manual_forbidden_references(instructions_dir: Path | None = None) -> set[str]:
|
||||
"""Instruction filenames mentioned somewhere they would be picked up
|
||||
automatically - forbidden for a `manual: true` instruction."""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
referenced: set[str] = set()
|
||||
for path in automatic_load_paths(root):
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except OSError: # pragma: no cover - unreadable file
|
||||
continue
|
||||
for candidate in instruction_files(root):
|
||||
if candidate.name in text:
|
||||
referenced.add(candidate.name)
|
||||
return referenced
|
||||
|
||||
|
||||
def dev_only_forbidden_references(instructions_dir: Path | None = None) -> set[str]:
|
||||
"""Instruction/skill names under instructions/dev/ mentioned from outside
|
||||
it - forbidden. `dist export` prunes instructions/dev/ wholesale
|
||||
(dist_cmd.INSTRUCTIONS_EXCLUDE_DIRS), so a reference from outside it
|
||||
would either dangle in a distributed instance or leak a dev-only
|
||||
procedure into a regular content path.
|
||||
|
||||
A mention inside a <!-- dist:strip-start/end --> block is exempt: it is
|
||||
stripped from the haystack text before the scan (`dist_cmd.strip_markers`,
|
||||
the same utility `dist export` itself uses), because `dist export` drops
|
||||
that block and instructions/dev/ together - nothing is left dangling.
|
||||
|
||||
The haystack is wider here than in `referenced_names()`: README.md is
|
||||
scanned too, because this check is about what would *dangle* in a shipped
|
||||
document rather than about what an agent can reach, and `dist export`
|
||||
copies README.md verbatim (dist_cmd.ROOT_FILES). CHANGES.md is the one
|
||||
shipped-looking file left out - `dist export` always replaces it wholesale
|
||||
with dist_templates/CHANGES.md regardless of its content, so a historical
|
||||
mention there never reaches a distributed instance in the first place."""
|
||||
root = instructions_dir or config.INSTRUCTIONS_DIR
|
||||
dev_names = {p.name for p in instruction_files(root) if is_dev_only(p, root)}
|
||||
dev_names |= {d.name for d in skill_dirs(root) if is_dev_only(d, root)}
|
||||
if not dev_names:
|
||||
return set()
|
||||
|
||||
referenced: set[str] = set()
|
||||
haystacks = _reference_haystacks(
|
||||
root, include_dev=False, root_files=AGENT_ROOT_FILES + SHIPPED_DOC_ROOT_FILES
|
||||
)
|
||||
for path in haystacks:
|
||||
try:
|
||||
text = dist_cmd.strip_markers(path.read_text(encoding="utf-8"))
|
||||
except OSError: # pragma: no cover - unreadable file
|
||||
continue
|
||||
for name in dev_names:
|
||||
if name in text:
|
||||
referenced.add(name)
|
||||
return referenced
|
||||
|
||||
|
||||
@app.command("sync")
|
||||
def sync(
|
||||
force: bool = typer.Option(
|
||||
False,
|
||||
"--force",
|
||||
help="Replace a target directory whose contents differ from the source and was not "
|
||||
"generated by sync. Without this, sync refuses instead of deleting content it did "
|
||||
"not create.",
|
||||
),
|
||||
):
|
||||
"""Publish every instructions/<name>/SKILL.md into the harness skill directories."""
|
||||
sources = skill_dirs()
|
||||
if not sources:
|
||||
fail(f"No skills found under {rel_path(config.INSTRUCTIONS_DIR)}.")
|
||||
|
||||
names = {source.name for source in sources}
|
||||
published = []
|
||||
removed = []
|
||||
for target_root in target_dirs():
|
||||
target_root.mkdir(parents=True, exist_ok=True)
|
||||
for source in sources:
|
||||
_copy_skill(source, target_root / source.name, force)
|
||||
# A skill that no longer exists must not keep being offered.
|
||||
for stale in sorted(target_root.iterdir()):
|
||||
if stale.name not in names:
|
||||
removed.append(rel_path(stale))
|
||||
if stale.is_dir() and not stale.is_symlink():
|
||||
shutil.rmtree(stale)
|
||||
else:
|
||||
stale.unlink()
|
||||
published.append(rel_path(target_root))
|
||||
|
||||
suffix = f", removed {len(removed)} stale" if removed else ""
|
||||
success(f"Published {len(sources)} skill(s) to {' and '.join(published)}{suffix}")
|
||||
|
||||
|
||||
@app.command("verify")
|
||||
def verify():
|
||||
"""Check instructions/ against its type, and every published copy against its source."""
|
||||
sources = skill_dirs()
|
||||
instructions = instruction_files()
|
||||
if not sources and not instructions:
|
||||
fail(f"Nothing found under {rel_path(config.INSTRUCTIONS_DIR)}.")
|
||||
|
||||
issues: list[str] = []
|
||||
manual: set[str] = set()
|
||||
|
||||
# 1. Flat instructions validate against the instruction type-spec.
|
||||
for path in instructions:
|
||||
frontmatter, error = _read_frontmatter(path)
|
||||
if frontmatter is None:
|
||||
issues.append(f"{path.name}: {error}")
|
||||
continue
|
||||
if frontmatter.get("type") != INSTRUCTION_TYPE:
|
||||
issues.append(f"{path.name}: `type:` should be {INSTRUCTION_TYPE}")
|
||||
continue
|
||||
try:
|
||||
resolver.validate_frontmatter(frontmatter, INSTRUCTION_TYPE)
|
||||
except ValueError as exc:
|
||||
issues.append(f"{path.name}: {exc}")
|
||||
continue
|
||||
if frontmatter.get("name") != path.stem:
|
||||
issues.append(
|
||||
f"{path.name}: frontmatter `name` ({frontmatter.get('name')!r}) "
|
||||
"does not match the filename"
|
||||
)
|
||||
if frontmatter.get("manual"):
|
||||
manual.add(path.name)
|
||||
|
||||
# 2. Skills carry the frontmatter the harness reads. That frontmatter is
|
||||
# the harness's contract, not this repo's type system's, so it is checked
|
||||
# directly rather than against a schema.
|
||||
for source in sources:
|
||||
frontmatter, error = _read_frontmatter(source / SKILL_FILE)
|
||||
if frontmatter is None:
|
||||
issues.append(f"{source.name}: SKILL.md {error}")
|
||||
continue
|
||||
if frontmatter.get("name") != source.name:
|
||||
issues.append(
|
||||
f"{source.name}: SKILL.md `name` ({frontmatter.get('name')!r}) "
|
||||
"does not match the folder name"
|
||||
)
|
||||
if not frontmatter.get("description"):
|
||||
issues.append(f"{source.name}: SKILL.md is missing (or has an empty) `description`")
|
||||
|
||||
# 3. Published copies match their sources. Missing *everywhere* is a clean
|
||||
# checkout, not a fault - say what to run instead of reporting drift.
|
||||
expected = len(sources) * len(target_dirs())
|
||||
missing = 0
|
||||
drifted: list[str] = []
|
||||
for target_root in target_dirs():
|
||||
for source in sources:
|
||||
difference = drift(source, target_root / source.name)
|
||||
if difference == "missing":
|
||||
missing += 1
|
||||
elif difference:
|
||||
drifted.append(f"{rel_path(target_root / source.name)}: {difference}")
|
||||
if target_root.is_dir():
|
||||
for extra in sorted(target_root.iterdir()):
|
||||
if extra.name not in {s.name for s in sources}:
|
||||
drifted.append(f"{rel_path(extra)}: published but has no source")
|
||||
issues.extend(drifted)
|
||||
bootstrap_needed = missing and missing == expected and not drifted
|
||||
if missing and not bootstrap_needed:
|
||||
issues.append(f"{missing} published copy/copies missing - run `wikitool instructions sync`")
|
||||
|
||||
# 4. An instruction nothing loads is inert. Nothing else would report it -
|
||||
# unless it is `manual: true`, which inverts the rule over a narrower
|
||||
# haystack: that instruction must not be linked from AGENTS.md or a
|
||||
# skill (automatic pickup), though a CONTRACT.md mentioning it by name
|
||||
# as documentation is fine and expected.
|
||||
referenced = referenced_names()
|
||||
forbidden = manual_forbidden_references()
|
||||
for path in instructions:
|
||||
if path.name in manual:
|
||||
if path.name in forbidden:
|
||||
issues.append(
|
||||
f"{path.name}: marked `manual` but linked from AGENTS.md, CLAUDE.md, or a "
|
||||
"skill - that would load it automatically, exactly what `manual` exists to "
|
||||
"prevent. Remove the link, or drop `manual: true` if it should run routinely."
|
||||
)
|
||||
continue
|
||||
if path.name not in referenced:
|
||||
issues.append(
|
||||
f"{path.name}: nothing references it - it deploys to no one. "
|
||||
"Link it from a skill, a contract, AGENTS.md, or CLAUDE.md, or delete it."
|
||||
)
|
||||
|
||||
# 5. instructions/dev/ is a hard boundary: `dist export` prunes it whole,
|
||||
# so nothing outside it may depend on something inside it staying
|
||||
# around in a distributed instance. See dev_only_forbidden_references's
|
||||
# docstring for the dist:strip exemption.
|
||||
for name in sorted(dev_only_forbidden_references()):
|
||||
issues.append(
|
||||
f"{name}: lives under instructions/dev/ but is referenced from outside it and "
|
||||
"outside a dist:strip block - `dist export` removes instructions/dev/ wholesale, so "
|
||||
"that reference would dangle in a distributed instance. Remove the reference, or "
|
||||
"wrap it in a <!-- dist:strip-start/end --> block if it belongs only to this dev "
|
||||
"instance."
|
||||
)
|
||||
|
||||
if issues:
|
||||
fail("Instruction layer issues:\n - " + "\n - ".join(issues))
|
||||
|
||||
if bootstrap_needed:
|
||||
success(
|
||||
f"{len(instructions)} instruction(s) and {len(sources)} skill(s) valid. "
|
||||
"No skills published yet - run `wikitool instructions sync` "
|
||||
"(see instructions/bootstrap.md)."
|
||||
)
|
||||
return
|
||||
|
||||
success(
|
||||
f"{len(instructions)} instruction(s) and {len(sources)} skill(s) valid, "
|
||||
f"{expected} published copy/copies match their source."
|
||||
)
|
||||
|
||||
|
||||
@app.command("list")
|
||||
def list_instructions(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the listing as JSON"),
|
||||
):
|
||||
"""List the flat instructions with their descriptions.
|
||||
|
||||
This is how the layer is discovered. `wikitool search` deliberately covers
|
||||
`kb/` only: a page is found by what it says, an instruction by what it is
|
||||
for, and that is exactly what `description` carries.
|
||||
"""
|
||||
import json as _json
|
||||
|
||||
rows = []
|
||||
for path in instruction_files():
|
||||
frontmatter, _error = _read_frontmatter(path)
|
||||
rows.append(
|
||||
{
|
||||
"name": (frontmatter or {}).get("name", path.stem),
|
||||
"path": rel_path(path),
|
||||
"description": (frontmatter or {}).get("description", ""),
|
||||
}
|
||||
)
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(rows, indent=2))
|
||||
return
|
||||
|
||||
if not rows:
|
||||
typer.echo("No instructions found.")
|
||||
return
|
||||
for row in rows:
|
||||
typer.echo(f"{row['name']} ({row['path']})")
|
||||
typer.echo(f" {row['description']}")
|
||||
typer.echo("")
|
||||
typer.echo(f"{len(rows)} instruction(s). Skills are listed by the agent harness itself.")
|
||||
@@ -0,0 +1,470 @@
|
||||
"""Deterministic structural health checks for the wiki.
|
||||
|
||||
This intentionally covers only what can be computed mechanically: broken
|
||||
wikilinks, orphan pages, index/page drift, frontmatter schema gaps, and
|
||||
filename/title mismatches. Semantic judgment (contradictions, staleness,
|
||||
what's worth writing about next) stays with the LLM - this report gives it a
|
||||
verified factual foundation instead of requiring it to re-derive these facts
|
||||
by reading every page.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import rel_path, success
|
||||
from chemenu.frontmatter_io import frontmatter_error
|
||||
from chemenu.markdown_code import strip_code_spans
|
||||
from chemenu.provenance import broken_raw_refs as find_broken_raw_refs
|
||||
from chemenu.provenance import duplicate_raw_file_owners as find_duplicate_raw_file_owners
|
||||
from chemenu.provenance import extract_inline_cites
|
||||
from chemenu.provenance import legacy_citation_markers as find_legacy_citation_markers
|
||||
from chemenu.provenance import legacy_source_pages as find_legacy_source_pages
|
||||
from chemenu.provenance import orphan_footnote_defs as find_orphan_footnote_defs
|
||||
from chemenu.provenance import uncovered_raw_files as find_uncovered_raw_files
|
||||
from chemenu.provenance import undefined_footnote_refs as find_undefined_footnote_refs
|
||||
from chemenu.kb_scan import (
|
||||
GENERATED_INDEX,
|
||||
WIKILINK_RE,
|
||||
build_link_graph,
|
||||
find_duplicate_title_paths,
|
||||
inbound_links,
|
||||
load_kb_pages,
|
||||
)
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
# Style guide's one mechanically-checkable rule (hard oracle: a plain count).
|
||||
# The rest of the style guide (tone, AI-phrase avoidance) is a soft/proxy judgment
|
||||
# and stays with the LLM - see wiki-manage/wiki-ingest skill guidance, not lint.
|
||||
#
|
||||
# The unit is a quote, not a `>` line. It used to be the line, which measured
|
||||
# the wrap width the rule has no opinion about: one quotation written long
|
||||
# counted 1 and the same quotation wrapped at 100 columns counted 4. An author
|
||||
# who took the finding seriously made the page harder to read to quiet it.
|
||||
QUOTE_LIMIT = 2
|
||||
|
||||
# How many hub pages `most_linked` reports. Purely informational (wiki-status
|
||||
# surfaces it); not a finding, so the cutoff only bounds report size.
|
||||
MOST_LINKED_COUNT = 10
|
||||
|
||||
|
||||
def count_quote_blocks(body: str) -> int:
|
||||
"""How many distinct blockquotes `body` carries.
|
||||
|
||||
A run of consecutive `>` lines is one quote; a blank line or any
|
||||
non-quoted line ends it. Code is masked out first, so a `>` inside a
|
||||
fenced shell transcript is a prompt, not a quotation.
|
||||
|
||||
Lazy continuation - a quote whose wrapped lines drop the `>` - reads here
|
||||
as two quotes rather than one. That over-counts in the direction the limit
|
||||
already errs on, and the corpus prefixes every line, so the alternative
|
||||
(tracking paragraph state) buys nothing.
|
||||
"""
|
||||
count, in_quote = 0, False
|
||||
for line in strip_code_spans(body).splitlines():
|
||||
is_quote = line.lstrip().startswith(">")
|
||||
if is_quote and not in_quote:
|
||||
count += 1
|
||||
in_quote = is_quote
|
||||
return count
|
||||
|
||||
|
||||
def run_lint(kb_dir: Path) -> dict:
|
||||
pages = load_kb_pages(kb_dir)
|
||||
duplicate_titles = find_duplicate_title_paths(kb_dir, config.ROOT)
|
||||
|
||||
# Pages whose frontmatter can't be parsed read back as `{}` everywhere
|
||||
# else, which would let them slip past every frontmatter-driven check
|
||||
# below with no finding at all - so they are detected explicitly.
|
||||
frontmatter_errors = []
|
||||
for title, page in sorted(pages.items()):
|
||||
reason = frontmatter_error(page.path)
|
||||
if reason is None and not page.frontmatter.get("type"):
|
||||
reason = "missing `type:` field"
|
||||
if reason is not None:
|
||||
frontmatter_errors.append({"page": title, "error": reason})
|
||||
|
||||
graph = build_link_graph(pages)
|
||||
broken_links = [
|
||||
{"page": title, "target": target}
|
||||
for title, targets in graph.items()
|
||||
for target in sorted(targets)
|
||||
if target not in pages
|
||||
]
|
||||
|
||||
inbound = inbound_links({t: v for t, v in graph.items() if t != "index"})
|
||||
orphan_pages = sorted(
|
||||
title
|
||||
for title, sources in inbound.items()
|
||||
if not sources
|
||||
and title not in ("index", "log")
|
||||
# comparison pages are not linked to by design; index.md is sufficient coverage
|
||||
and pages[title].kind != "comparison"
|
||||
)
|
||||
|
||||
# Same link graph, opposite end: the most-linked-to pages are the wiki's
|
||||
# hubs. Reported (not judged) so `wiki-status` can show them without
|
||||
# re-deriving the graph.
|
||||
inbound_counts = {title: len(sources) for title, sources in inbound.items()}
|
||||
most_linked = [
|
||||
{"page": title, "inbound": count}
|
||||
for title, count in sorted(inbound_counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
||||
if count > 0
|
||||
][:MOST_LINKED_COUNT]
|
||||
|
||||
# The catalog is sharded: `kb/index.md` is a map carrying counts and links,
|
||||
# and the page rows live in a generated INDEX.md per collection/area. Both
|
||||
# halves have to be read, or every page reads as missing from the index.
|
||||
index_text = "".join(
|
||||
path.read_text(encoding="utf-8")
|
||||
for path in [kb_dir / "index.md", *sorted(kb_dir.rglob(GENERATED_INDEX))]
|
||||
if path.exists()
|
||||
)
|
||||
index_links = {m.group(1).strip() for m in WIKILINK_RE.finditer(index_text)}
|
||||
missing_from_index = sorted(set(pages) - index_links - {"index", "log"})
|
||||
dangling_index_entries = sorted(index_links - set(pages))
|
||||
|
||||
title_mismatches = []
|
||||
for title, page in sorted(pages.items()):
|
||||
if page.kind not in ("entity", "concept"):
|
||||
continue
|
||||
h1 = page.h1_title
|
||||
if h1 is not None and h1 != title:
|
||||
title_mismatches.append({"page": title, "h1": h1})
|
||||
|
||||
unmarked_provenance = []
|
||||
for title, page in sorted(pages.items()):
|
||||
if page.kind not in ("entity", "concept"):
|
||||
continue
|
||||
sources_list = page.frontmatter.get("sources") or []
|
||||
if not sources_list and page.frontmatter.get("provenance") != "general":
|
||||
unmarked_provenance.append(title)
|
||||
|
||||
citation_frontmatter_drift = []
|
||||
for title, page in sorted(pages.items()):
|
||||
sources_list = set(page.frontmatter.get("sources") or [])
|
||||
cited = {cited_title for cited_title, _file in extract_inline_cites(page.body)}
|
||||
cited.discard(title) # a source page citing itself for a specific file within it is not drift
|
||||
for missing_source in sorted(cited - sources_list):
|
||||
citation_frontmatter_drift.append({"page": title, "cited_but_not_in_sources": missing_source})
|
||||
|
||||
legacy_citation_markers = find_legacy_citation_markers(pages)
|
||||
undefined_footnote_refs = find_undefined_footnote_refs(pages)
|
||||
orphan_footnote_defs = find_orphan_footnote_defs(pages)
|
||||
|
||||
# The frontmatter half of the link graph. `broken_links` above only walks
|
||||
# `[[wikilinks]]` in page *bodies*, so a `related:`/`sources:`/`entities:`
|
||||
# entry naming a page that does not exist - a rename that was not
|
||||
# propagated, a deleted page, or a URL pasted where a title belongs - used
|
||||
# to pass every check. Which fields hold page titles is declared by each
|
||||
# type-spec's `page_ref_fields:`, not hardcoded here.
|
||||
dangling_frontmatter_refs = []
|
||||
for title, page in sorted(pages.items()):
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
continue
|
||||
try:
|
||||
ref_fields = resolver.get_page_ref_fields(type_path, page.path)
|
||||
except ValueError:
|
||||
continue # unresolvable type is already reported as type_resolution_errors
|
||||
for field in ref_fields:
|
||||
for target in page.frontmatter.get(field) or []:
|
||||
if target not in pages:
|
||||
dangling_frontmatter_refs.append(
|
||||
{"page": title, "field": field, "target": target}
|
||||
)
|
||||
|
||||
quote_limit_violations = []
|
||||
for title, page in sorted(pages.items()):
|
||||
quote_count = count_quote_blocks(page.body)
|
||||
if quote_count > QUOTE_LIMIT:
|
||||
quote_limit_violations.append({"page": title, "quote_count": quote_count})
|
||||
|
||||
# Type system validation. Lint reports are not validated here: they are
|
||||
# written to `reports/` outside kb/ and are never pages, so nothing this
|
||||
# loop scans can be one.
|
||||
invalid_type_paths = []
|
||||
type_resolution_errors = []
|
||||
schema_validation_errors = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
continue
|
||||
|
||||
# Check if type path is valid
|
||||
if not type_path.endswith('.md'):
|
||||
invalid_type_paths.append({"page": title, "type": type_path, "error": "Type path must end with .md"})
|
||||
continue
|
||||
|
||||
# Try to resolve and validate the type
|
||||
try:
|
||||
resolver.load_type_spec(type_path, page.path)
|
||||
|
||||
# Try schema validation
|
||||
try:
|
||||
resolver.validate_frontmatter(page.frontmatter, type_path, page.path)
|
||||
except ValueError as schema_error:
|
||||
schema_validation_errors.append({"page": title, "type": type_path, "error": str(schema_error)})
|
||||
|
||||
except ValueError as resolution_error:
|
||||
type_resolution_errors.append({"page": title, "type": type_path, "error": str(resolution_error)})
|
||||
|
||||
return {
|
||||
"generated": date.today().isoformat(),
|
||||
"page_count": len(pages),
|
||||
"frontmatter_errors": frontmatter_errors,
|
||||
"broken_links": broken_links,
|
||||
"orphan_pages": orphan_pages,
|
||||
"most_linked": most_linked,
|
||||
"inbound_counts": inbound_counts,
|
||||
"missing_from_index": missing_from_index,
|
||||
"dangling_index_entries": dangling_index_entries,
|
||||
"title_mismatches": title_mismatches,
|
||||
"duplicate_titles": duplicate_titles,
|
||||
"uncovered_raw_files": find_uncovered_raw_files(config.RAW_DIR, pages),
|
||||
"broken_raw_refs": find_broken_raw_refs(pages),
|
||||
"duplicate_raw_file_owners": find_duplicate_raw_file_owners(pages),
|
||||
"legacy_source_pages": find_legacy_source_pages(pages),
|
||||
"unmarked_provenance": unmarked_provenance,
|
||||
"citation_frontmatter_drift": citation_frontmatter_drift,
|
||||
"legacy_citation_markers": legacy_citation_markers,
|
||||
"undefined_footnote_refs": undefined_footnote_refs,
|
||||
"orphan_footnote_defs": orphan_footnote_defs,
|
||||
"dangling_frontmatter_refs": dangling_frontmatter_refs,
|
||||
"quote_limit_violations": quote_limit_violations,
|
||||
"invalid_type_paths": invalid_type_paths,
|
||||
"type_resolution_errors": type_resolution_errors,
|
||||
"schema_validation_errors": schema_validation_errors,
|
||||
}
|
||||
|
||||
|
||||
def _section(lines: list[str], title: str, items: list, formatter) -> None:
|
||||
lines.append(f"## {title}")
|
||||
lines.append("")
|
||||
if not items:
|
||||
lines.append("None found.")
|
||||
else:
|
||||
for item in items:
|
||||
lines.append(f"- {formatter(item)}")
|
||||
lines.append("")
|
||||
|
||||
|
||||
def render_markdown(report: dict) -> str:
|
||||
lines = [f"# Structural Lint Report ({report['generated']})", ""]
|
||||
lines.append(f"Scanned {report['page_count']} pages under `wiki/`. This report covers only")
|
||||
lines.append("mechanically-verifiable structural issues; see the Semantic Review section")
|
||||
lines.append("below for judgment calls the LLM should complete.")
|
||||
lines.append("")
|
||||
|
||||
_section(
|
||||
lines, "Unreadable Frontmatter", report["frontmatter_errors"],
|
||||
lambda i: f"[[{i['page']}]] - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Broken Wikilinks", report["broken_links"],
|
||||
lambda i: f"[[{i['page']}]] links to missing [[{i['target']}]]",
|
||||
)
|
||||
_section(lines, "Orphan Pages (no inbound links)", report["orphan_pages"], lambda i: f"[[{i}]]")
|
||||
_section(
|
||||
lines, f"Most-Linked Pages (top {MOST_LINKED_COUNT} hubs)", report["most_linked"],
|
||||
lambda i: f"[[{i['page']}]] - {i['inbound']} inbound link(s)",
|
||||
)
|
||||
_section(lines, "Pages Missing from index.md", report["missing_from_index"], lambda i: f"[[{i}]]")
|
||||
_section(lines, "Dangling index.md Entries", report["dangling_index_entries"], lambda i: f"[[{i}]]")
|
||||
_section(
|
||||
lines, "Duplicate Titles (naming collisions)", report["duplicate_titles"],
|
||||
lambda i: f"`{i['stem']}` -> {', '.join(f'`{p}`' for p in i['paths'])}",
|
||||
)
|
||||
_section(
|
||||
lines, "Filename / H1 Title Mismatches", report["title_mismatches"],
|
||||
lambda i: f"[[{i['page']}]] H1 is '{i['h1']}'",
|
||||
)
|
||||
_section(
|
||||
lines, "Uncovered Raw Files (no source page)", report["uncovered_raw_files"],
|
||||
lambda i: f"`{i}`",
|
||||
)
|
||||
_section(
|
||||
lines, "Broken raw_files References", report["broken_raw_refs"],
|
||||
lambda i: f"[[{i['page']}]] -> `{i['raw_path']}` (does not exist)",
|
||||
)
|
||||
_section(
|
||||
lines, "Raw Files With More Than One Owner", report["duplicate_raw_file_owners"],
|
||||
lambda i: f"`{i['raw_file']}` is claimed by " + ", ".join(f"[[{t}]]" for t in i["owners"]),
|
||||
)
|
||||
_section(
|
||||
lines, "Legacy source: Field (not yet migrated to raw_files:)", report["legacy_source_pages"],
|
||||
lambda i: f"[[{i['page']}]] source: `{i['source']}` ({i['reason']})",
|
||||
)
|
||||
_section(
|
||||
lines, "Pages Missing provenance: general Marker", report["unmarked_provenance"],
|
||||
lambda i: f"[[{i}]] has no sources and is not marked `provenance: general`",
|
||||
)
|
||||
_section(
|
||||
lines, "Citation / Frontmatter Drift", report["citation_frontmatter_drift"],
|
||||
lambda i: f"[[{i['page']}]] cites [[{i['cited_but_not_in_sources']}]] inline but it is missing from frontmatter `sources:`",
|
||||
)
|
||||
_section(
|
||||
lines, "Legacy Citation Markers (pre-migration `^[[...]]`)", report["legacy_citation_markers"],
|
||||
lambda i: f"[[{i['page']}]] still has `{i['marker']}` - run `wikitool cite add` and replace it with the `[^cite-id]` it prints",
|
||||
)
|
||||
_section(
|
||||
lines, "Undefined Footnote References", report["undefined_footnote_refs"],
|
||||
lambda i: f"[[{i['page']}]] references `[^{i['ref']}]`, which has no `[^{i['ref']}]: [[...]]` definition",
|
||||
)
|
||||
_section(
|
||||
lines, "Orphan Footnote Definitions", report["orphan_footnote_defs"],
|
||||
lambda i: f"[[{i['page']}]] defines `[^{i['id']}]` (-> [[{i['source']}]]) but nothing references it - run `wikitool cite sync`",
|
||||
)
|
||||
_section(
|
||||
lines, "Dangling Frontmatter References", report["dangling_frontmatter_refs"],
|
||||
lambda i: f"[[{i['page']}]] `{i['field']}:` names `{i['target']}`, which is not a page",
|
||||
)
|
||||
_section(
|
||||
lines, "Invalid Type Paths", report["invalid_type_paths"],
|
||||
lambda i: f"[[{i['page']}]] has type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Type Resolution Errors", report["type_resolution_errors"],
|
||||
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Schema Validation Errors", report["schema_validation_errors"],
|
||||
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, f"Pages Exceeding Quote Limit (>{QUOTE_LIMIT}/page)", report["quote_limit_violations"],
|
||||
lambda i: f"[[{i['page']}]] has {i['quote_count']} quotes - trim or confirm they're load-bearing",
|
||||
)
|
||||
|
||||
lines.append("## Semantic Review (LLM to complete)")
|
||||
lines.append("")
|
||||
lines.append("- Contradictions across pages: TODO")
|
||||
lines.append("- Stale claims (unconfirmed >6 months): TODO")
|
||||
lines.append("- Suggested new pages / missing cross-references: TODO")
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# Sections that always carry content but are not findings, so the summary
|
||||
# handles them separately: a hub list is a statistic, and the semantic review
|
||||
# is the checklist that follows the report rather than part of it.
|
||||
INFORMATIONAL_SECTIONS = ("Most-Linked Pages",)
|
||||
SEMANTIC_REVIEW_SECTION = "Semantic Review"
|
||||
|
||||
|
||||
def _split_sections(markdown: str) -> tuple[str, list[tuple[str, str]]]:
|
||||
"""Cut a rendered report into its preamble and (title, body) sections."""
|
||||
preamble, *rest = markdown.split("\n## ")
|
||||
sections = []
|
||||
for part in rest:
|
||||
title, _, body = part.partition("\n")
|
||||
sections.append((title.strip(), body.strip()))
|
||||
return preamble.rstrip(), sections
|
||||
|
||||
|
||||
def render_summary(report: dict) -> str:
|
||||
"""The same report with the empty sections removed.
|
||||
|
||||
On a healthy corpus the full report is better than 90% "None found.", so
|
||||
reading it in the terminal means paging past the answer. The file on disk
|
||||
stays complete - this is what gets printed, and the written path underneath
|
||||
it is how the rest is reached without running lint a second time.
|
||||
"""
|
||||
preamble, sections = _split_sections(render_markdown(report))
|
||||
findings, trailing = [], []
|
||||
for title, body in sections:
|
||||
if title.startswith(SEMANTIC_REVIEW_SECTION):
|
||||
trailing.append((title, body))
|
||||
elif body != "None found." and not title.startswith(INFORMATIONAL_SECTIONS):
|
||||
findings.append((title, body))
|
||||
lines = [preamble, ""]
|
||||
if not findings:
|
||||
lines += ["No structural findings.", ""]
|
||||
for title, body in findings + trailing:
|
||||
lines += [f"## {title}", "", body, ""]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def default_report_path(report: dict) -> Path:
|
||||
"""Where a report goes when the caller names no path.
|
||||
|
||||
`reports/` is derived and gitignored ([reports/CONTRACT.md]), so writing
|
||||
here by default costs the tree nothing.
|
||||
"""
|
||||
return config.ROOT / "reports" / f"Lint Report {report['generated']}.md"
|
||||
|
||||
|
||||
# Findings that make a tree structurally wrong rather than merely untidy.
|
||||
# `orphan_pages` is deliberately absent: many pages are validly reachable
|
||||
# through the index or navigation only. `quote_limit_violations` is advisory
|
||||
# too - it flags a habit, not a broken tree.
|
||||
#
|
||||
# One definition, used by `lint --fail-on-error` and by the eval scorecard: if
|
||||
# the two disagreed, a run could pass its score while lint refused it.
|
||||
HARD_ERROR_KEYS = (
|
||||
"frontmatter_errors",
|
||||
"broken_links",
|
||||
"dangling_index_entries",
|
||||
"duplicate_titles",
|
||||
"broken_raw_refs",
|
||||
"duplicate_raw_file_owners",
|
||||
"legacy_source_pages",
|
||||
"citation_frontmatter_drift",
|
||||
"legacy_citation_markers",
|
||||
"undefined_footnote_refs",
|
||||
"orphan_footnote_defs",
|
||||
"dangling_frontmatter_refs",
|
||||
"invalid_type_paths",
|
||||
"type_resolution_errors",
|
||||
"schema_validation_errors",
|
||||
)
|
||||
|
||||
|
||||
def has_hard_errors(report: dict) -> bool:
|
||||
return any(report.get(key) for key in HARD_ERROR_KEYS)
|
||||
|
||||
|
||||
def lint_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the raw findings as JSON and write no report"),
|
||||
markdown_out: Optional[Path] = typer.Option(None, "--markdown", help="Write the markdown report here instead of the default reports/Lint Report <date>.md"),
|
||||
full: bool = typer.Option(False, "--full", help="Print the whole report instead of only the sections with findings"),
|
||||
fail_on_error: bool = typer.Option(False, "--fail-on-error", help="Exit non-zero if hard errors were found"),
|
||||
):
|
||||
"""Run structural lint checks against kb/.
|
||||
|
||||
Unless `--json` is given, the full report is always written to a file and
|
||||
its path is printed. That path is the point: a lint report is long, and an
|
||||
agent that only saw it on stdout had no way back to the part it scrolled
|
||||
past except by running lint again - two budget slots for one look at the
|
||||
corpus.
|
||||
"""
|
||||
report = run_lint(config.KB_DIR)
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(report, indent=2))
|
||||
if fail_on_error and has_hard_errors(report):
|
||||
raise typer.Exit(code=1)
|
||||
return
|
||||
|
||||
typer.echo(render_markdown(report) if full else render_summary(report))
|
||||
|
||||
target = markdown_out or default_report_path(report)
|
||||
frontmatter = (
|
||||
"---\n"
|
||||
"type: types/lint-report.md\n"
|
||||
f"created: {report['generated']}\n"
|
||||
f"summary: Structural lint report - {report['page_count']} pages scanned\n"
|
||||
"---\n\n"
|
||||
)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_text(frontmatter + render_markdown(report) + "\n", encoding="utf-8")
|
||||
success(f"Full report written to {rel_path(target)}")
|
||||
|
||||
if fail_on_error and has_hard_errors(report):
|
||||
raise typer.Exit(code=1)
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Append correctly-formatted entries to wiki/log.md."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, rel_path, success, today_iso
|
||||
|
||||
app = typer.Typer(help="Manage wiki/log.md.")
|
||||
|
||||
VALID_OPS = ["ingest", "query", "lint", "create", "update", "delete", "rename"]
|
||||
|
||||
# Matches the "## [YYYY-MM-DD] op | title" heading `format_log_entry` writes,
|
||||
# in file order (oldest first, since entries are appended).
|
||||
LOG_ENTRY_RE = re.compile(r"^## \[(\d{4}-\d{2}-\d{2})\] (\S+) \| (.+)$", re.MULTILINE)
|
||||
|
||||
|
||||
def format_log_entry(op: str, title: str, body: str = "", today: Optional[str] = None) -> str:
|
||||
today = today or today_iso()
|
||||
entry = f"## [{today}] {op} | {title}\n"
|
||||
if body.strip():
|
||||
entry += f"\n{body.strip()}\n"
|
||||
entry += "\n---\n"
|
||||
return entry
|
||||
|
||||
|
||||
def parse_log_entries(text: str) -> list[tuple[str, str, str]]:
|
||||
"""Return every logged (date, op, title) entry in file order."""
|
||||
return [(m.group(1), m.group(2), m.group(3)) for m in LOG_ENTRY_RE.finditer(text)]
|
||||
|
||||
|
||||
def ingests_since_last_lint(entries: list[tuple[str, str, str]]) -> int:
|
||||
"""Count `ingest` entries logged after the most recent `lint` entry (or
|
||||
since the start of the log, if it has never been linted).
|
||||
|
||||
This is the deterministic count behind the Maintenance Schedule's "every
|
||||
10 sources" full-lint cadence: nothing else in the system tracks it, so
|
||||
without this the claim was prose with no enforcement - an agent (or user)
|
||||
had to remember to count."""
|
||||
count = 0
|
||||
for _date, op, _title in entries:
|
||||
if op == "lint":
|
||||
count = 0
|
||||
elif op == "ingest":
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
@app.command("append")
|
||||
def log_append(
|
||||
op: str = typer.Option(..., "--op", help="|".join(VALID_OPS)),
|
||||
title: str = typer.Option(..., "--title", help="Brief description, e.g. a source path"),
|
||||
body: str = typer.Option("", "--body", help="Optional multi-line details"),
|
||||
body_file: Optional[Path] = typer.Option(None, "--body-file", help="Read the body from a file instead of --body"),
|
||||
):
|
||||
if op not in VALID_OPS:
|
||||
fail(f"--op must be one of {VALID_OPS}")
|
||||
text = body
|
||||
if body_file:
|
||||
text = body_file.read_text(encoding="utf-8")
|
||||
entry = format_log_entry(op, title, text)
|
||||
with config.LOG_FILE.open("a", encoding="utf-8") as f:
|
||||
f.write("\n" + entry)
|
||||
success(f"Appended log entry to {rel_path(config.LOG_FILE)}")
|
||||
|
||||
|
||||
@app.command("status")
|
||||
def log_status():
|
||||
"""Report how many `ingest` operations have been logged since the last
|
||||
`lint` - the deterministic trigger for the Maintenance Schedule's "every
|
||||
10 sources" full-lint cadence. Read-only."""
|
||||
if not config.LOG_FILE.exists():
|
||||
success("No wiki/log.md yet; nothing logged.")
|
||||
return
|
||||
entries = parse_log_entries(config.LOG_FILE.read_text(encoding="utf-8"))
|
||||
count = ingests_since_last_lint(entries)
|
||||
typer.echo(f"Ingests since last lint: {count}")
|
||||
if count >= 10:
|
||||
typer.echo("Threshold reached (>=10) - run the wiki-lint skill (`tools/wikitool lint ...`) next.")
|
||||
success(f"{len(entries)} total log entries in {rel_path(config.LOG_FILE)}.")
|
||||
@@ -0,0 +1,381 @@
|
||||
"""`wikitool migrate` - content migrations: what this instance still owes, and
|
||||
whether a bulk rewrite broke anything.
|
||||
|
||||
Five commands around two facts. `.wikitool-kb.json` (see `chemenu/kb_state.py`)
|
||||
records what shape the content is in, so the chain of outstanding migrations is
|
||||
computed rather than guessed. `migrate verify` compares the corpus against a git
|
||||
revision on the invariants a migration must not change (see
|
||||
`chemenu/corpus_diff.py`).
|
||||
|
||||
**There is no `migrate run`.** An `assisted` migration is a procedure an agent
|
||||
carries out page by page; the tool keeps the books and checks the result. A
|
||||
`run` would claim an ability that does not exist - it arrives when mechanical
|
||||
primitives do.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json as _json
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, corpus_diff, kb_scan, kb_state, version as version_mod
|
||||
from chemenu.commands._util import console, fail, rel_path, success, today_iso
|
||||
from chemenu.frontmatter_io import read_page
|
||||
from chemenu.page import Page
|
||||
from chemenu.version import Version, VersionError
|
||||
|
||||
app = typer.Typer(help="Content migrations: outstanding chain, bookkeeping, and verification.")
|
||||
|
||||
|
||||
# --- shared state loading --------------------------------------------------
|
||||
|
||||
|
||||
def _versions() -> tuple[Version, Optional[Version]]:
|
||||
"""(stack version, kb version).
|
||||
|
||||
An unreadable VERSION or a corrupt state file exits 1 here (via `fail`,
|
||||
which raises); a *missing* kb version returns None, because that is a
|
||||
state each command explains in its own words rather than an error."""
|
||||
try:
|
||||
return version_mod.read_version(), kb_state.read_kb_version()
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
raise # unreachable: fail() raises typer.Exit
|
||||
|
||||
|
||||
# --- migrate list ----------------------------------------------------------
|
||||
|
||||
|
||||
@app.command("list")
|
||||
def list_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the migrations as JSON"),
|
||||
):
|
||||
"""List every migration document, oldest target first. Read-only."""
|
||||
migrations = kb_state.load_migrations()
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
[
|
||||
{
|
||||
"name": m.name,
|
||||
"migrates_to": str(m.target),
|
||||
"migration_kind": m.kind,
|
||||
"description": m.description,
|
||||
"path": m.relative_path,
|
||||
}
|
||||
for m in migrations
|
||||
],
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
if not migrations:
|
||||
success(f"No migration documents under {rel_path(kb_state.migrations_dir())}.")
|
||||
return
|
||||
for migration in migrations:
|
||||
console.print(f"[bold]{migration.target}[/bold] {migration.name} ({migration.kind})")
|
||||
if migration.description:
|
||||
console.print(f" {migration.description}")
|
||||
|
||||
|
||||
# --- migrate status --------------------------------------------------------
|
||||
|
||||
|
||||
@app.command("status")
|
||||
def status_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the chain as JSON"),
|
||||
):
|
||||
"""Show the migrations this instance still owes, in the order they run.
|
||||
|
||||
Read-only. Exit 1 only when the KB version is undeclared - that is a
|
||||
question the tool refuses to answer by guessing."""
|
||||
stack, kb_version = _versions()
|
||||
migrations = kb_state.load_migrations()
|
||||
|
||||
if kb_version is None:
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
{"stack_version": str(stack), "kb_version": None, "pending": None}, indent=2
|
||||
)
|
||||
)
|
||||
fail(
|
||||
f"{kb_state.KB_STATE_FILENAME} is missing - this instance has never declared what "
|
||||
f"shape its content is in, and guessing would be wrong exactly when it matters.\n"
|
||||
f"Declare it once: `wikitool migrate baseline <version>` (use {stack} if this "
|
||||
f"instance's content has never been migrated behind its machinery)."
|
||||
)
|
||||
return
|
||||
|
||||
pending = kb_state.chain(migrations, kb_version, stack)
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
{
|
||||
"stack_version": str(stack),
|
||||
"kb_version": str(kb_version),
|
||||
"pending": [
|
||||
{"name": m.name, "migrates_to": str(m.target), "migration_kind": m.kind}
|
||||
for m in pending
|
||||
],
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
console.print(f"stack {stack}, content {kb_version}")
|
||||
if not pending:
|
||||
if kb_version < stack:
|
||||
console.print(
|
||||
f"[green]Nothing outstanding[/green] - no migration targets the range "
|
||||
f"({kb_version}, {stack}]."
|
||||
)
|
||||
else:
|
||||
console.print("[green]Nothing outstanding[/green] - content matches the machinery.")
|
||||
return
|
||||
|
||||
console.print(f"[cyan]{len(pending)} migration(s) outstanding, in this order:[/cyan]")
|
||||
for position, migration in enumerate(pending, start=1):
|
||||
console.print(f" {position}. {migration.target} {migration.name} ({migration.kind})")
|
||||
if migration.description:
|
||||
console.print(f" {migration.description}")
|
||||
console.print(f" {migration.relative_path}")
|
||||
console.print(
|
||||
f"\nRun the first one, then record it: `wikitool migrate done {pending[0].target}`.\n"
|
||||
"The procedure is instructions/migrate-corpus.md."
|
||||
)
|
||||
|
||||
|
||||
# --- migrate done / baseline ----------------------------------------------
|
||||
|
||||
|
||||
@app.command("done")
|
||||
def done_command(
|
||||
version: str = typer.Argument(..., help="The migration's target version, e.g. 1.4.0"),
|
||||
pages: Optional[int] = typer.Option(None, "--pages", help="How many pages it touched"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Report without writing"),
|
||||
):
|
||||
"""Record one migration as applied, advancing the KB version to its target.
|
||||
|
||||
Refuses any version that is not the *next* link in the chain: skipping a
|
||||
migration is how a corpus ends up in a shape no version describes, and an
|
||||
interrupted multi-step upgrade has to be resumable rather than guessable."""
|
||||
stack, kb_version = _versions()
|
||||
if kb_version is None:
|
||||
fail(
|
||||
f"{kb_state.KB_STATE_FILENAME} is missing - run `wikitool migrate baseline <version>` "
|
||||
"before recording a migration."
|
||||
)
|
||||
return
|
||||
|
||||
try:
|
||||
target = Version.parse(version)
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
migrations = kb_state.load_migrations()
|
||||
expected = kb_state.next_link(migrations, kb_version, stack)
|
||||
if expected is None:
|
||||
fail(
|
||||
f"Nothing is outstanding: content is at {kb_version}, machinery at {stack}, and no "
|
||||
f"migration targets the range in between."
|
||||
)
|
||||
return
|
||||
if expected.target != target:
|
||||
fail(
|
||||
f"{target} is not the next migration. The chain from {kb_version} continues with "
|
||||
f"{expected.target} ({expected.name}) - applying them out of order leaves the corpus "
|
||||
f"in a shape no version describes.\nRun `wikitool migrate status` to see the order."
|
||||
)
|
||||
return
|
||||
|
||||
state = kb_state.read_kb_state() or {}
|
||||
applied = list(state.get("applied") or [])
|
||||
entry = {"migration": expected.name, "at": today_iso()}
|
||||
if pages is not None:
|
||||
entry["pages"] = pages
|
||||
applied.append(entry)
|
||||
|
||||
if dry_run:
|
||||
success(f"Dry run: content {kb_version} -> {target} ({expected.name}). Nothing written.")
|
||||
return
|
||||
|
||||
kb_state.write_kb_state(target, applied)
|
||||
remaining = kb_state.chain(migrations, target, stack)
|
||||
success(
|
||||
f"Content is now {target} ({expected.name}). "
|
||||
+ (
|
||||
f"{len(remaining)} migration(s) still outstanding - next is {remaining[0].target}."
|
||||
if remaining
|
||||
else "Nothing outstanding."
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@app.command("baseline")
|
||||
def baseline_command(
|
||||
version: str = typer.Argument(..., help="The shape this instance's content is already in"),
|
||||
force: bool = typer.Option(
|
||||
False, "--force", help="Overwrite an existing declaration (not a substitute for `done`)"
|
||||
),
|
||||
):
|
||||
"""Declare the KB version once, for an instance that never had one.
|
||||
|
||||
Only for a tree predating `.wikitool-kb.json`. Advancing the version after
|
||||
a migration is `migrate done`, which checks the chain; this command does
|
||||
not, which is why it refuses to overwrite silently."""
|
||||
try:
|
||||
target = Version.parse(version)
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
try:
|
||||
existing = kb_state.read_kb_version()
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
if existing is not None and not force:
|
||||
fail(
|
||||
f"This instance already declares content version {existing}. Use "
|
||||
f"`wikitool migrate done <version>` to advance it after a migration, or --force "
|
||||
f"if the declaration itself is wrong."
|
||||
)
|
||||
return
|
||||
|
||||
state = kb_state.read_kb_state() or {}
|
||||
kb_state.write_kb_state(target, list(state.get("applied") or []))
|
||||
success(f"Content version declared as {target}.")
|
||||
|
||||
|
||||
# --- migrate verify --------------------------------------------------------
|
||||
|
||||
|
||||
def _git_show(rev: str, relative: str) -> Optional[str]:
|
||||
result = subprocess.run(
|
||||
["git", "show", f"{rev}:{relative}"],
|
||||
cwd=config.ROOT,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
return result.stdout if result.returncode == 0 else None
|
||||
|
||||
|
||||
def _paths_at(rev: str) -> Optional[list[str]]:
|
||||
result = subprocess.run(
|
||||
["git", "ls-tree", "-r", "--name-only", "-z", rev, "--", "kb"],
|
||||
cwd=config.ROOT,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
# The same page/not-a-page rule the working tree is read with. Answering it
|
||||
# differently on the two sides reported every COLLECTION.md and INDEX.md as
|
||||
# a page that had since disappeared.
|
||||
return [
|
||||
path
|
||||
for path in result.stdout.split("\0")
|
||||
if path.startswith("kb/") and kb_scan.is_page_path(path[len("kb/"):])
|
||||
]
|
||||
|
||||
|
||||
def _shapes_at_revision(rev: str, wanted: set[str]) -> dict[str, corpus_diff.PageShape]:
|
||||
"""Page shapes as of `rev`, keyed by repo-relative path."""
|
||||
import tempfile
|
||||
|
||||
shapes: dict[str, corpus_diff.PageShape] = {}
|
||||
paths = _paths_at(rev)
|
||||
if paths is None:
|
||||
fail(f"`git show {rev}` failed - is {rev} a revision in this repository?")
|
||||
return shapes
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
for relative in paths:
|
||||
if wanted and not any(relative.startswith(prefix) for prefix in wanted):
|
||||
continue
|
||||
text = _git_show(rev, relative)
|
||||
if text is None:
|
||||
continue
|
||||
# read_page owns frontmatter parsing (and its error contract), so the
|
||||
# historical blob is materialised under its real filename - the stem
|
||||
# is the page title, which PageShape compares.
|
||||
scratch = Path(tmp) / Path(relative).name
|
||||
scratch.write_text(text, encoding="utf-8")
|
||||
try:
|
||||
frontmatter, body = read_page(scratch)
|
||||
except Exception: # noqa: BLE001 - an unparseable historical page is not this tool's error
|
||||
continue
|
||||
shapes[relative] = corpus_diff.PageShape.of(Page(Path(relative), frontmatter, body))
|
||||
return shapes
|
||||
|
||||
|
||||
def _shapes_now(wanted: set[str]) -> dict[str, corpus_diff.PageShape]:
|
||||
shapes: dict[str, corpus_diff.PageShape] = {}
|
||||
for path in kb_scan.iter_kb_pages(config.KB_DIR):
|
||||
relative = path.relative_to(config.ROOT).as_posix()
|
||||
if wanted and not any(relative.startswith(prefix) for prefix in wanted):
|
||||
continue
|
||||
try:
|
||||
frontmatter, body = read_page(path)
|
||||
except Exception: # noqa: BLE001 - lint reports unreadable frontmatter
|
||||
continue
|
||||
shapes[relative] = corpus_diff.PageShape.of(Page(path, frontmatter, body))
|
||||
return shapes
|
||||
|
||||
|
||||
@app.command("verify")
|
||||
def verify_command(
|
||||
from_rev: str = typer.Option(..., "--from", help="Git revision to compare against, e.g. HEAD"),
|
||||
path: Optional[list[str]] = typer.Option(
|
||||
None, "--path", help="Limit to a subtree, repeatable (e.g. kb/concepts)"
|
||||
),
|
||||
expect_body_change: bool = typer.Option(
|
||||
False, "--expect-body-change", help="Also report pages whose body did not change at all"
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the diff as JSON"),
|
||||
fail_on_error: bool = typer.Option(
|
||||
False, "--fail-on-error", help="Exit 1 if any invariant changed"
|
||||
),
|
||||
):
|
||||
"""Compare kb/ against a git revision on the invariants a content migration
|
||||
must not change: wikilink and citation *counts*, footnote definitions, H1,
|
||||
and structural frontmatter.
|
||||
|
||||
Not migration-specific - worth running after any bulk rewrite. `lint` cannot
|
||||
answer this: it reads one revision, so a reference that went missing is
|
||||
invisible to it."""
|
||||
wanted = {p.rstrip("/") for p in (path or [])}
|
||||
before = _shapes_at_revision(from_rev, wanted)
|
||||
after = _shapes_now(wanted)
|
||||
diff = corpus_diff.compare(before, after, expect_body_change=expect_body_change)
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
{
|
||||
"from": from_rev,
|
||||
"compared": diff.compared,
|
||||
"added": diff.added,
|
||||
"removed": diff.removed,
|
||||
"findings": [
|
||||
{"path": f.path, "kind": f.kind, "detail": f.detail} for f in diff.findings
|
||||
],
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
else:
|
||||
typer.echo(corpus_diff.render_report(diff, from_rev))
|
||||
|
||||
if diff.findings and fail_on_error:
|
||||
raise typer.Exit(code=1)
|
||||
@@ -0,0 +1,331 @@
|
||||
"""Scaffold new wiki pages from type-spec templates.
|
||||
|
||||
These commands produce structurally-correct frontmatter and a body
|
||||
skeleton with TODO placeholders by loading templates from type-spec files.
|
||||
The prose (Description, Summary, Key Takeaways, ...) is still written by the
|
||||
LLM afterwards with its normal edit tool.
|
||||
|
||||
The split between type-spec templates and LLM-provided prose is intentional:
|
||||
type definitions (naming, frontmatter shape, directory placement, templates) are
|
||||
deterministic and stored in /types/; the content is judgment and provided by the LLM.
|
||||
|
||||
Frontmatter defaults, enum validity, and required-ness all come from the
|
||||
type's `.schema.yaml` (via `TypeResolver`) - nothing here re-declares them.
|
||||
Directory placement for subtype-driven types (currently just entities) also
|
||||
comes from the type-spec, via its `layout:` frontmatter (see
|
||||
`TypeResolver.get_layout`) - not a hand-maintained Python dict.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
import datetime
|
||||
from typing import Any, Dict, Optional
|
||||
import re
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import (
|
||||
check_collision,
|
||||
check_raw_files_exist,
|
||||
fail,
|
||||
parse_set_fields,
|
||||
rel_path,
|
||||
success,
|
||||
)
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
|
||||
def _default_summary(summary: str) -> str:
|
||||
"""Scaffold-time placeholder for an unfilled --summary, so schema
|
||||
validation's `summary` minLength requirement doesn't block page creation.
|
||||
Matches the same "TODO: add summary" text index_build.py already falls
|
||||
back to when a page has no frontmatter summary."""
|
||||
return summary.strip() or "TODO: add summary"
|
||||
|
||||
|
||||
def _resolve_type_and_get_template(type_path: str, source_dir: Path = None):
|
||||
"""Resolve a type path, load the type-spec, and extract its template."""
|
||||
type_spec = resolver.load_type_spec(type_path, source_dir)
|
||||
template = resolver.extract_template(type_spec)
|
||||
return type_spec, template
|
||||
|
||||
|
||||
def _enum_help(type_path: str, field_name: str) -> str:
|
||||
"""Build a help string from the schema's own enum, so any text listing a
|
||||
field's valid values can never drift from what the schema accepts."""
|
||||
return f"One of: {'|'.join(resolver.get_enum(type_path, field_name))}"
|
||||
|
||||
|
||||
def _parse_date(field_name: str, text: str) -> datetime.date:
|
||||
"""A `--set <date field>=<value>` as a real date, or a refusal naming it."""
|
||||
try:
|
||||
return datetime.date.fromisoformat(text)
|
||||
except ValueError:
|
||||
fail(f"--set {field_name}={text!r} must be YYYY-MM-DD.")
|
||||
|
||||
|
||||
def _build_frontmatter(
|
||||
type_path: str, schema: Optional[Dict[str, Any]], today: datetime.date, explicit: Dict[str, Any]
|
||||
) -> Dict[str, Any]:
|
||||
"""Build a page's frontmatter dict in schema-declaration order.
|
||||
|
||||
`explicit` supplies every CLI-derived value the caller already has;
|
||||
fields not in `explicit` get a type-appropriate default (today's date for
|
||||
date-formatted fields, the scaffold placeholder for `summary`, the
|
||||
schema's own `default:` where declared, an empty list for arrays), or are
|
||||
omitted entirely if optional with no sensible default (e.g.
|
||||
`source_url`). This is what lets frontmatter shape - and scaffold-time
|
||||
defaults like `provenance: general` or `confidence: 0.5` - follow the
|
||||
schema instead of being hand-declared per CLI command.
|
||||
"""
|
||||
frontmatter: Dict[str, Any] = {"type": type_path}
|
||||
for field_name, field_schema in (schema or {}).get("properties", {}).items():
|
||||
if field_name == "type":
|
||||
continue
|
||||
if field_name in explicit:
|
||||
value = explicit[field_name]
|
||||
# A `--set date=2026-08-23` arrives as a string; the corpus stores
|
||||
# dates as `datetime.date`, and the schema already says which fields
|
||||
# those are. Converting here keeps one representation on disk instead
|
||||
# of leaving it to whoever wrote the CLI call.
|
||||
if field_schema.get("format") == "date" and isinstance(value, str):
|
||||
value = _parse_date(field_name, value)
|
||||
frontmatter[field_name] = value
|
||||
elif field_name == "summary":
|
||||
frontmatter[field_name] = _default_summary("")
|
||||
elif field_name == "author":
|
||||
resolved_author = config.default_author()
|
||||
if resolved_author is None:
|
||||
fail(
|
||||
"No author configured for this instance. Set `git config user.name`, "
|
||||
"or export WIKI_AUTHOR to override it, then retry."
|
||||
)
|
||||
frontmatter[field_name] = resolved_author
|
||||
elif field_schema.get("format") == "date":
|
||||
frontmatter[field_name] = today
|
||||
elif "default" in field_schema:
|
||||
frontmatter[field_name] = field_schema["default"]
|
||||
elif field_schema.get("type") == "array":
|
||||
frontmatter[field_name] = []
|
||||
|
||||
# Carry through any caller-supplied field the schema doesn't declare,
|
||||
# rather than silently dropping it: a typo'd `--set` must surface as a
|
||||
# validation error (via `additionalProperties: false`) instead of being
|
||||
# quietly ignored, and a schema that does allow extra properties should
|
||||
# keep them.
|
||||
for field_name, value in explicit.items():
|
||||
if field_name not in frontmatter:
|
||||
frontmatter[field_name] = value
|
||||
return frontmatter
|
||||
|
||||
|
||||
def _filter_bullets(value: Any) -> str:
|
||||
"""Render a list frontmatter field as `- [[item]]` bullet lines."""
|
||||
items = value or []
|
||||
return "\n".join(f"- [[{item}]]" for item in items) if items else "- None identified"
|
||||
|
||||
|
||||
def _filter_join(value: Any) -> str:
|
||||
"""Comma-join a list frontmatter field."""
|
||||
return ", ".join(value or [])
|
||||
|
||||
|
||||
def _filter_capitalize(value: Any) -> str:
|
||||
return str(value).capitalize()
|
||||
|
||||
|
||||
def _filter_table_header(value: Any) -> str:
|
||||
"""Render an array field as wikilinked markdown table column headers."""
|
||||
return " | ".join(f"[[{item}]]" for item in (value or []))
|
||||
|
||||
|
||||
def _filter_table_sep(value: Any) -> str:
|
||||
"""Render the markdown table separator row for an array field, one
|
||||
column per item."""
|
||||
return "|".join("--------" for _ in (value or [])) or "--------"
|
||||
|
||||
|
||||
def _filter_table_cells(value: Any) -> str:
|
||||
"""Render a placeholder table body row, one cell per array item."""
|
||||
return " | ".join("..." for _ in (value or []))
|
||||
|
||||
|
||||
_TEMPLATE_FILTERS = {
|
||||
"bullets": _filter_bullets,
|
||||
"join": _filter_join,
|
||||
"capitalize": _filter_capitalize,
|
||||
"table_header": _filter_table_header,
|
||||
"table_sep": _filter_table_sep,
|
||||
"table_cells": _filter_table_cells,
|
||||
}
|
||||
|
||||
|
||||
def _apply_template_variables(template: str, variables: Dict[str, Any]) -> str:
|
||||
"""Apply variable substitutions to a template string.
|
||||
|
||||
Supports:
|
||||
- `{field}` - plain substitution from `variables[field]`
|
||||
- `{field|filter}` - apply a named filter (bullets, join, capitalize)
|
||||
to `variables[field]`'s value, so templates can render list/enum
|
||||
frontmatter fields directly instead of the caller precomputing a
|
||||
separate display-only variable for each one
|
||||
- `{field|literal text}` - literal fallback if `field` isn't in
|
||||
`variables` at all and the suffix isn't a recognized filter name
|
||||
"""
|
||||
def replace_match(match: re.Match) -> str:
|
||||
full_match = match.group(0)
|
||||
var_name = match.group(1)
|
||||
if '|' in var_name:
|
||||
var_name, suffix = var_name.split('|', 1)
|
||||
var_name = var_name.strip()
|
||||
suffix = suffix.strip()
|
||||
if var_name not in variables:
|
||||
return suffix
|
||||
value = variables[var_name]
|
||||
filter_fn = _TEMPLATE_FILTERS.get(suffix)
|
||||
return filter_fn(value) if filter_fn else str(value)
|
||||
return str(variables.get(var_name, full_match))
|
||||
|
||||
pattern = r'\{([^}]+)\}'
|
||||
return re.sub(pattern, replace_match, template)
|
||||
|
||||
|
||||
def _page_subdir(subtype: Optional[str], type_path: str) -> Optional[str]:
|
||||
"""Return the subtype-driven subdirectory under a type's `base_dir`, from
|
||||
the type-spec's own `layout:` frontmatter. Returns None for types with no
|
||||
`layout:` (flat directory). Falls back to `<subtype>s` for a subtype the
|
||||
layout doesn't list, matching the previous hand-maintained behavior."""
|
||||
if subtype is None:
|
||||
return None
|
||||
try:
|
||||
layout = resolver.get_layout(type_path)
|
||||
except ValueError:
|
||||
layout = None
|
||||
if layout is None:
|
||||
return None
|
||||
return layout.get(subtype, {}).get("dir", subtype + "s")
|
||||
|
||||
|
||||
def _target_dir(type_path: str, frontmatter: Dict[str, Any]) -> Path:
|
||||
"""Resolve where an instance of this type is written: `<root>/<base_dir>`,
|
||||
plus a subtype subdirectory when the type declares a `layout:`.
|
||||
|
||||
`base_dir` is resolved against `config.KB_DIR` by default, so tests that
|
||||
point KB_DIR at a temporary fixture wiki can never write into the real
|
||||
`kb/`. A type-spec declaring `root: repo` resolves against `config.ROOT`
|
||||
instead - for artifacts that are agent-directed material rather than
|
||||
knowledge, and so live outside the knowledge layer."""
|
||||
base_dir = resolver.get_base_dir(type_path)
|
||||
if not base_dir:
|
||||
fail(
|
||||
f"Type {type_path} declares no `base_dir:` and cannot be "
|
||||
f"instantiated as a page"
|
||||
)
|
||||
try:
|
||||
root = resolver.get_root(type_path)
|
||||
except ValueError as exc:
|
||||
fail(str(exc))
|
||||
target = (config.ROOT if root == "repo" else config.KB_DIR) / base_dir
|
||||
subtype_field = resolver.get_subtype_field(type_path)
|
||||
if subtype_field:
|
||||
subdir = _page_subdir(frontmatter.get(subtype_field), type_path)
|
||||
if subdir:
|
||||
target = target / subdir
|
||||
return target
|
||||
|
||||
|
||||
def _validate_or_fail(frontmatter: Dict[str, Any], type_path: str, source_dir: Path) -> None:
|
||||
"""Validate frontmatter against its type-spec schema, converting a
|
||||
ValueError into the CLI's normal friendly-failure path instead of an
|
||||
uncaught traceback. The schema's own error message (which already names
|
||||
the offending field and, for enums, lists the valid values) is shown
|
||||
as-is - there is no separate hand-maintained validity check to keep in
|
||||
sync with it."""
|
||||
try:
|
||||
resolver.validate_frontmatter(frontmatter, type_path, source_dir)
|
||||
except ValueError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
|
||||
def _load_type_or_fail(type_path: str, source_dir: Path):
|
||||
"""Resolve a type path and load its type-spec + template, converting an
|
||||
unresolvable/invalid `--type` into the CLI's normal friendly-failure path."""
|
||||
try:
|
||||
return _resolve_type_and_get_template(type_path, source_dir)
|
||||
except ValueError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
|
||||
def new_page_command(
|
||||
type_name: str = typer.Argument(
|
||||
...,
|
||||
help="Type name, e.g. entity|concept|source|comparison (see `wikitool types list`)",
|
||||
),
|
||||
name: str = typer.Option(..., "--name", help="Page title (a type's title_prefix is added automatically)"),
|
||||
type_path_override: str = typer.Option(
|
||||
"", "--type", help="Override the type-spec path (defaults to the one named by TYPE_NAME)"
|
||||
),
|
||||
set_fields: Optional[list[str]] = typer.Option(
|
||||
None,
|
||||
"--set",
|
||||
help="Frontmatter field, repeatable: --set entity_type=tool --set tags=a,b. Array values split on commas (escape a literal one as \\,); repeating --set for an array field appends instead of replacing",
|
||||
),
|
||||
):
|
||||
"""Scaffold a new wiki page of any type.
|
||||
|
||||
The type's own type-spec drives everything: which frontmatter fields
|
||||
exist and are required (its `.schema.yaml`), their scaffold defaults
|
||||
(schema `default:`), where the page is written (`base_dir` + `layout`),
|
||||
what prefixes its title (`title_prefix`), and its body skeleton (the
|
||||
type-spec's template). Adding a new type therefore needs no change here.
|
||||
"""
|
||||
type_path = type_path_override or resolver.find_type_by_name(type_name)
|
||||
if not type_path:
|
||||
available = sorted(fm.get("name") for _, fm in resolver.list_type_specs())
|
||||
fail(f"No type-spec named '{type_name}'. Available: {', '.join(available)}")
|
||||
|
||||
today = datetime.date.today()
|
||||
|
||||
try:
|
||||
title_prefix = resolver.get_title_prefix(type_path)
|
||||
root = resolver.get_root(type_path)
|
||||
except ValueError as exc:
|
||||
fail(str(exc))
|
||||
page_title = f"{title_prefix}{name}"
|
||||
if root == "kb":
|
||||
# Title collisions matter because wikilinks resolve by title alone, so
|
||||
# two pages sharing a stem are indistinguishable to every link in the
|
||||
# wiki. Artifacts outside kb/ are not addressed by title and are not
|
||||
# part of that namespace, so the check does not apply to them.
|
||||
check_collision(page_title)
|
||||
|
||||
schema = resolver.get_schema(type_path)
|
||||
explicit = parse_set_fields(set_fields, schema)
|
||||
declared = (schema or {}).get("properties", {})
|
||||
if "summary" in declared:
|
||||
explicit.setdefault("summary", _default_summary(""))
|
||||
if "name" in declared:
|
||||
# The CLI already has this value; a type that stores its own name in
|
||||
# frontmatter should not have to be told it twice.
|
||||
explicit.setdefault("name", name)
|
||||
if "description" in declared:
|
||||
explicit.setdefault("description", "TODO: add description")
|
||||
|
||||
frontmatter = _build_frontmatter(type_path, schema, today, explicit)
|
||||
|
||||
target_dir = _target_dir(type_path, frontmatter)
|
||||
_type_spec, template = _load_type_or_fail(type_path, target_dir)
|
||||
_validate_or_fail(frontmatter, type_path, target_dir)
|
||||
|
||||
if "raw_files" in frontmatter:
|
||||
check_raw_files_exist(frontmatter["raw_files"])
|
||||
|
||||
path = target_dir / f"{page_title}.md"
|
||||
body = _apply_template_variables(
|
||||
template, {**frontmatter, "name": name, "today": today.isoformat()}
|
||||
)
|
||||
|
||||
write_page(path, frontmatter, body)
|
||||
success(f"Created {rel_path(path)}")
|
||||
@@ -0,0 +1,377 @@
|
||||
"""`wikitool rename` / `wikitool rm` - the two page mutations that had no command.
|
||||
|
||||
A page's title is the wiki's only identifier for it, so renaming or deleting a
|
||||
page is never just a filesystem operation: the title appears in every other
|
||||
page's body `[[wikilinks]]`, in the `[[Title]]` a `[^cite-id]` footnote
|
||||
definition points at, and in page-reference frontmatter arrays (`related:`,
|
||||
`sources:`, `entities:`, `concepts:`).
|
||||
|
||||
Doing this by hand is what left four pages citing
|
||||
`Source - Docker Cheatsheet.md` when the page is
|
||||
`Source - Docker Cheatsheet` - and because `lint`'s broken-link check
|
||||
only walked page bodies, nothing ever reported it.
|
||||
|
||||
Which frontmatter fields hold page titles comes from each type-spec's
|
||||
`page_ref_fields:`, so a new type needs no change here.
|
||||
|
||||
Neither command is atomic: both write one page at a time. Both are idempotent
|
||||
per page, so a retry after a partial failure is safe.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import check_collision, fail, rel_path, success
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.page import Page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.provenance import (
|
||||
CITE_REF_RE,
|
||||
cite_id,
|
||||
cite_block_heading,
|
||||
render_page_body,
|
||||
split_cite_block,
|
||||
unique_cite_id,
|
||||
)
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
# `[[Target]]`, `[[Target|alias]]`, `[[Target#anchor]]` - including the
|
||||
# `[[Target]]` inside a `[^cite-id]: [[Target]]` Footnotes definition, which
|
||||
# is exactly what lets retarget_body() repoint a citation's link target on a
|
||||
# rename. Group 1 is the target title; group 2 keeps any alias/anchor suffix
|
||||
# untouched. The id itself is a separate concern - see retarget_cite_ids().
|
||||
LINK_RE = re.compile(r"\[\[([^\[\]|#]+)((?:[|#][^\[\]]*)?)\]\]")
|
||||
|
||||
|
||||
def page_ref_fields(page: Page) -> list[str]:
|
||||
"""The page's declared page-title frontmatter fields, or [] if its type
|
||||
can't be resolved (lint reports that separately)."""
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
return []
|
||||
try:
|
||||
return resolver.get_page_ref_fields(type_path, page.path)
|
||||
except ValueError:
|
||||
return []
|
||||
|
||||
|
||||
def retarget_body(body: str, old: str, new: str) -> str:
|
||||
"""Repoint every wikilink and citation marker aimed at `old` to `new`,
|
||||
preserving any `|alias` or `#anchor` suffix."""
|
||||
|
||||
def replace(match: re.Match) -> str:
|
||||
target, suffix = match.group(1), match.group(2)
|
||||
if target.strip() != old:
|
||||
return match.group(0)
|
||||
return f"[[{new}{suffix}]]"
|
||||
|
||||
return LINK_RE.sub(replace, body)
|
||||
|
||||
|
||||
def retarget_cite_ids(body: str, old: str, new: str) -> str:
|
||||
"""After retarget_body() has already repointed a Footnotes definition's
|
||||
`[[old]]` link target to `[[new]]`, also refresh a citation id that was
|
||||
*derived* from `old`'s slug - `[^s-old-title]` -> `[^s-new-title]` - in
|
||||
both the definition and every inline `[^id]` reference to it.
|
||||
|
||||
An id not derived from `old` (hand-picked, or a `-2`/`-3` collision
|
||||
suffix from an unrelated pair) is left untouched; body is returned
|
||||
unchanged if nothing needs renaming.
|
||||
"""
|
||||
head, definitions = split_cite_block(body)
|
||||
if not definitions:
|
||||
return body
|
||||
|
||||
renames: dict[str, str] = {}
|
||||
new_definitions: dict[str, tuple[str, Optional[str]]] = {}
|
||||
reserved = set(definitions)
|
||||
for cid, (title, qualifier) in definitions.items():
|
||||
if title == new and cid == cite_id(old, qualifier):
|
||||
new_id = unique_cite_id(reserved - {cid}, new, qualifier)
|
||||
renames[cid] = new_id
|
||||
reserved.add(new_id)
|
||||
new_definitions[new_id] = (title, qualifier)
|
||||
else:
|
||||
new_definitions[cid] = (title, qualifier)
|
||||
|
||||
if not renames:
|
||||
return body
|
||||
|
||||
new_head = CITE_REF_RE.sub(lambda m: f"[^{renames.get(m.group(1), m.group(1))}]", head)
|
||||
return render_page_body(new_head, new_definitions, cite_block_heading(body))
|
||||
|
||||
|
||||
def retarget_frontmatter(page: Page, old: str, new: str) -> bool:
|
||||
"""Repoint `old` to `new` in every declared page-ref field. Returns True if
|
||||
anything changed."""
|
||||
changed = False
|
||||
for field in page_ref_fields(page):
|
||||
values = page.frontmatter.get(field)
|
||||
if not values:
|
||||
continue
|
||||
updated = [new if value == old else value for value in values]
|
||||
if updated != values:
|
||||
page.frontmatter[field] = updated
|
||||
changed = True
|
||||
return changed
|
||||
|
||||
|
||||
def _ref_fields_to_sweep(page: Page) -> list[str]:
|
||||
"""Every field this page might hold a reference in.
|
||||
|
||||
The type's own declaration, plus any field already present on the page that
|
||||
*some* type declares as a reference field. The second half exists because a
|
||||
command can leave a reference in a field this type does not declare - `xref
|
||||
add` wrote `related:` on source pages until 1.6.0 - and clearing exactly
|
||||
that kind of leftover is what `xref remove` promises to be for. Sweeping
|
||||
only declared fields made the state unreachable.
|
||||
|
||||
The extra names come from the type-specs rather than a constant here, so a
|
||||
new reference field is swept without a code change.
|
||||
"""
|
||||
declared = page_ref_fields(page)
|
||||
known = {
|
||||
field
|
||||
for _path, frontmatter in resolver.list_type_specs()
|
||||
for field in (frontmatter.get("page_ref_fields") or [])
|
||||
}
|
||||
extra = [f for f in page.frontmatter if f not in declared and f in known]
|
||||
return declared + extra
|
||||
|
||||
|
||||
def strip_frontmatter_ref(page: Page, title: str) -> bool:
|
||||
"""Drop `title` from every page-ref field. Returns True if anything changed.
|
||||
|
||||
An *undeclared* field that ends up empty is removed outright rather than
|
||||
left as `field: []`: it was never valid for this type, and leaving the key
|
||||
keeps the page failing schema validation for a reference that is gone.
|
||||
"""
|
||||
changed = False
|
||||
declared = page_ref_fields(page)
|
||||
for field in _ref_fields_to_sweep(page):
|
||||
values = page.frontmatter.get(field)
|
||||
if not values:
|
||||
continue
|
||||
updated = [value for value in values if value != title]
|
||||
if updated == values:
|
||||
continue
|
||||
if not updated and field not in declared:
|
||||
del page.frontmatter[field]
|
||||
else:
|
||||
page.frontmatter[field] = updated
|
||||
changed = True
|
||||
return changed
|
||||
|
||||
|
||||
def strip_link_bullets(body: str, title: str) -> str:
|
||||
"""Remove whole-line list bullets that exist only to point at `title` -
|
||||
`- [[Title]]` (See Also) and `- **label:** [[Title]]` (Relationships).
|
||||
|
||||
Deliberately narrow: a bullet carrying prose alongside the link, and a
|
||||
`[^cite-id]: [[Title]]` Footnotes definition line (which never starts
|
||||
with `-`, so the pattern below cannot match it), are left alone. Removing
|
||||
a citation is an editorial judgment about a claim, not a mechanical
|
||||
de-linking.
|
||||
"""
|
||||
escaped = re.escape(title)
|
||||
pattern = re.compile(
|
||||
rf"^[ \t]*-[ \t]+(?:\*\*[^*\n]+:\*\*[ \t]+)?\[\[{escaped}\]\][ \t]*\n?",
|
||||
re.MULTILINE,
|
||||
)
|
||||
return pattern.sub("", body)
|
||||
|
||||
|
||||
def body_references(body: str, title: str) -> int:
|
||||
"""How many wikilinks in `body` still point at `title`."""
|
||||
return sum(1 for match in LINK_RE.finditer(body) if match.group(1).strip() == title)
|
||||
|
||||
|
||||
def inbound_pages(pages: dict[str, Page], title: str) -> list[str]:
|
||||
"""Every page (other than `title` itself) referencing it from its body or
|
||||
from a declared page-ref frontmatter field."""
|
||||
found = set()
|
||||
for other_title, page in pages.items():
|
||||
if other_title == title:
|
||||
continue
|
||||
if body_references(page.body, title):
|
||||
found.add(other_title)
|
||||
continue
|
||||
if any(title in (page.frontmatter.get(f) or []) for f in page_ref_fields(page)):
|
||||
found.add(other_title)
|
||||
return sorted(found)
|
||||
|
||||
|
||||
def rename_command(
|
||||
old: str = typer.Option(..., "--from", help="Current page title, exactly as it appears"),
|
||||
new: str = typer.Option(..., "--to", help="New page title"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="List what would change instead of writing"),
|
||||
):
|
||||
"""Rename a page, or repoint references that name a page that never existed.
|
||||
|
||||
Two modes, chosen by whether `--from` is an actual page:
|
||||
|
||||
- `--from` is a page: it is renamed to `--to` (which must be free) and every
|
||||
reference follows.
|
||||
- `--from` is not a page but is referenced: references are repointed to
|
||||
`--to`, which must already exist. This is the cleanup case - a reference
|
||||
spelled `act_runner` when the page is `Act Runner`, or
|
||||
`Source - X.md` when the page is `Source - X`. Nothing moves on disk.
|
||||
"""
|
||||
if old == new:
|
||||
fail("--from and --to are the same title; nothing to rename.")
|
||||
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
target = pages.get(old)
|
||||
references_only = target is None
|
||||
|
||||
if references_only:
|
||||
if new not in pages:
|
||||
fail(
|
||||
f"Neither '{old}' nor '{new}' is a page under wiki/. Repointing references "
|
||||
f"to '{new}' would just move the dangling reference; create the page first "
|
||||
"with `wikitool new ...`, or drop the reference with `wikitool xref remove`."
|
||||
)
|
||||
elif not dry_run:
|
||||
check_collision(new)
|
||||
elif new in pages:
|
||||
fail(f"A page titled '{new}' already exists at {rel_path(pages[new].path)}")
|
||||
|
||||
touched: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
new_body = retarget_body(page.body, old, new)
|
||||
new_body = retarget_cite_ids(new_body, old, new)
|
||||
if title == old and page.h1_title == old:
|
||||
new_body = re.sub(rf"^# {re.escape(old)}$", f"# {new}", new_body, count=1, flags=re.MULTILINE)
|
||||
frontmatter_changed = retarget_frontmatter(page, old, new)
|
||||
if new_body == page.body and not frontmatter_changed:
|
||||
continue
|
||||
touched.append(title)
|
||||
if not dry_run:
|
||||
try:
|
||||
write_page(page.path, page.frontmatter, new_body)
|
||||
except OSError as exc:
|
||||
failed.append(f"{title} ({exc})")
|
||||
|
||||
if failed:
|
||||
fail(
|
||||
f"Updated references in {len(touched) - len(failed)}/{len(touched)} page(s) before a write "
|
||||
f"failed: {', '.join(failed)}. Nothing was renamed on disk, so '{old}' is unchanged - check "
|
||||
"`git status`, resolve the write failure (permissions/disk), then re-run the full `rename` "
|
||||
"command (safe to retry - each page's rewrite is idempotent)."
|
||||
)
|
||||
|
||||
for title in touched:
|
||||
typer.echo(f" updated references in '{title}'")
|
||||
|
||||
if references_only:
|
||||
if not touched:
|
||||
success(f"Nothing references '{old}'; nothing to repoint.")
|
||||
return
|
||||
if dry_run:
|
||||
typer.echo(f"[dry-run] would repoint {len(touched)} page(s) to '{new}'. No files written.")
|
||||
return
|
||||
success(
|
||||
f"Repointed references from '{old}' to the existing page '{new}' in "
|
||||
f"{len(touched)} page(s). No file was moved ('{old}' was not a page)."
|
||||
)
|
||||
return
|
||||
|
||||
new_path = target.path.parent / f"{new}.md"
|
||||
if dry_run:
|
||||
typer.echo(f"[dry-run] would rename {rel_path(target.path)} -> {rel_path(new_path)}")
|
||||
typer.echo(f"[dry-run] would update {len(touched)} page(s). No files written.")
|
||||
return
|
||||
|
||||
target.path.rename(new_path)
|
||||
success(
|
||||
f"Renamed '{old}' -> '{new}' ({rel_path(new_path)}); "
|
||||
f"updated references in {len(touched)} page(s). "
|
||||
"Run `wikitool index rebuild` and `wikitool sources rebuild-index` next."
|
||||
)
|
||||
|
||||
|
||||
def rm_command(
|
||||
page_title: str = typer.Option(..., "--page", help="Exact title of the page to delete"),
|
||||
yes: bool = typer.Option(
|
||||
False,
|
||||
"--yes",
|
||||
"-y",
|
||||
help="Confirm deletion of a page that other pages still reference. Only pass this after "
|
||||
"a human has reviewed the inbound list - never set it automatically.",
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="List what would change instead of writing"),
|
||||
):
|
||||
"""Delete a page and mechanically de-link it from the rest of the wiki."""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
target = pages.get(page_title)
|
||||
if target is None:
|
||||
fail(f"No page titled '{page_title}' found under wiki/.")
|
||||
|
||||
inbound = inbound_pages(pages, page_title)
|
||||
if inbound and not yes:
|
||||
listed = "\n".join(f"- {t}" for t in inbound)
|
||||
fail(
|
||||
f"'{page_title}' is still referenced by {len(inbound)} page(s). Deleting it will leave "
|
||||
"their prose pointing at nothing. Show the user this list and only re-run with --yes "
|
||||
f"once they have approved:\n{listed}"
|
||||
)
|
||||
|
||||
touched: list[str] = []
|
||||
failed: list[str] = []
|
||||
leftover: list[tuple[str, int]] = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
if title == page_title:
|
||||
continue
|
||||
new_body = strip_link_bullets(page.body, page_title)
|
||||
frontmatter_changed = strip_frontmatter_ref(page, page_title)
|
||||
if new_body != page.body or frontmatter_changed:
|
||||
touched.append(title)
|
||||
if not dry_run:
|
||||
try:
|
||||
write_page(page.path, page.frontmatter, new_body)
|
||||
except OSError as exc:
|
||||
failed.append(f"{title} ({exc})")
|
||||
remaining = body_references(new_body, page_title)
|
||||
if remaining:
|
||||
leftover.append((title, remaining))
|
||||
|
||||
if failed:
|
||||
fail(
|
||||
f"De-linked {len(touched) - len(failed)}/{len(touched)} page(s) before a write failed: "
|
||||
f"{', '.join(failed)}. '{page_title}' was NOT deleted, so nothing is orphaned - check "
|
||||
"`git status`, resolve the write failure, then re-run `rm` (safe to retry)."
|
||||
)
|
||||
|
||||
for title in touched:
|
||||
typer.echo(f" de-linked '{title}'")
|
||||
|
||||
if dry_run:
|
||||
typer.echo(f"[dry-run] would delete {rel_path(target.path)}")
|
||||
typer.echo(f"[dry-run] would update {len(touched)} page(s). No files written.")
|
||||
else:
|
||||
target.path.unlink()
|
||||
|
||||
if leftover:
|
||||
typer.echo("")
|
||||
typer.echo(
|
||||
"Prose references left in place - these carry claims, so removing them is an "
|
||||
"editorial call, not a mechanical one:"
|
||||
)
|
||||
for title, count in leftover:
|
||||
typer.echo(f" - {title}: {count} remaining [[{page_title}]] reference(s)")
|
||||
typer.echo("Fix them, then re-run `wikitool lint`.")
|
||||
|
||||
if dry_run:
|
||||
return
|
||||
success(
|
||||
f"Deleted '{page_title}' ({rel_path(target.path)}); de-linked {len(touched)} page(s). "
|
||||
"Run `wikitool index rebuild` and `wikitool sources rebuild-index` next."
|
||||
)
|
||||
@@ -0,0 +1,184 @@
|
||||
"""`wikitool sources ...` - raw-file <-> wiki provenance tooling.
|
||||
|
||||
This is the deterministic backbone for citation backtracing: it never guesses
|
||||
which raw file backs a claim, it only reports what the frontmatter and inline
|
||||
`[^cite-id]` footnotes already declare. Filling in those declarations
|
||||
correctly is still the LLM's job.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, rel_path, success
|
||||
from chemenu.provenance import (
|
||||
broken_raw_refs,
|
||||
citing_pages,
|
||||
legacy_source_pages,
|
||||
page_raw_files,
|
||||
source_pages_by_raw_file,
|
||||
source_raw_files,
|
||||
uncovered_raw_files,
|
||||
)
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
|
||||
app = typer.Typer(help="Trace and lint raw-file <-> wiki-page provenance.")
|
||||
|
||||
|
||||
def _normalize_raw_path(raw: str) -> str:
|
||||
"""Accept an absolute path or a path relative to the repo root or to raw/,
|
||||
and return it as a path relative to the repo root (matching how it is
|
||||
stored in `raw_files:`)."""
|
||||
candidate = Path(raw)
|
||||
if candidate.is_absolute():
|
||||
try:
|
||||
return str(candidate.relative_to(config.ROOT))
|
||||
except ValueError:
|
||||
return str(candidate)
|
||||
if candidate.exists():
|
||||
return str(candidate)
|
||||
if (config.ROOT / candidate).exists():
|
||||
return str(candidate)
|
||||
if (config.RAW_DIR / candidate).exists():
|
||||
return str((Path("raw") / candidate))
|
||||
return raw
|
||||
|
||||
|
||||
@app.command("coverage")
|
||||
def coverage(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
|
||||
"""Report raw files with no source page, broken raw_files: references, and
|
||||
source pages still using a legacy directory/URL-only `source:` field."""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
report = {
|
||||
"uncovered_raw_files": uncovered_raw_files(config.RAW_DIR, pages),
|
||||
"broken_raw_refs": broken_raw_refs(pages),
|
||||
"legacy_source_pages": legacy_source_pages(pages),
|
||||
}
|
||||
if json_out:
|
||||
typer.echo(json.dumps(report, indent=2))
|
||||
return
|
||||
|
||||
typer.echo(f"Uncovered raw files: {len(report['uncovered_raw_files'])}")
|
||||
for f in report["uncovered_raw_files"]:
|
||||
typer.echo(f" - {f}")
|
||||
typer.echo(f"Broken raw_files references: {len(report['broken_raw_refs'])}")
|
||||
for item in report["broken_raw_refs"]:
|
||||
typer.echo(f" - [[{item['page']}]] -> {item['raw_path']}")
|
||||
typer.echo(f"Legacy (directory/URL-only) source pages: {len(report['legacy_source_pages'])}")
|
||||
for item in report["legacy_source_pages"]:
|
||||
typer.echo(f" - [[{item['page']}]] ({item['reason']}): {item['source']}")
|
||||
|
||||
|
||||
@app.command("trace")
|
||||
def trace(
|
||||
raw: Optional[str] = typer.Option(None, "--raw", help="Raw file path to trace forward from"),
|
||||
page: Optional[str] = typer.Option(None, "--page", help="Wiki page title to trace backward from"),
|
||||
):
|
||||
"""Trace provenance in either direction: --raw shows which source pages
|
||||
cover a raw file and which wiki pages cite it; --page shows which sources
|
||||
and raw files back a given wiki page."""
|
||||
if bool(raw) == bool(page):
|
||||
fail("Provide exactly one of --raw or --page")
|
||||
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
|
||||
if raw:
|
||||
raw_key = _normalize_raw_path(raw)
|
||||
by_raw = source_pages_by_raw_file(pages)
|
||||
source_titles = by_raw.get(raw_key, [])
|
||||
if not source_titles:
|
||||
typer.echo(f"No source page covers {raw_key}")
|
||||
raise typer.Exit(code=1)
|
||||
for source_title in source_titles:
|
||||
typer.echo(f"{raw_key}")
|
||||
typer.echo(f" covered by: [[{source_title}]]")
|
||||
citers = citing_pages(pages, source_title)
|
||||
if citers:
|
||||
for c in citers:
|
||||
typer.echo(f" cited by: [[{c}]]")
|
||||
else:
|
||||
typer.echo(" cited by: (nothing yet)")
|
||||
return
|
||||
|
||||
target_page = pages.get(page)
|
||||
if target_page is None:
|
||||
fail(f"No page titled '{page}' found")
|
||||
sources = target_page.frontmatter.get("sources") or []
|
||||
typer.echo(f"[[{page}]]")
|
||||
if not sources:
|
||||
typer.echo(" sources: (none listed)")
|
||||
for source_title in sources:
|
||||
typer.echo(f" sources: [[{source_title}]]")
|
||||
source_page = pages.get(source_title)
|
||||
if source_page is None:
|
||||
typer.echo(" (source page not found)")
|
||||
continue
|
||||
for raw_path in source_raw_files(source_page):
|
||||
typer.echo(f" raw file: {raw_path}")
|
||||
raw_files = page_raw_files(pages, target_page)
|
||||
typer.echo(f" all raw files (incl. inline citations): {raw_files or '(none)'}")
|
||||
|
||||
|
||||
def build_provenance_index(kb_dir: Path, raw_dir: Path) -> str:
|
||||
pages = load_kb_pages(kb_dir)
|
||||
by_raw = source_pages_by_raw_file(pages)
|
||||
all_raw = sorted(str(p.relative_to(config.ROOT)) for p in config.iter_raw_files(raw_dir))
|
||||
uncovered = uncovered_raw_files(raw_dir, pages)
|
||||
|
||||
lines: list[str] = []
|
||||
lines.append("# Provenance Index")
|
||||
lines.append("")
|
||||
lines.append("Generated by `tools/wikitool sources rebuild-index`. Do not hand-edit.")
|
||||
lines.append("")
|
||||
lines.append("Maps every raw source file to the wiki source page(s) that cover it, and")
|
||||
lines.append("every wiki page that cites that source (via frontmatter `sources:` or an")
|
||||
lines.append("inline `[^cite-id]` footnote).")
|
||||
lines.append("")
|
||||
lines.append("## Coverage Summary")
|
||||
lines.append("")
|
||||
lines.append(f"- **Total raw files:** {len(all_raw)}")
|
||||
lines.append(f"- **Covered:** {len(all_raw) - len(uncovered)}")
|
||||
lines.append(f"- **Uncovered:** {len(uncovered)}")
|
||||
lines.append("")
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
lines.append("## Raw Files")
|
||||
lines.append("")
|
||||
|
||||
for raw_path in all_raw:
|
||||
lines.append(f"### `{raw_path}`")
|
||||
lines.append("")
|
||||
source_titles = by_raw.get(raw_path, [])
|
||||
if not source_titles:
|
||||
lines.append("No source page covers this file yet.")
|
||||
lines.append("")
|
||||
continue
|
||||
for source_title in source_titles:
|
||||
lines.append(f"- Covered by: [[{source_title}]]")
|
||||
citers = citing_pages(pages, source_title)
|
||||
if citers:
|
||||
lines.append(f" - Cited by: {', '.join(f'[[{c}]]' for c in citers)}")
|
||||
else:
|
||||
lines.append(" - Cited by: (nothing yet)")
|
||||
lines.append("")
|
||||
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
@app.command("rebuild-index")
|
||||
def rebuild_index(
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Print the result instead of writing wiki/provenance.md"),
|
||||
):
|
||||
content = build_provenance_index(config.KB_DIR, config.RAW_DIR)
|
||||
provenance_file = config.KB_DIR / "provenance.md"
|
||||
if dry_run:
|
||||
# nl=False so the preview is byte-identical to the file that would be
|
||||
# written; see the same note in index_build.py.
|
||||
typer.echo(content, nl=False)
|
||||
return
|
||||
provenance_file.write_text(content, encoding="utf-8")
|
||||
success(f"Rebuilt {rel_path(provenance_file)}")
|
||||
@@ -0,0 +1,355 @@
|
||||
"""Iteration/cost budget gate: a hard, code-enforced cap on how many wikitool
|
||||
commands a single agent session may run before requiring explicit human
|
||||
confirmation, plus a loop-breaker that trips immediately if the last few
|
||||
calls are near-identical (same command + same arguments).
|
||||
|
||||
This closes the gap documented in AGENTS.md's "Gates" section: unlike a
|
||||
prompt instruction ("stop after N steps"), this check runs
|
||||
in-process on every `wikitool` invocation and cannot be skipped by the
|
||||
calling agent "politely trying again". It mirrors the Mass-Update Gate
|
||||
pattern (see git_publish.py / wiki/concepts/Mass-Update Gate.md), but that
|
||||
gate is scoped to the *size* of a single publish, while this one is scoped to
|
||||
*iteration volume* across a whole session (e.g. a wiki-ingest or wiki-lint run
|
||||
that could otherwise loop unbounded over many entity/concept pages).
|
||||
|
||||
Session scoping: a "session" is approximated by the parent process of this
|
||||
CLI invocation (the agent's shell), via the WIKITOOL_SESSION_ID env var if the
|
||||
caller sets one, otherwise os.getppid(). A new terminal/session therefore
|
||||
starts with a fresh budget.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, success
|
||||
from chemenu.session import session_id as _shared_session_id
|
||||
from chemenu.session import session_id_source as _shared_session_id_source
|
||||
from chemenu.telemetry import emit
|
||||
|
||||
app = typer.Typer(help="Session iteration/cost budget gate (see the tooling contract's 'Iteration and Cost Limits').")
|
||||
|
||||
STATE_DIR = config.ROOT / "tools" / ".wikitool_session"
|
||||
STATE_FILE = STATE_DIR / "budget.json"
|
||||
LOCK_FILE = STATE_DIR / "budget.lock"
|
||||
|
||||
# Calibration, measured in this instance rather than inherited: ~5-15 calls for
|
||||
# a simple task, ~20-35 for a complex multi-tool workflow such as an ingest.
|
||||
#
|
||||
# The upper band used to read 15-25, taken from an industry rule of thumb (see
|
||||
# kb/concepts/Iteration and Cost Limits.md, which still cites it as such). Four
|
||||
# consecutive real ingests measured 24, 26, 29 and 30 calls - every one of them
|
||||
# at or above the old band's ceiling while doing nothing unusual. A guideline
|
||||
# that the normal case exceeds is not a guideline; it teaches an agent that the
|
||||
# numbers are decorative.
|
||||
#
|
||||
# The limit sits well above the band on purpose. It is not a target but the
|
||||
# point past which a session is presumed stuck. At 30 the ingest of 2026-08-30
|
||||
# hit it on overhead alone - reading a report back, a corrected retry, checking
|
||||
# the tree before publishing - which is the gate firing on the tool rather than
|
||||
# on the task.
|
||||
DEFAULT_CALL_LIMIT = 60
|
||||
|
||||
# Loop-breaker: abort if the last N calls all share the same command + args,
|
||||
# even if the overall call limit hasn't been reached yet.
|
||||
DEFAULT_LOOP_WINDOW = 3
|
||||
|
||||
# Sessions untouched for this long are dropped on the next write. Without this
|
||||
# the state file grows one entry per session forever - and the getppid()
|
||||
# fallback makes new keys cheap (a new shell is a new session).
|
||||
SESSION_TTL_SECONDS = 7 * 24 * 3600
|
||||
|
||||
# Never gate the gate's *read* side, or reporting the situation to the user
|
||||
# would become impossible exactly when the limit trips. `budget reset` is
|
||||
# deliberately NOT exempt: it clears the counter, so exempting it would make
|
||||
# the whole gate a formality an agent could step around by resetting first.
|
||||
# It is gated on `--yes` instead, the same way `publish` is.
|
||||
#
|
||||
# `eval score` and `eval sessions` read a trace and re-run lint's checks in
|
||||
# process. Reading back what a session already did is not iteration on the wiki,
|
||||
# and charging for it would discourage checking one's own work.
|
||||
# `version show`/`check`/`notes` only read - `VERSION`, the release stamp, the
|
||||
# changelog, or a remote release feed. `version bump` writes two files and
|
||||
# stays counted like every other mutation.
|
||||
SKIP_COMMAND_PATHS = {
|
||||
("budget", "status"),
|
||||
("eval", "score"),
|
||||
("eval", "sessions"),
|
||||
("cite", "id"),
|
||||
("version", "show"),
|
||||
("version", "check"),
|
||||
("version", "notes"),
|
||||
# Bare `wikitool version` (and `version --json`) is an alias for `show`;
|
||||
# the subcommand slot is empty, so it needs its own entry to be exempt
|
||||
# alongside the command it delegates to.
|
||||
("version", ""),
|
||||
# `migrate list/status/verify` only read - the migration documents, the KB
|
||||
# state file, and git history. `verify` especially: a migration runs it
|
||||
# once per unit by design, and charging for the check would push an agent
|
||||
# toward skipping the one step that catches a dropped reference.
|
||||
# `migrate done`/`baseline` write the state file and stay counted.
|
||||
("migrate", "list"),
|
||||
("migrate", "status"),
|
||||
("migrate", "verify"),
|
||||
}
|
||||
|
||||
# Commands exempt regardless of their first argument, because that argument is
|
||||
# a query rather than a subcommand. `search` is here because retrieval is
|
||||
# reading, not iterating: the budget exists to stop an agent looping over the
|
||||
# wiki's *state*, and charging for a search would penalise the one habit that
|
||||
# lowers cost - looking before reading. `doctor` is here for the same reason:
|
||||
# it only reads and reports, never mutates anything. Every command that
|
||||
# mutates anything stays counted.
|
||||
SKIP_COMMANDS = {"search", "doctor"}
|
||||
|
||||
|
||||
def is_exempt(command: str, args: list[str]) -> bool:
|
||||
"""Whether this invocation is outside the budget entirely."""
|
||||
if command in SKIP_COMMANDS:
|
||||
return True
|
||||
subcommand = args[0] if args and not args[0].startswith("-") else ""
|
||||
return (command, subcommand) in SKIP_COMMAND_PATHS
|
||||
|
||||
|
||||
def _session_id() -> str:
|
||||
return _shared_session_id()
|
||||
|
||||
|
||||
def _session_id_source() -> str:
|
||||
return _shared_session_id_source()
|
||||
|
||||
|
||||
def _load_state() -> dict:
|
||||
if not STATE_FILE.exists():
|
||||
return {}
|
||||
try:
|
||||
return json.loads(STATE_FILE.read_text())
|
||||
except (json.JSONDecodeError, OSError):
|
||||
return {}
|
||||
|
||||
|
||||
def prune_state(state: dict, now: float, ttl: float = SESSION_TTL_SECONDS) -> dict:
|
||||
"""Drop sessions whose last recorded call is older than the TTL. Entries
|
||||
written before `last_seen` existed are kept (they get a timestamp on their
|
||||
next recorded call)."""
|
||||
return {
|
||||
session_id: entry
|
||||
for session_id, entry in state.items()
|
||||
if "last_seen" not in entry or now - entry["last_seen"] <= ttl
|
||||
}
|
||||
|
||||
|
||||
def _save_state(state: dict) -> None:
|
||||
"""Write the state atomically: build the payload, write it to a sibling
|
||||
temp file, then rename it over the real file. A crash or concurrent read
|
||||
mid-write can never observe a truncated/partial JSON file this way -
|
||||
os.replace() is atomic on POSIX."""
|
||||
STATE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
payload = json.dumps(prune_state(state, time.time()), indent=2)
|
||||
tmp_file = STATE_FILE.with_suffix(STATE_FILE.suffix + ".tmp")
|
||||
tmp_file.write_text(payload)
|
||||
os.replace(tmp_file, STATE_FILE)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _state_lock():
|
||||
"""Exclusive cross-process lock guarding the budget state's load-modify-
|
||||
save cycle. Without this, two `wikitool` calls racing in the same session
|
||||
(e.g. two parallel subagents) can both load count=N, both compute N+1, and
|
||||
both save - losing an increment and letting the session run past the gate
|
||||
it exists to enforce. POSIX-only (fcntl); best-effort no-op if unavailable,
|
||||
since the loop-breaker's identical-call check still degrades gracefully."""
|
||||
STATE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
try:
|
||||
import fcntl
|
||||
except ImportError: # pragma: no cover - non-POSIX platform
|
||||
yield
|
||||
return
|
||||
with open(LOCK_FILE, "w") as lock_fh:
|
||||
fcntl.flock(lock_fh, fcntl.LOCK_EX)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
||||
|
||||
|
||||
def loop_breaker_message(call_signature: str, loop_window: int) -> str:
|
||||
return (
|
||||
f"Loop-Breaker: the last {loop_window} wikitool calls in this session were "
|
||||
f"identical ('{call_signature}'). This usually means the agent is stuck retrying "
|
||||
"the same failing operation instead of changing approach - per the tooling contract's "
|
||||
"Tool Error Contracts, that is exactly the case for stopping and escalating rather than "
|
||||
"retrying again. Stop, explain the situation to the user, and get explicit direction "
|
||||
"before continuing. Only re-run with --override-budget once the user has confirmed "
|
||||
"the repeat is intentional - never add it on the agent's own initiative."
|
||||
)
|
||||
|
||||
|
||||
def call_limit_message(count: int, call_limit: int) -> str:
|
||||
return (
|
||||
f"Iteration Budget Gate: this session has made {count} wikitool calls, exceeding the "
|
||||
f"limit of {call_limit}. Per the tooling contract's 'Iteration and Cost Limits' section, a "
|
||||
"single task should typically need roughly 5-15 calls (simple) or 20-35 (complex multi-tool "
|
||||
"workflow like wiki-ingest/wiki-lint). This far past that band is a documented sign of poor "
|
||||
"task decomposition or a stuck loop. Stop, summarize progress and the blocker to the "
|
||||
"user, and get explicit direction before continuing. Only re-run with --override-budget "
|
||||
"once the user has approved continuing this session - never add it unprompted."
|
||||
)
|
||||
|
||||
|
||||
def record_and_check(
|
||||
command: str,
|
||||
args: list[str],
|
||||
override: bool,
|
||||
call_limit: int = DEFAULT_CALL_LIMIT,
|
||||
loop_window: int = DEFAULT_LOOP_WINDOW,
|
||||
) -> bool:
|
||||
"""Record this invocation against the session budget and enforce the gate.
|
||||
Called once per process from main() before Typer dispatches to a
|
||||
subcommand, so every wikitool command is covered uniformly.
|
||||
|
||||
A refused call is *not* recorded: it never ran, so counting it would keep
|
||||
inflating the number quoted back to the user on every subsequent attempt.
|
||||
The loop-breaker still trips on the next identical call, because the
|
||||
history that made it identical is already stored.
|
||||
|
||||
Returns whether a slot was actually charged, so the caller knows whether
|
||||
there is anything to hand back via `refund()`.
|
||||
"""
|
||||
if not command or is_exempt(command, args):
|
||||
return False
|
||||
|
||||
with _state_lock():
|
||||
session_id = _session_id()
|
||||
state = _load_state()
|
||||
entry = state.setdefault(session_id, {"count": 0, "recent": []})
|
||||
recent = entry["recent"]
|
||||
|
||||
call_signature = f"{command} {' '.join(args)}".strip()
|
||||
|
||||
# Checked against history *before* this call is appended, so it answers
|
||||
# "were the last `loop_window` calls already identical to this one?".
|
||||
is_repeat_of_recent = (
|
||||
len(recent) >= loop_window
|
||||
and all(c == call_signature for c in recent[-loop_window:])
|
||||
)
|
||||
|
||||
if not override:
|
||||
if is_repeat_of_recent:
|
||||
emit(
|
||||
"wikitool",
|
||||
"gate.refused",
|
||||
{
|
||||
"gate": "loop-breaker",
|
||||
"command": command,
|
||||
"args": args,
|
||||
"call_signature": call_signature,
|
||||
"loop_window": loop_window,
|
||||
"count": entry["count"],
|
||||
},
|
||||
)
|
||||
fail(loop_breaker_message(call_signature, loop_window))
|
||||
if entry["count"] + 1 > call_limit:
|
||||
emit(
|
||||
"wikitool",
|
||||
"gate.refused",
|
||||
{
|
||||
"gate": "iteration-budget",
|
||||
"command": command,
|
||||
"args": args,
|
||||
"count": entry["count"] + 1,
|
||||
"limit": call_limit,
|
||||
},
|
||||
)
|
||||
fail(call_limit_message(entry["count"] + 1, call_limit))
|
||||
|
||||
entry["count"] += 1
|
||||
recent.append(call_signature)
|
||||
entry["recent"] = recent[-max(loop_window, 10):]
|
||||
entry["last_seen"] = time.time()
|
||||
_save_state(state)
|
||||
return True
|
||||
|
||||
|
||||
def refund() -> None:
|
||||
"""Give the current session its last charged slot back.
|
||||
|
||||
Called when the command declined instead of acting: a rejected argument,
|
||||
or a read-only check reporting findings (`_util.fail`, exit 1). The
|
||||
tooling contract answers a rejected argument with "fix it and retry once",
|
||||
so charging for the rejection makes the prescribed response cost two slots
|
||||
for one operation - and the budget exists to bound iteration on the wiki,
|
||||
which a call that changed nothing did not do.
|
||||
|
||||
The call stays in `recent`. Repeating the same broken invocation is a real
|
||||
failure, and the loop-breaker is the instrument for it: it needs the
|
||||
history, not the counter.
|
||||
"""
|
||||
with _state_lock():
|
||||
state = _load_state()
|
||||
entry = state.get(_session_id())
|
||||
if not entry or entry.get("count", 0) <= 0:
|
||||
return
|
||||
entry["count"] -= 1
|
||||
_save_state(state)
|
||||
|
||||
|
||||
def status_command():
|
||||
"""Show the current session's call count and recent command history."""
|
||||
state = _load_state()
|
||||
entry = state.get(_session_id())
|
||||
typer.echo(f"Session: {_session_id()} (from {_session_id_source()})")
|
||||
if not entry:
|
||||
success("No recorded calls yet for this session.")
|
||||
return
|
||||
typer.echo(f"Calls so far: {entry['count']} (limit {DEFAULT_CALL_LIMIT})")
|
||||
typer.echo("Recent calls:")
|
||||
for c in entry["recent"]:
|
||||
typer.echo(f" - {c}")
|
||||
|
||||
|
||||
def reset_message() -> str:
|
||||
return (
|
||||
"`budget reset` clears the Iteration Budget Gate - the check that exists to stop a "
|
||||
"session looping or sprawling unnoticed. Resetting it on the agent's own initiative "
|
||||
"would make the gate advisory, which is exactly what it was built not to be. Stop, "
|
||||
"summarize what the session has done so far and why it needs more calls, and only "
|
||||
"re-run with --yes once the user has approved continuing."
|
||||
)
|
||||
|
||||
|
||||
def reset_command(
|
||||
all_sessions: bool = typer.Option(
|
||||
False, "--all", help="Clear every session's budget, not just the current one."
|
||||
),
|
||||
yes: bool = typer.Option(
|
||||
False,
|
||||
"--yes",
|
||||
"-y",
|
||||
help="Confirm clearing the budget. Only pass this after a human has approved "
|
||||
"continuing the session - never set it automatically to work around the gate.",
|
||||
),
|
||||
):
|
||||
"""Clear the current session's (or all sessions') recorded budget."""
|
||||
if not yes:
|
||||
fail(reset_message())
|
||||
with _state_lock():
|
||||
if all_sessions:
|
||||
if STATE_FILE.exists():
|
||||
STATE_FILE.unlink()
|
||||
success("Cleared budget state for all sessions.")
|
||||
return
|
||||
state = _load_state()
|
||||
if state.pop(_session_id(), None) is not None:
|
||||
_save_state(state)
|
||||
success(f"Cleared budget state for session {_session_id()}.")
|
||||
|
||||
|
||||
app.command("status")(status_command)
|
||||
app.command("reset")(reset_command)
|
||||
@@ -0,0 +1,221 @@
|
||||
"""`wikitool search` - find pages without reading `kb/index.md`.
|
||||
|
||||
This command exists to make retrieval cheap. Before it, the documented way to
|
||||
find a page was to read the whole generated index; at a few hundred pages that
|
||||
is tens of thousands of tokens spent to learn three filenames. A search returns
|
||||
the same pointers for a fraction of it.
|
||||
|
||||
Two halves, deliberately kept separate:
|
||||
|
||||
- Text search is answered by a pluggable backend (`rg` today) - see
|
||||
`chemenu/search/`.
|
||||
- Frontmatter predicates (`--field`) are evaluated here, in-process, on the
|
||||
structured YAML rather than on its rendering. With no text at all this is a
|
||||
pure structured query, which is how "systems below 0.6 confidence, oldest
|
||||
first" is asked without a second command.
|
||||
|
||||
Scope is `kb/` only. `instructions/` is discovered through
|
||||
`wikitool instructions list`, because a procedure is found by what it is *for*
|
||||
(its description), not by keywords in its body.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, today_iso
|
||||
from chemenu.frontmatter_io import read_page
|
||||
from chemenu.kb_scan import iter_kb_pages
|
||||
from chemenu.page import Page
|
||||
from chemenu.search import filters
|
||||
from chemenu.search.base import page_key
|
||||
from chemenu.search.filters import PredicateError
|
||||
from chemenu.search.fuse import reciprocal_rank_fusion
|
||||
from chemenu.search.registry import UnknownBackend, resolve
|
||||
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing, build_hit
|
||||
from chemenu.search.types import Predicate, SearchHit, SearchQuery
|
||||
|
||||
TITLE_WIDTH = 34
|
||||
SUMMARY_WIDTH = 84
|
||||
|
||||
|
||||
def load_pages_by_path(kb_dir: Path | None = None, root: Path | None = None) -> dict[str, Page]:
|
||||
"""Every page under `kb/`, keyed by repo-relative path.
|
||||
|
||||
Path-keyed rather than title-keyed on purpose: `load_kb_pages()` drops one
|
||||
of two pages sharing a stem, and search should still find both - a
|
||||
duplicate title is a lint finding, not a reason to hide a page.
|
||||
"""
|
||||
kb_dir = kb_dir or config.KB_DIR
|
||||
root = root or config.ROOT
|
||||
pages: dict[str, Page] = {}
|
||||
for path in iter_kb_pages(kb_dir):
|
||||
frontmatter, body = read_page(path)
|
||||
pages[page_key(path, root)] = Page(path=path, frontmatter=frontmatter, body=body)
|
||||
return pages
|
||||
|
||||
|
||||
def _sort_key(hit: SearchHit, field: str):
|
||||
value = hit.as_dict().get(field)
|
||||
if value is None:
|
||||
# Missing values sort last in either direction rather than crashing on
|
||||
# a None comparison.
|
||||
return (1, "")
|
||||
if isinstance(value, (int, float)):
|
||||
return (0, value)
|
||||
return (0, str(value).lower())
|
||||
|
||||
|
||||
def sort_hits(hits: list[SearchHit], sort: str | None) -> list[SearchHit]:
|
||||
"""Sort by a hit field. A leading `-` reverses, e.g. `--sort -confidence`."""
|
||||
if not sort:
|
||||
return hits
|
||||
descending = sort.startswith("-")
|
||||
field = sort.lstrip("-")
|
||||
ordered = sorted(hits, key=lambda h: _sort_key(h, field), reverse=descending)
|
||||
return ordered
|
||||
|
||||
|
||||
def run_search(
|
||||
query: SearchQuery,
|
||||
pages: dict[str, Page],
|
||||
backends: list,
|
||||
kb_dir: Path | None = None,
|
||||
) -> list[SearchHit]:
|
||||
"""Answer a query. Pure: no I/O beyond whatever a backend does."""
|
||||
filters.validate_fields(query.predicates, pages)
|
||||
|
||||
if query.text:
|
||||
rankings = [backend.search(query, pages) for backend in backends]
|
||||
hits = rankings[0] if len(rankings) == 1 else reciprocal_rank_fusion(rankings)
|
||||
allowed = filters.apply_predicates(pages, query.predicates, kb_dir)
|
||||
hits = [hit for hit in hits if hit.path in allowed]
|
||||
else:
|
||||
selected = filters.apply_predicates(pages, query.predicates, kb_dir)
|
||||
hits = [
|
||||
build_hit(page, key, [], query, backend="frontmatter", kb_dir=kb_dir)
|
||||
for key, page in selected.items()
|
||||
]
|
||||
hits.sort(key=lambda h: h.title.lower())
|
||||
|
||||
hits = sort_hits(hits, query.sort)
|
||||
return hits[: query.limit] if query.limit else hits
|
||||
|
||||
|
||||
def _truncate(text: str, width: int) -> str:
|
||||
text = " ".join(text.split())
|
||||
return text if len(text) <= width else text[: width - 1] + "\u2026"
|
||||
|
||||
|
||||
def render_table(hits: list[SearchHit], show_matches: bool) -> str:
|
||||
if not hits:
|
||||
return "No matches."
|
||||
lines = []
|
||||
for hit in hits:
|
||||
kind = hit.kind or "?"
|
||||
if hit.subtype:
|
||||
kind = f"{kind}/{hit.subtype}"
|
||||
lines.append(
|
||||
f"{hit.score:6.1f} {_truncate(hit.title, TITLE_WIDTH):<{TITLE_WIDTH}} "
|
||||
f"{kind:<18} {_truncate(hit.summary, SUMMARY_WIDTH)}"
|
||||
)
|
||||
if show_matches:
|
||||
for match in hit.matches:
|
||||
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
|
||||
lines.append("")
|
||||
lines.append(f"{len(hits)} result(s).")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def search_command(
|
||||
text: str = typer.Argument(
|
||||
None,
|
||||
help="Text to search for. Omit it to run a pure frontmatter query.",
|
||||
),
|
||||
field: list[str] = typer.Option(
|
||||
None,
|
||||
"--field",
|
||||
"-f",
|
||||
help="Frontmatter predicate, repeatable (AND). Forms: field=value, "
|
||||
"field~substring, 'field>=value', 'field:*' (present), '!field' (absent).",
|
||||
),
|
||||
kind: str = typer.Option(None, "--kind", help="Shorthand for --field kind=<value>."),
|
||||
subtype: str = typer.Option(None, "--subtype", help="Shorthand for --field subtype=<value>."),
|
||||
collection: str = typer.Option(
|
||||
None, "--collection", help="Shorthand for --field collection=<value>."
|
||||
),
|
||||
tag: str = typer.Option(None, "--tag", help="Shorthand for --field tags=<value>."),
|
||||
regex: bool = typer.Option(
|
||||
False, "--regex", help="Treat the query as a regex. Off by default: terms are literal."
|
||||
),
|
||||
limit: int = typer.Option(20, "--limit", help="Maximum number of results. 0 for no limit."),
|
||||
sort: str = typer.Option(
|
||||
None, "--sort", help="Sort by a result field; prefix with '-' to reverse, e.g. -confidence."
|
||||
),
|
||||
backend: str = typer.Option(
|
||||
None,
|
||||
"--backend",
|
||||
help="Search backend(s), comma-separated. Default 'rg' (or $WIKITOOL_SEARCH_BACKEND).",
|
||||
),
|
||||
show_matches: bool = typer.Option(
|
||||
False, "--matches", help="Print the matching lines under each result."
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the results as JSON."),
|
||||
):
|
||||
"""Search kb/ by text, by frontmatter, or by both."""
|
||||
raw_predicates = list(field or [])
|
||||
for value, name in ((kind, "kind"), (subtype, "subtype"), (collection, "collection")):
|
||||
if value:
|
||||
raw_predicates.append(f"{name}={value}")
|
||||
if tag:
|
||||
raw_predicates.append(f"tags={tag}")
|
||||
|
||||
if not text and not raw_predicates:
|
||||
fail("Nothing to search for: give a query, or at least one --field predicate.")
|
||||
|
||||
try:
|
||||
predicates: tuple[Predicate, ...] = tuple(
|
||||
filters.parse_predicate(raw) for raw in raw_predicates
|
||||
)
|
||||
except PredicateError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
try:
|
||||
backends = resolve(backend)
|
||||
except UnknownBackend as exc:
|
||||
fail(str(exc))
|
||||
|
||||
query = SearchQuery(
|
||||
text=text,
|
||||
predicates=predicates,
|
||||
regex=regex,
|
||||
limit=limit,
|
||||
sort=sort,
|
||||
)
|
||||
|
||||
pages = load_pages_by_path()
|
||||
try:
|
||||
hits = run_search(query, pages, backends)
|
||||
except PredicateError as exc:
|
||||
fail(str(exc))
|
||||
except RipgrepMissing as exc:
|
||||
fail(str(exc))
|
||||
except RipgrepFailed as exc:
|
||||
fail(str(exc))
|
||||
|
||||
if json_out:
|
||||
payload = {
|
||||
"generated": today_iso(),
|
||||
"query": text,
|
||||
"predicates": [p.render() for p in predicates],
|
||||
"backend": ",".join(b.name for b in backends),
|
||||
"count": len(hits),
|
||||
"results": [hit.as_dict() for hit in hits],
|
||||
}
|
||||
typer.echo(json.dumps(payload, indent=2))
|
||||
return
|
||||
|
||||
typer.echo(render_table(hits, show_matches))
|
||||
@@ -0,0 +1,319 @@
|
||||
"""`wikitool touch` - update the self-describing frontmatter fields of a page.
|
||||
|
||||
`modified:`, `summary:`, `provenance:` and `confidence_base:` describe the page
|
||||
itself rather than its relationships, so they were the one part of frontmatter
|
||||
the skills still told the LLM to edit by hand - a carve-out in the otherwise
|
||||
absolute "never hand-write frontmatter" rule. Bumping a date and rewriting a
|
||||
one-line summary are mechanical, so they belong here: the field name is chosen
|
||||
from the type's own schema (`modified` for entity/concept, `date` for source),
|
||||
and the result is schema-validated before it is written.
|
||||
|
||||
Only `modified:` is bumped automatically. A source's `date:` is the publication
|
||||
date of the material itself, not a record of when we last edited the page, so it
|
||||
changes only on an explicit `--date`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
import typer
|
||||
|
||||
# Hard, non-optional dependency - see type_resolver.py's import comment.
|
||||
from jsonschema import Draft202012Validator, FormatChecker
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import (
|
||||
check_raw_files_exist,
|
||||
fail,
|
||||
parse_set_fields,
|
||||
rel_path,
|
||||
success,
|
||||
)
|
||||
from chemenu.frontmatter_io import normalize_dates
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
# Ordered by preference: whichever the page's schema declares is the one that
|
||||
# records "when was this page's content last confirmed?".
|
||||
DATE_FIELDS = ("modified", "date")
|
||||
|
||||
# Fields `--set` refuses, each with the command that owns it instead. This is a
|
||||
# denylist rather than an allowlist on purpose: an allowlist is a second copy of
|
||||
# the schema, and the copy is the one that drifts - a field added to a type-spec
|
||||
# would silently stay unwritable until someone remembered to widen the list.
|
||||
# Everything the schema declares is settable unless there is a reason here.
|
||||
UNSETTABLE = {
|
||||
"type": (
|
||||
"changing it changes the page's schema *and* the directory it belongs in - "
|
||||
"see instructions/page-lifecycle.md"
|
||||
),
|
||||
"confidence": (
|
||||
"derived, not authored: set `--confidence-base` and run "
|
||||
"`wikitool confidence decay --apply` to recompute it"
|
||||
),
|
||||
"related": "page-reference field - use `wikitool xref add` / `xref remove`",
|
||||
"sources": (
|
||||
"page-reference field - written from the other side by "
|
||||
"`wikitool xref link-source`, or cleared with `xref remove`"
|
||||
),
|
||||
"entities": (
|
||||
"page-reference field - use `wikitool xref link-source --source <this page> "
|
||||
"--entities <titles>`, which writes both directions; `xref remove` clears one"
|
||||
),
|
||||
"concepts": (
|
||||
"page-reference field - use `wikitool xref link-source --source <this page> "
|
||||
"--entities <titles>`, which writes both directions; `xref remove` clears one"
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _date_field(schema: Optional[Dict[str, Any]], frontmatter: Dict[str, Any]) -> Optional[str]:
|
||||
properties = (schema or {}).get("properties", {})
|
||||
for field in DATE_FIELDS:
|
||||
if field in properties or field in frontmatter:
|
||||
return field
|
||||
return None
|
||||
|
||||
|
||||
def _parse_date(text: str) -> datetime.date:
|
||||
"""`--date` as a real date, or a refusal naming the expected shape."""
|
||||
try:
|
||||
return datetime.date.fromisoformat(text)
|
||||
except ValueError:
|
||||
fail(f"--date must be YYYY-MM-DD, got '{text}'.")
|
||||
|
||||
|
||||
def validate_fields(
|
||||
frontmatter: Dict[str, Any], schema: Optional[Dict[str, Any]], fields: set[str]
|
||||
) -> Optional[str]:
|
||||
"""Validate only the fields this command is writing.
|
||||
|
||||
Whole-document validation would refuse to bump `modified:` on a page that
|
||||
is invalid for some unrelated, pre-existing reason - which is exactly the
|
||||
page most in need of maintenance. Errors whose path points outside the
|
||||
touched fields (missing required fields elsewhere, legacy extra keys) are
|
||||
left for `wikitool lint` to report.
|
||||
"""
|
||||
if schema is None:
|
||||
return None
|
||||
validator = Draft202012Validator(schema, format_checker=FormatChecker())
|
||||
messages = [
|
||||
error.message
|
||||
for error in validator.iter_errors(normalize_dates(frontmatter))
|
||||
if error.path and error.path[0] in fields
|
||||
]
|
||||
return "; ".join(messages) if messages else None
|
||||
|
||||
|
||||
def _settable_or_fail(field: str, schema: Optional[Dict[str, Any]], type_path: str) -> Dict[str, Any]:
|
||||
"""Refuse a field this command must not write, and return its subschema.
|
||||
|
||||
Two refusals, deliberately worded differently. A field on `UNSETTABLE` is
|
||||
writable in principle but belongs to another command, so the message names
|
||||
that command. A field the schema does not declare is not a routing problem
|
||||
but a typo or a wrong page type, so the message lists what this page
|
||||
actually has - the value is knowing that `tag` should have been `tags`.
|
||||
"""
|
||||
if field in UNSETTABLE:
|
||||
fail(f"`{field}` cannot be set with --set: {UNSETTABLE[field]}")
|
||||
properties = (schema or {}).get("properties", {})
|
||||
if field not in properties:
|
||||
settable = sorted(set(properties) - set(UNSETTABLE))
|
||||
fail(
|
||||
f"Type {type_path} declares no field '{field}'.\n"
|
||||
f" Settable fields for this page: {', '.join(settable) or '(none)'}"
|
||||
)
|
||||
return properties[field]
|
||||
|
||||
|
||||
def _apply_set(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
|
||||
if frontmatter.get(field) == value:
|
||||
return None
|
||||
before = frontmatter.get(field)
|
||||
frontmatter[field] = value
|
||||
return f"{field}: {before!r} -> {value!r}"
|
||||
|
||||
|
||||
def _apply_add(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
|
||||
"""Append list elements not already present, preserving order."""
|
||||
if not isinstance(value, list):
|
||||
fail(f"--add works on array fields only; '{field}' is not one. Use --set.")
|
||||
current = list(frontmatter.get(field) or [])
|
||||
added = [item for item in value if item not in current]
|
||||
if not added:
|
||||
return None
|
||||
frontmatter[field] = current + added
|
||||
return f"{field}: added {', '.join(repr(i) for i in added)}"
|
||||
|
||||
|
||||
def _apply_remove(frontmatter: Dict[str, Any], field: str, value: Any) -> Optional[str]:
|
||||
"""Drop list elements, reporting the ones that were not there.
|
||||
|
||||
Removing something absent succeeds rather than failing - `xref remove` is
|
||||
idempotent for the same reason, and a repair command that refuses to run
|
||||
twice is a repair command nobody dares script. But it is *reported*: a
|
||||
silent no-op is how a mistyped element name looks exactly like a successful
|
||||
removal.
|
||||
"""
|
||||
if not isinstance(value, list):
|
||||
fail(f"--remove works on array fields only; '{field}' is not one. Use --set.")
|
||||
current = list(frontmatter.get(field) or [])
|
||||
present = [item for item in value if item in current]
|
||||
absent = [item for item in value if item not in current]
|
||||
if absent:
|
||||
typer.echo(f" {field}: not present, nothing removed: {', '.join(repr(i) for i in absent)}")
|
||||
if not present:
|
||||
return None
|
||||
frontmatter[field] = [item for item in current if item not in present]
|
||||
return f"{field}: removed {', '.join(repr(i) for i in present)}"
|
||||
|
||||
|
||||
def touch_command(
|
||||
page_title: str = typer.Option(..., "--page", help="Exact page title, e.g. 'Docker Cheatsheet'"),
|
||||
summary: Optional[str] = typer.Option(None, "--summary", help="Replace the page's 1-line summary"),
|
||||
provenance: Optional[str] = typer.Option(
|
||||
None, "--provenance", help="Replace the page's provenance marker (sourced|general|mixed)"
|
||||
),
|
||||
confidence_base: Optional[float] = typer.Option(
|
||||
None,
|
||||
"--confidence-base",
|
||||
help="Re-assess the page's undecayed confidence (0.0-1.0). `confidence` itself is derived - "
|
||||
"run `wikitool confidence decay --apply` afterwards to recompute it.",
|
||||
),
|
||||
date: Optional[str] = typer.Option(
|
||||
None, "--date",
|
||||
help="Date to record (YYYY-MM-DD). `modified:` defaults to today; a source's "
|
||||
"`date:` is its publication date and changes only when given here.",
|
||||
),
|
||||
set_fields: Optional[list[str]] = typer.Option(
|
||||
None,
|
||||
"--set",
|
||||
help="Replace a frontmatter field, repeatable: --set tags=a,b. Array values split on "
|
||||
"commas (escape a literal one as \\,); repeating --set for one array field appends "
|
||||
"within this call. Page-reference fields belong to `xref`, not here",
|
||||
),
|
||||
add_fields: Optional[list[str]] = typer.Option(
|
||||
None,
|
||||
"--add",
|
||||
help="Append elements to an array field without naming the whole list: --add tags=x. "
|
||||
"Elements already present are left alone",
|
||||
),
|
||||
remove_fields: Optional[list[str]] = typer.Option(
|
||||
None,
|
||||
"--remove",
|
||||
help="Drop elements from an array field: --remove tags=x. Removing an absent element "
|
||||
"succeeds and says so",
|
||||
),
|
||||
no_date: bool = typer.Option(
|
||||
False, "--no-date", help="Only change the given fields; leave the modified/date field alone"
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview the new frontmatter instead of writing"),
|
||||
):
|
||||
"""Bump a page's `modified:` date and optionally rewrite its other frontmatter fields.
|
||||
|
||||
`--summary`/`--provenance`/`--confidence-base` are shorthands for the three
|
||||
fields worth their own flag; `--set`/`--add`/`--remove` reach every other
|
||||
field the page's type declares. Before they existed, a field `new` wrote
|
||||
once - `tags:`, `raw_files:` - could never be corrected: `touch` did not
|
||||
know it, hand-editing frontmatter is what the tool exists to prevent, and
|
||||
deleting the page to recreate it breaks every reference already pointing at
|
||||
it. A mistyped `--set tags=` at creation was therefore permanent, and `new`
|
||||
is not idempotent, so the window to get it right was exactly one command.
|
||||
"""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
page = pages.get(page_title)
|
||||
if page is None:
|
||||
fail(f"No page titled '{page_title}' found under wiki/. Create it first with `wikitool new ...`.")
|
||||
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
fail(f"Page '{page_title}' has no `type:` frontmatter - fix it before touching it.")
|
||||
|
||||
try:
|
||||
schema = resolver.get_schema(type_path, page.path)
|
||||
except ValueError as exc:
|
||||
fail(str(exc))
|
||||
|
||||
frontmatter = dict(page.frontmatter)
|
||||
changes: list[str] = []
|
||||
touched: set[str] = set()
|
||||
|
||||
if not no_date:
|
||||
field = _date_field(schema, frontmatter)
|
||||
if field is None:
|
||||
fail(f"Type {type_path} declares no modified/date field - pass --no-date to skip it.")
|
||||
# `modified:` is ours to bump - it records when we last touched the page.
|
||||
# `date:` is not: on a source it is the source material's own publication
|
||||
# date, a fact about the world that today's date is simply wrong for.
|
||||
# Auto-bumping it silently replaced a raw file's real date with the day
|
||||
# the summary happened to be rewritten, and left the page contradicting
|
||||
# the `**Datum:**` line in its own body. It is still writable, but only
|
||||
# when the caller says so with an explicit `--date`.
|
||||
if field == "date" and date is None:
|
||||
touched.discard(field)
|
||||
else:
|
||||
# A `datetime.date`, not a string: that is what `yaml.safe_load`
|
||||
# yields for every page already on disk, and writing anything else
|
||||
# made an unchanged date compare unequal to itself - so `touch`
|
||||
# reported a change on every run, and `dump_frontmatter` had to
|
||||
# guess whether to quote what it was handed.
|
||||
new_date = _parse_date(date) if date else datetime.date.today()
|
||||
touched.add(field)
|
||||
if frontmatter.get(field) != new_date:
|
||||
frontmatter[field] = new_date
|
||||
changes.append(f"{field}: {page.frontmatter.get(field)} -> {new_date.isoformat()}")
|
||||
|
||||
if summary is not None:
|
||||
frontmatter["summary"] = summary
|
||||
touched.add("summary")
|
||||
changes.append("summary updated")
|
||||
if provenance is not None:
|
||||
frontmatter["provenance"] = provenance
|
||||
touched.add("provenance")
|
||||
changes.append(f"provenance: {page.frontmatter.get('provenance')} -> {provenance}")
|
||||
if confidence_base is not None:
|
||||
frontmatter["confidence_base"] = round(confidence_base, 2)
|
||||
touched.add("confidence_base")
|
||||
changes.append(
|
||||
f"confidence_base: {page.frontmatter.get('confidence_base')} -> {round(confidence_base, 2)}"
|
||||
)
|
||||
|
||||
# --set/--add/--remove last, so an explicit field always wins over the
|
||||
# shorthand flags rather than depending on option order.
|
||||
for flag, values, apply in (
|
||||
("--set", set_fields, _apply_set),
|
||||
("--add", add_fields, _apply_add),
|
||||
("--remove", remove_fields, _apply_remove),
|
||||
):
|
||||
parsed = parse_set_fields(values, schema, flag=flag)
|
||||
for field, value in parsed.items():
|
||||
_settable_or_fail(field, schema, type_path)
|
||||
change = apply(frontmatter, field, value)
|
||||
touched.add(field)
|
||||
if change:
|
||||
changes.append(change)
|
||||
|
||||
# Filesystem check, not a data-shape one, so the schema cannot carry it -
|
||||
# and `touch` writes this field now, so it owes the same check `new` does.
|
||||
if "raw_files" in touched:
|
||||
check_raw_files_exist(frontmatter.get("raw_files"))
|
||||
|
||||
error = validate_fields(frontmatter, schema, touched)
|
||||
if error:
|
||||
fail(f"Invalid value for type {type_path}: {error}")
|
||||
|
||||
if not changes:
|
||||
success(f"'{page_title}' already up to date; nothing to change.")
|
||||
return
|
||||
|
||||
for change in changes:
|
||||
typer.echo(f" {change}")
|
||||
|
||||
if dry_run:
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
write_page(page.path, frontmatter, page.body)
|
||||
success(f"Touched {rel_path(page.path)}")
|
||||
@@ -0,0 +1,122 @@
|
||||
"""`wikitool types ...` - discover and describe the wiki's type-spec contracts.
|
||||
|
||||
Lets an LLM (or human) find out what page types exist and what a given type
|
||||
requires by asking wikitool, instead of reading raw type-spec markdown files
|
||||
into context on every skill invocation. A type-spec's own frontmatter
|
||||
(`name`, `description`, `schema`, `subtype_field`, `base_dir`) and its
|
||||
declared `.schema.yaml` are the single source of truth; this command only
|
||||
formats what `TypeResolver` already resolves - it does not duplicate or
|
||||
re-derive any type knowledge.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Dict
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu.commands._util import fail
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
app = typer.Typer(help="Discover and describe Chemenu type-spec contracts.")
|
||||
|
||||
|
||||
@app.command("list")
|
||||
def list_types(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
|
||||
"""List every type-spec under types/, with its name, schema, subtype
|
||||
field (if any), base directory, and description."""
|
||||
rows: list[Dict[str, Any]] = []
|
||||
for type_path, frontmatter in resolver.list_type_specs():
|
||||
rows.append({
|
||||
"name": frontmatter.get("name"),
|
||||
"type_path": type_path,
|
||||
"schema": frontmatter.get("schema"),
|
||||
"subtype_field": frontmatter.get("subtype_field"),
|
||||
"root": frontmatter.get("root") or "kb",
|
||||
"base_dir": frontmatter.get("base_dir"),
|
||||
"description": frontmatter.get("description"),
|
||||
})
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(rows, indent=2))
|
||||
return
|
||||
|
||||
for row in rows:
|
||||
typer.echo(f"{row['name']} ({row['type_path']})")
|
||||
typer.echo(f" schema: {row['schema']}")
|
||||
if row["subtype_field"]:
|
||||
typer.echo(f" subtype_field: {row['subtype_field']}")
|
||||
if row["base_dir"]:
|
||||
typer.echo(f" base_dir: {row['root']}/{row['base_dir']}")
|
||||
typer.echo(f" {row['description']}")
|
||||
typer.echo("")
|
||||
|
||||
|
||||
@app.command("describe")
|
||||
def describe_type(
|
||||
name: str = typer.Argument(..., help="Type name, e.g. 'entity' (see `types list`)"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON"),
|
||||
):
|
||||
"""Print one type's full contract: frontmatter fields (required/optional,
|
||||
with enums where declared), its subtype field if any, and its authoring
|
||||
body - the same information an LLM would otherwise gather by reading the
|
||||
raw type-spec and `.schema.yaml` files directly."""
|
||||
type_path = resolver.find_type_by_name(name)
|
||||
if type_path is None:
|
||||
available = sorted(fm.get("name") for _, fm in resolver.list_type_specs())
|
||||
fail(f"No type-spec named '{name}'. Available: {', '.join(available)}")
|
||||
return # unreachable; keeps type-checkers happy about `type_path` below
|
||||
|
||||
type_spec = resolver.load_type_spec(type_path)
|
||||
frontmatter = type_spec["frontmatter"]
|
||||
body = type_spec["body"]
|
||||
schema = resolver.get_schema(type_path)
|
||||
|
||||
fields: list[Dict[str, Any]] = []
|
||||
if schema is not None:
|
||||
required = set(schema.get("required", []))
|
||||
for field_name, field_schema in schema.get("properties", {}).items():
|
||||
fields.append({
|
||||
"field": field_name,
|
||||
"required": field_name in required,
|
||||
"type": field_schema.get("type"),
|
||||
"enum": field_schema.get("enum"),
|
||||
})
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps({
|
||||
"name": frontmatter.get("name"),
|
||||
"type_path": type_path,
|
||||
"description": frontmatter.get("description"),
|
||||
"schema": frontmatter.get("schema"),
|
||||
"subtype_field": frontmatter.get("subtype_field"),
|
||||
"base_dir": frontmatter.get("base_dir"),
|
||||
"title_prefix": frontmatter.get("title_prefix"),
|
||||
"fields": fields,
|
||||
"body": body.strip(),
|
||||
}, indent=2))
|
||||
return
|
||||
|
||||
typer.echo(f"# {frontmatter.get('name')} ({type_path})")
|
||||
typer.echo(frontmatter.get("description", ""))
|
||||
typer.echo("")
|
||||
if frontmatter.get("subtype_field"):
|
||||
typer.echo(f"subtype_field: {frontmatter['subtype_field']}")
|
||||
if frontmatter.get("base_dir"):
|
||||
typer.echo(f"base_dir: {frontmatter.get('root') or 'kb'}/{frontmatter['base_dir']}")
|
||||
if frontmatter.get("title_prefix"):
|
||||
typer.echo(f"title_prefix: {frontmatter['title_prefix']!r}")
|
||||
typer.echo("")
|
||||
|
||||
if not fields:
|
||||
typer.echo("(no schema declared for this type)")
|
||||
else:
|
||||
typer.echo("## Frontmatter fields")
|
||||
for field in fields:
|
||||
marker = "required" if field["required"] else "optional"
|
||||
extra = f", enum: {field['enum']}" if field["enum"] else ""
|
||||
typer.echo(f"- `{field['field']}` ({marker}, {field['type']}{extra})")
|
||||
typer.echo("")
|
||||
|
||||
typer.echo("## Authoring guidance")
|
||||
typer.echo(body.strip())
|
||||
@@ -0,0 +1,276 @@
|
||||
"""`wikitool version` - report, bump, and check the stack's version.
|
||||
|
||||
Three jobs that all hang off one number (see `chemenu/version.py` for what
|
||||
that number means):
|
||||
|
||||
- `version show` answers "which stack is this instance running", offline, from
|
||||
`VERSION` plus the release stamp `dist export` writes.
|
||||
- `version bump` moves it, and writes the changelog *heading* that has to
|
||||
accompany the move - the same structure-by-tool/prose-by-author split as
|
||||
`new`. `docs verify` then holds the two together.
|
||||
- `version check` is the one command in `wikitool` that makes a network call.
|
||||
It is deliberately its own command: nothing else reaches for it implicitly,
|
||||
it needs no key, it times out, and a feed that cannot be reached is reported
|
||||
as an error rather than silently answered as "up to date".
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json as _json
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, version as version_mod
|
||||
from chemenu.commands._util import console, fail, rel_path, success, today_iso
|
||||
from chemenu.version import Version, VersionError
|
||||
|
||||
app = typer.Typer(
|
||||
help="Report, bump, and check the stack version (see tools/CONTRACT.md).",
|
||||
invoke_without_command=True,
|
||||
)
|
||||
|
||||
|
||||
@app.callback()
|
||||
def version_callback(ctx: typer.Context) -> None:
|
||||
"""Bare `wikitool version` is a convenience alias for `version show`."""
|
||||
if ctx.invoked_subcommand is None:
|
||||
show_command(json_out=False)
|
||||
|
||||
|
||||
def _describe_origin(stamp: Optional[dict]) -> str:
|
||||
if not stamp:
|
||||
return "development tree (no release stamp)"
|
||||
parts = []
|
||||
exported = stamp.get("exported_at")
|
||||
if exported:
|
||||
parts.append(f"exported {exported}")
|
||||
commit = str(stamp.get("source_commit") or "")
|
||||
if commit:
|
||||
parts.append(f"from commit {commit[:12]}")
|
||||
repo = stamp.get("source_repo")
|
||||
if repo:
|
||||
parts.append(str(repo))
|
||||
return "distribution: " + ", ".join(parts) if parts else "distribution"
|
||||
|
||||
|
||||
@app.command("show")
|
||||
def show_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the version and stamp as JSON"),
|
||||
):
|
||||
"""Print this instance's stack version and where it came from. Read-only,
|
||||
offline, and exempt from the Iteration Budget Gate."""
|
||||
try:
|
||||
current = version_mod.read_version()
|
||||
stamp = version_mod.read_stamp()
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
{
|
||||
"version": str(current),
|
||||
"compat_key": list(current.compat_key),
|
||||
"stamp": stamp,
|
||||
"update_url": version_mod.update_url(stamp),
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
console.print(f"[bold]{current}[/bold] ({_describe_origin(stamp)})")
|
||||
if stamp and stamp.get("release_url"):
|
||||
console.print(f"release: {stamp['release_url']}")
|
||||
|
||||
|
||||
@app.command("check")
|
||||
def check_command(
|
||||
url: Optional[str] = typer.Option(
|
||||
None, "--url", help="Release feed to ask (default: the stamp's, else the built-in origin)"
|
||||
),
|
||||
timeout: float = typer.Option(10.0, "--timeout", help="Seconds to wait for the feed"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print the result as JSON"),
|
||||
):
|
||||
"""Ask the origin's release feed whether a newer stack exists.
|
||||
|
||||
The only networked command in `wikitool`. Exits 1 if the feed cannot be
|
||||
reached or does not answer with a release - an unreachable feed is not the
|
||||
same answer as "up to date", and must never be reported as one."""
|
||||
import os
|
||||
|
||||
try:
|
||||
current = version_mod.read_version()
|
||||
stamp = version_mod.read_stamp()
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
feed = url or version_mod.update_url(stamp)
|
||||
token = os.environ.get(version_mod.UPDATE_TOKEN_ENV, "").strip() or None
|
||||
|
||||
try:
|
||||
latest, release_url, published = version_mod.fetch_latest_release(feed, token, timeout)
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
status = version_mod.UpdateStatus(
|
||||
local=current,
|
||||
latest=latest,
|
||||
state=version_mod.compare(current, latest),
|
||||
release_url=release_url,
|
||||
published_at=published,
|
||||
)
|
||||
|
||||
if json_out:
|
||||
typer.echo(
|
||||
_json.dumps(
|
||||
{
|
||||
"local": str(status.local),
|
||||
"latest": str(status.latest),
|
||||
"state": status.state,
|
||||
"requires_migration": status.state == "migration",
|
||||
"release_url": status.release_url,
|
||||
"published_at": status.published_at,
|
||||
"feed": feed,
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
color = {"current": "green", "ahead": "yellow", "update": "cyan", "migration": "bold yellow"}
|
||||
console.print(f"[{color[status.state]}]{status.headline}[/{color[status.state]}]")
|
||||
if status.release_url:
|
||||
console.print(f"release: {status.release_url}")
|
||||
if status.state in ("update", "migration"):
|
||||
console.print(
|
||||
"Applying it is a separate, manual step - see INSTALL.md "
|
||||
"§ 'Eine Instanz aktualisieren'."
|
||||
)
|
||||
|
||||
|
||||
@app.command("notes")
|
||||
def notes_command(
|
||||
version: Optional[str] = typer.Option(
|
||||
None, "--version", help="Which entry to print (default: this tree's VERSION)"
|
||||
),
|
||||
):
|
||||
"""Print one version's `CHANGES.md` entry, for use as release notes.
|
||||
|
||||
Mechanical extraction, so the release workflow never has to parse markdown
|
||||
in shell."""
|
||||
try:
|
||||
wanted = Version.parse(version) if version else version_mod.read_version()
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
changes = version_mod.changes_file()
|
||||
if not changes.is_file():
|
||||
fail(f"{version_mod.CHANGES_FILENAME} is missing - there are no release notes to print")
|
||||
return
|
||||
|
||||
section = version_mod.changes_section(changes.read_text(encoding="utf-8"), wanted)
|
||||
if section is None:
|
||||
fail(
|
||||
f"{version_mod.CHANGES_FILENAME} has no entry for {wanted} - "
|
||||
f"run `wikitool version bump` before releasing, or write the entry"
|
||||
)
|
||||
return
|
||||
typer.echo(section, nl=False)
|
||||
|
||||
|
||||
@app.command("bump")
|
||||
def bump_command(
|
||||
major: bool = typer.Option(False, "--major", help="Bump MAJOR (resets MINOR and PATCH)"),
|
||||
minor: bool = typer.Option(False, "--minor", help="Bump MINOR (resets PATCH)"),
|
||||
patch: bool = typer.Option(False, "--patch", help="Bump PATCH"),
|
||||
title: str = typer.Option(..., "--title", help="One-line title for the new CHANGES.md entry"),
|
||||
no_migration: Optional[str] = typer.Option(
|
||||
None,
|
||||
"--no-migration",
|
||||
help="Why this boundary-crossing bump needs no content migration (recorded in CHANGES.md)",
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Report the change without writing"),
|
||||
):
|
||||
"""Raise the stack version and open its `CHANGES.md` entry.
|
||||
|
||||
Writes `VERSION` and inserts the entry's heading, date and author - the
|
||||
entry's body stays the author's to write, the same way `new` produces
|
||||
frontmatter and leaves the prose. `docs verify` afterwards enforces that
|
||||
the two agree, so a bump with no entry cannot reach a release.
|
||||
|
||||
A bump that crosses the compatibility boundary additionally requires a
|
||||
migration document for the new version, or `--no-migration "<reason>"`.
|
||||
An instance learning that it must migrate, with nothing telling it how, is
|
||||
the gap this closes."""
|
||||
selected = [name for name, chosen in (("major", major), ("minor", minor), ("patch", patch)) if chosen]
|
||||
if len(selected) != 1:
|
||||
fail("Pass exactly one of --major / --minor / --patch")
|
||||
return
|
||||
if not title.strip():
|
||||
fail("--title must not be empty - it becomes the changelog entry's heading")
|
||||
return
|
||||
|
||||
try:
|
||||
current = version_mod.read_version()
|
||||
new_version = current.bumped(selected[0])
|
||||
except VersionError as exc:
|
||||
fail(str(exc))
|
||||
return
|
||||
|
||||
changes = version_mod.changes_file()
|
||||
if not changes.is_file():
|
||||
fail(f"{version_mod.CHANGES_FILENAME} is missing - a bump has nowhere to record itself")
|
||||
return
|
||||
text = changes.read_text(encoding="utf-8")
|
||||
existing = version_mod.top_changes_version(text)
|
||||
if existing is not None and existing >= new_version:
|
||||
fail(
|
||||
f"{version_mod.CHANGES_FILENAME} already documents {existing}, which is not older "
|
||||
f"than {new_version} - bump past it, or fix the changelog"
|
||||
)
|
||||
return
|
||||
|
||||
author = config.default_author() or "unknown"
|
||||
crossing = new_version.compat_key != current.compat_key
|
||||
boundary = " (crosses a compatibility boundary - instances must migrate)" if crossing else ""
|
||||
|
||||
if crossing and not no_migration:
|
||||
from chemenu import kb_state
|
||||
|
||||
if not any(m.target == new_version for m in kb_state.load_migrations()):
|
||||
fail(
|
||||
f"{current} -> {new_version} crosses the compatibility boundary, so every existing "
|
||||
f"instance must migrate - but no migration document targets {new_version}.\n"
|
||||
f"Write one under {rel_path(kb_state.migrations_dir())}/{new_version}-<slug>.md "
|
||||
f"(see instructions/migrate-corpus.md), or, if no content actually has to change, "
|
||||
f're-run with --no-migration "<reason>".'
|
||||
)
|
||||
return
|
||||
if no_migration and not crossing:
|
||||
fail(
|
||||
f"--no-migration only applies to a bump that crosses the compatibility boundary; "
|
||||
f"{current} -> {new_version} does not."
|
||||
)
|
||||
return
|
||||
|
||||
if dry_run:
|
||||
success(f"Dry run: {current} -> {new_version}{boundary}. Nothing written.")
|
||||
return
|
||||
|
||||
version_mod.write_version(new_version)
|
||||
changes.write_text(
|
||||
version_mod.insert_changes_entry(
|
||||
text, new_version, today_iso(), title.strip(), author,
|
||||
no_migration_reason=no_migration.strip() if no_migration else None,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
success(
|
||||
f"{current} -> {new_version}{boundary}. Wrote {version_mod.VERSION_FILENAME} and opened "
|
||||
f"the {version_mod.CHANGES_FILENAME} entry - write its body before publishing."
|
||||
)
|
||||
@@ -0,0 +1,244 @@
|
||||
"""`wikitool work` - scaffold and inspect workshop runs under `work/`.
|
||||
|
||||
A workshop is the tracked, transient scratch directory for a task that does not
|
||||
fit in one session (see work/CONTRACT.md). The only mechanical part of it is the
|
||||
run key: it is derived from the input path, it is the directory name, and a
|
||||
collision means the same tree is already being ingested. Doing that by hand is
|
||||
how a second identifier and a `-2` suffix creep in, so it lives here instead.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import shutil
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, rel_path, success, today_iso
|
||||
|
||||
app = typer.Typer(help="Workshop runs under work/ (see work/CONTRACT.md).")
|
||||
|
||||
RUN_KEY_PREFIX = "ingest-"
|
||||
|
||||
# Everything outside this set is folded to a single hyphen, so a run key is
|
||||
# always a safe directory name and always reproducible from the same input.
|
||||
_UNSAFE_RE = re.compile(r"[^a-z0-9]+")
|
||||
|
||||
REQUIRED_FILES = ("README.md", "plan.md")
|
||||
|
||||
|
||||
def derive_run_key(input_path: str) -> str:
|
||||
"""Run key for an input path under `raw/`.
|
||||
|
||||
Derived from the path *below* `raw/` with separators flattened, never from
|
||||
the basename: `raw/documents/handbook` and `raw/articles/handbook` share a
|
||||
basename but are different sources.
|
||||
"""
|
||||
relative = input_path.strip().strip("/")
|
||||
if relative == "raw":
|
||||
return ""
|
||||
if relative.startswith("raw/"):
|
||||
relative = relative[len("raw/"):]
|
||||
slug = _UNSAFE_RE.sub("-", relative.lower()).strip("-")
|
||||
if not slug:
|
||||
return ""
|
||||
return f"{RUN_KEY_PREFIX}{slug}"
|
||||
|
||||
|
||||
def normalize_run_key(key: str) -> str:
|
||||
"""Run key for a run with no raw input, given explicitly by the caller.
|
||||
|
||||
Not every multi-session task is an ingest. A migration or a sweep across
|
||||
`kb/` has no input tree to derive a key from, and the alternative - opening
|
||||
no workshop at all - costs the run its `plan.md`, which is what makes taking
|
||||
a new session id per unit legitimate rather than a way around a refusal
|
||||
(instructions/gates.md).
|
||||
|
||||
The `ingest-` prefix stays reserved for derived keys, so a directory name
|
||||
always says which kind of run made it.
|
||||
"""
|
||||
slug = _UNSAFE_RE.sub("-", key.strip().lower()).strip("-")
|
||||
if not slug:
|
||||
return ""
|
||||
if slug.startswith(RUN_KEY_PREFIX):
|
||||
return ""
|
||||
return slug
|
||||
|
||||
|
||||
def readme_template(run_key: str, input_path: Optional[str]) -> str:
|
||||
input_line = f"`{input_path}`" if input_path else "none - this run is not an ingest"
|
||||
return f"""# Workshop: {run_key}
|
||||
|
||||
- **Run key:** `{run_key}` (this directory's name - there is no other identifier)
|
||||
- **Input:** {input_line}
|
||||
- **Started:** {today_iso()}
|
||||
- **Session id form:** `WIKITOOL_SESSION_ID="{run_key}/u<N>"`, one per unit
|
||||
|
||||
## Goal
|
||||
|
||||
TODO: what this run must produce.
|
||||
|
||||
## Closes when
|
||||
|
||||
TODO: the condition that ends the run - normally "every unit in plan.md is published and
|
||||
its `## Not Extracted` section is filled".
|
||||
|
||||
## Checklist
|
||||
|
||||
TODO: one line per unit from plan.md, e.g.
|
||||
|
||||
- [ ] u1 <unit> - extract / promote / publish
|
||||
|
||||
## Open decisions
|
||||
|
||||
- None yet. Record blockers as `DECISION NEEDED: <question>` and stop at them.
|
||||
"""
|
||||
|
||||
|
||||
def plan_template(run_key: str, input_path: Optional[str]) -> str:
|
||||
if input_path is None:
|
||||
return f"""# Plan: {run_key}
|
||||
|
||||
Cut the work into units. One unit is one session id and one `publish`, so it has to fit inside
|
||||
the 60-call iteration budget on its own - count the `wikitool` calls the unit needs before
|
||||
committing to its size.
|
||||
|
||||
| # | Unit | Job | Done when |
|
||||
|---|------|-----|-----------|
|
||||
| u1 | TODO | TODO | TODO |
|
||||
|
||||
## Deliberately excluded from this run
|
||||
|
||||
- TODO: what this run is not touching, and why.
|
||||
"""
|
||||
return f"""# Plan: {run_key}
|
||||
|
||||
Input tree: `{input_path}`
|
||||
|
||||
Cut the tree into units. One unit does one job and becomes one source page. A unit whose
|
||||
`raw_files` list would pass roughly 15 entries is still too coarse.
|
||||
|
||||
| # | Unit (input subtree) | Job | Planned source page | Why this cut |
|
||||
|---|----------------------|-----|---------------------|--------------|
|
||||
| u1 | TODO | TODO | `Source - TODO` | TODO |
|
||||
|
||||
## Deliberately excluded from this run
|
||||
|
||||
- TODO: parts of the tree that are not being ingested at all, and why.
|
||||
"""
|
||||
|
||||
|
||||
@app.command("new")
|
||||
def new_command(
|
||||
input_path: Optional[str] = typer.Option(
|
||||
None,
|
||||
"--input",
|
||||
help="Path to the raw source tree or file this run covers, e.g. raw/documents/handbook",
|
||||
),
|
||||
key: Optional[str] = typer.Option(
|
||||
None,
|
||||
"--key",
|
||||
help="Explicit run key for a run with no raw input (a migration, a sweep across kb/). "
|
||||
"Mutually exclusive with --input; may not start with 'ingest-'.",
|
||||
),
|
||||
again: bool = typer.Option(
|
||||
False,
|
||||
"--again",
|
||||
help="This is a deliberate re-ingest of a tree already processed before: append today's "
|
||||
"date to the run key instead of refusing the collision.",
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be created, write nothing"),
|
||||
):
|
||||
"""Scaffold `work/<runkey>/` for one workshop run."""
|
||||
if (input_path is None) == (key is None):
|
||||
fail(
|
||||
"Pass exactly one of --input (an ingest of raw material, key derived from the path) "
|
||||
"or --key (a run with no raw input, key given explicitly)."
|
||||
)
|
||||
|
||||
if key is not None:
|
||||
run_key = normalize_run_key(key)
|
||||
if not run_key:
|
||||
fail(
|
||||
f"`{key}` is not a usable run key - it must contain letters or digits and must not "
|
||||
f"start with `{RUN_KEY_PREFIX}`, which is reserved for keys derived from a raw path."
|
||||
)
|
||||
else:
|
||||
source = (config.ROOT / input_path).resolve()
|
||||
try:
|
||||
source.relative_to(config.RAW_DIR)
|
||||
except ValueError:
|
||||
fail(
|
||||
f"`{input_path}` is not under raw/. A run key is derived from the input path below "
|
||||
"raw/, so an --input workshop can only be opened for raw material. A run with no raw "
|
||||
"input takes --key instead."
|
||||
)
|
||||
if not source.exists():
|
||||
fail(f"`{input_path}` does not exist - a workshop is opened for material that is already in raw/")
|
||||
|
||||
run_key = derive_run_key(input_path)
|
||||
if not run_key:
|
||||
fail(f"`{input_path}` yields an empty run key - point --input at a subtree of raw/, not at raw/ itself")
|
||||
if again:
|
||||
run_key = f"{run_key}-{today_iso()}"
|
||||
|
||||
target = config.WORK_DIR / run_key
|
||||
if target.exists():
|
||||
if again:
|
||||
fail(
|
||||
f"`{rel_path(target)}` already exists - a second re-ingest of the same tree on the "
|
||||
"same day. Resume that run or close it first."
|
||||
)
|
||||
fail(
|
||||
f"`{rel_path(target)}` already exists, which means this tree is already being ingested. "
|
||||
"Resume that run, or - if the tree itself has changed since - re-run with --again to "
|
||||
"open a dated second pass. Never work around this with a numbered suffix."
|
||||
)
|
||||
|
||||
if dry_run:
|
||||
success(f"Would create {rel_path(target)}/ with {', '.join(REQUIRED_FILES)}")
|
||||
return
|
||||
|
||||
target.mkdir(parents=True)
|
||||
(target / "README.md").write_text(readme_template(run_key, input_path), encoding="utf-8")
|
||||
(target / "plan.md").write_text(plan_template(run_key, input_path), encoding="utf-8")
|
||||
|
||||
typer.echo(f"Run key: {run_key}")
|
||||
typer.echo(f"Workshop: {rel_path(target)}/")
|
||||
typer.echo(f"Next: fill in plan.md, then export WIKITOOL_SESSION_ID=\"{run_key}/u1\"")
|
||||
success(f"Created workshop {run_key}")
|
||||
|
||||
|
||||
@app.command("close")
|
||||
def close_command(
|
||||
run_key: str = typer.Option(..., "--run-key", help="The workshop directory name"),
|
||||
yes: bool = typer.Option(
|
||||
False,
|
||||
"--yes",
|
||||
"-y",
|
||||
help="Confirm deletion. Required: closing discards the only copy of the run's working "
|
||||
"notes, so the durable conclusions must already be in kb/.",
|
||||
),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be deleted, delete nothing"),
|
||||
):
|
||||
"""Delete a finished workshop. Its conclusions must already be in `kb/`."""
|
||||
target = config.WORK_DIR / run_key
|
||||
if not target.is_dir():
|
||||
fail(f"No workshop `{run_key}` under work/ - `ls work/` shows the open ones")
|
||||
|
||||
files = sorted(p for p in target.rglob("*") if p.is_file())
|
||||
if dry_run or not yes:
|
||||
listing = "\n".join(f"- {rel_path(p)}" for p in files)
|
||||
message = (
|
||||
f"Closing `{run_key}` deletes {len(files)} file(s):\n{listing}\n"
|
||||
"Nothing here is recoverable from the rest of the repo. Confirm the durable "
|
||||
"conclusions are already in kb/, then re-run with --yes."
|
||||
)
|
||||
if dry_run:
|
||||
typer.echo(message)
|
||||
return
|
||||
fail(message)
|
||||
|
||||
shutil.rmtree(target)
|
||||
success(f"Closed workshop {run_key} ({len(files)} file(s) deleted). Log it with `log append`.")
|
||||
@@ -0,0 +1,339 @@
|
||||
"""Bidirectional cross-reference management between wiki pages.
|
||||
|
||||
`xref add` keeps two pages' frontmatter `related:` lists AND their body
|
||||
"## Relationships" sections in sync in one operation, instead of the 3-5
|
||||
separate manual edits this used to take per pair of pages. It is idempotent:
|
||||
re-running it never duplicates a link.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config, sections
|
||||
from chemenu.commands._util import fail, parse_list, success
|
||||
from chemenu.commands.page_ops import strip_frontmatter_ref
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.page import Page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
|
||||
app = typer.Typer(help="Manage bidirectional cross-references between wiki pages.")
|
||||
|
||||
|
||||
def _find_page(pages: dict[str, Page], name: str) -> Page:
|
||||
if name not in pages:
|
||||
fail(f"No page titled '{name}' found under wiki/. Create it first with `wikitool new ...`.")
|
||||
return pages[name]
|
||||
|
||||
|
||||
def _declared_ref_fields(page: Page) -> list[str]:
|
||||
from chemenu.commands.page_ops import page_ref_fields
|
||||
|
||||
return page_ref_fields(page)
|
||||
|
||||
|
||||
def _require_related_field(page: Page, title: str) -> None:
|
||||
"""Refuse to write `related:` on a type that does not declare it.
|
||||
|
||||
`xref add` used to write the field unconditionally. On a source page -
|
||||
whose type declares `page_ref_fields: [entities, concepts]` - that produced
|
||||
frontmatter the schema rejects (`additionalProperties: false`), which
|
||||
`xref remove` then could not clear, because it only swept declared fields.
|
||||
One command created a state another could not undo.
|
||||
"""
|
||||
declared = _declared_ref_fields(page)
|
||||
if "related" in declared:
|
||||
return
|
||||
fail(
|
||||
f"'{title}' is a {page.frontmatter.get('type')} page, whose type does not declare a "
|
||||
f"`related:` field, so `xref add` has nothing to write there.\n"
|
||||
f" Reference fields this type declares: {', '.join(declared) or '(none)'}\n"
|
||||
f" For a source page, `wikitool xref link-source --source \"{title}\" "
|
||||
f"--entities <titles>` is the command that fills them."
|
||||
)
|
||||
|
||||
|
||||
def _back_reference_field(source: Page, target: Page) -> str | None:
|
||||
"""Which of `source`'s declared ref fields `target` belongs in.
|
||||
|
||||
Derived from the target's collection rather than a hardcoded type-to-field
|
||||
map: a page under `kb/entities/` belongs in `entities:`, one under
|
||||
`kb/concepts/` in `concepts:`. The collection directory *is* the field
|
||||
name, so a new collection needs no code change here - it needs a type that
|
||||
declares the matching field.
|
||||
"""
|
||||
try:
|
||||
collection = target.path.relative_to(config.KB_DIR).parts[0]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
return collection if collection in _declared_ref_fields(source) else None
|
||||
|
||||
|
||||
def add_related(frontmatter: dict, other_title: str) -> bool:
|
||||
"""Add other_title to frontmatter['related'] if not already present.
|
||||
Returns True if a change was made."""
|
||||
related = frontmatter.setdefault("related", [])
|
||||
if other_title in related:
|
||||
return False
|
||||
related.append(other_title)
|
||||
return True
|
||||
|
||||
|
||||
def _section_bounds(body: str, heading: str) -> tuple[int, int] | None:
|
||||
match = sections.heading_re(heading).search(body)
|
||||
if not match:
|
||||
return None
|
||||
start = match.end()
|
||||
next_heading = re.search(r"^## ", body[start:], re.MULTILINE)
|
||||
end = start + next_heading.start() if next_heading else len(body)
|
||||
return start, end
|
||||
|
||||
|
||||
def add_bullet_to_section(body: str, heading: str, bullet: str, dedup_link: str) -> str:
|
||||
"""Insert `bullet` into the `## {heading}` section of body, unless a
|
||||
wikilink to dedup_link already appears there. Creates the section
|
||||
(before the See Also section if present, else at the end) if missing.
|
||||
|
||||
`heading` is a canonical name from `sections`; an existing section is found
|
||||
under its aliases too, so a page that has not been translated yet is still
|
||||
appended to rather than given a duplicate section. A section this creates
|
||||
always carries the canonical name."""
|
||||
bounds = _section_bounds(body, heading)
|
||||
if bounds is None:
|
||||
section = f"## {heading}\n\n{bullet}\n\n"
|
||||
see_also = sections.heading_re(sections.SEE_ALSO).search(body)
|
||||
if heading != sections.SEE_ALSO and see_also:
|
||||
return body[: see_also.start()] + section + body[see_also.start() :]
|
||||
return body.rstrip("\n") + "\n\n" + section.rstrip("\n") + "\n"
|
||||
|
||||
start, end = bounds
|
||||
section_text = body[start:end]
|
||||
if f"[[{dedup_link}]]" in section_text:
|
||||
return body
|
||||
trimmed = section_text.rstrip("\n")
|
||||
new_section = trimmed + "\n" + bullet + "\n\n"
|
||||
return body[:start] + new_section + body[end:]
|
||||
|
||||
|
||||
def add_relationship_bullet(body: str, label: str, other_title: str) -> str:
|
||||
bullet = f"- **{label}:** [[{other_title}]]"
|
||||
return add_bullet_to_section(body, sections.RELATIONSHIPS, bullet, other_title)
|
||||
|
||||
|
||||
def add_see_also_bullet(body: str, other_title: str) -> str:
|
||||
return add_bullet_to_section(body, sections.SEE_ALSO, f"- [[{other_title}]]", other_title)
|
||||
|
||||
|
||||
@app.command("add")
|
||||
def xref_add(
|
||||
a: str = typer.Option(..., "--a", help="Exact title of page A"),
|
||||
b: str = typer.Option(..., "--b", help="Exact title of page B"),
|
||||
rel_a: str = typer.Option("related to", "--rel-a", help="Relationship label on A pointing to B"),
|
||||
rel_b: str = typer.Option("related to", "--rel-b", help="Relationship label on B pointing to A"),
|
||||
see_also: bool = typer.Option(True, "--see-also/--no-see-also", help="Also add reciprocal 'See Also' bullets"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes to both pages instead of writing"),
|
||||
):
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
page_a = _find_page(pages, a)
|
||||
page_b = _find_page(pages, b)
|
||||
|
||||
# Both refusals before either write, so a rejected pair leaves no half-link.
|
||||
_require_related_field(page_a, a)
|
||||
_require_related_field(page_b, b)
|
||||
|
||||
related_changed_a = add_related(page_a.frontmatter, b)
|
||||
related_changed_b = add_related(page_b.frontmatter, a)
|
||||
|
||||
body_a = add_relationship_bullet(page_a.body, rel_a, b)
|
||||
body_b = add_relationship_bullet(page_b.body, rel_b, a)
|
||||
if see_also:
|
||||
body_a = add_see_also_bullet(body_a, b)
|
||||
body_b = add_see_also_bullet(body_b, a)
|
||||
|
||||
changed_a = related_changed_a or body_a != page_a.body
|
||||
changed_b = related_changed_b or body_b != page_b.body
|
||||
|
||||
if dry_run:
|
||||
state_a = "would update" if changed_a else "already up to date"
|
||||
state_b = "would update" if changed_b else "already up to date"
|
||||
typer.echo(f"[dry-run] '{a}': {state_a} (related / Relationships / See Also)")
|
||||
typer.echo(f"[dry-run] '{b}': {state_b} (related / Relationships / See Also)")
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
try:
|
||||
write_page(page_a.path, page_a.frontmatter, body_a)
|
||||
except OSError as exc:
|
||||
fail(f"Failed to write '{a}': {exc}. '{b}' was not touched - fix the write failure and retry once.")
|
||||
|
||||
try:
|
||||
write_page(page_b.path, page_b.frontmatter, body_b)
|
||||
except OSError as exc:
|
||||
fail(
|
||||
f"'{a}' was updated but writing '{b}' failed: {exc}. The link is now one-directional - "
|
||||
f"fix the write failure, then re-run `xref add --a \"{a}\" --b \"{b}\"` (idempotent, safe to retry)."
|
||||
)
|
||||
success(f"Linked '{a}' <-> '{b}' ({rel_a} / {rel_b})")
|
||||
|
||||
|
||||
def remove_related(frontmatter: dict, other_title: str) -> bool:
|
||||
"""Drop other_title from frontmatter['related'] if present. Returns True if
|
||||
a change was made."""
|
||||
related = frontmatter.get("related")
|
||||
if not related or other_title not in related:
|
||||
return False
|
||||
frontmatter["related"] = [title for title in related if title != other_title]
|
||||
return True
|
||||
|
||||
def remove_link_bullets(body: str, other_title: str) -> str:
|
||||
"""Remove the whole-line Relationships/See Also bullets `xref add` writes -
|
||||
`- **label:** [[Other]]` and `- [[Other]]`.
|
||||
|
||||
Deliberately narrow, matching `xref add`'s own output: a bullet carrying
|
||||
prose alongside the link is left for the author to edit.
|
||||
"""
|
||||
escaped = re.escape(other_title)
|
||||
pattern = re.compile(
|
||||
rf"^[ \t]*-[ \t]+(?:\*\*[^*\n]+:\*\*[ \t]+)?\[\[{escaped}\]\][ \t]*\n?",
|
||||
re.MULTILINE,
|
||||
)
|
||||
return pattern.sub("", body)
|
||||
|
||||
|
||||
@app.command("remove")
|
||||
def xref_remove(
|
||||
a: str = typer.Option(..., "--a", help="Exact title of page A (must exist)"),
|
||||
b: str = typer.Option(..., "--b", help="Title to unlink from A; need not still exist as a page"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview changes instead of writing"),
|
||||
):
|
||||
"""Remove a cross-reference: the inverse of `xref add`.
|
||||
|
||||
Clears `--b` from *every* page-ref frontmatter field the type declares
|
||||
(`related:`, `sources:`, `entities:`, `concepts:`), not just `related:`,
|
||||
so it is equally the inverse of `xref link-source`.
|
||||
|
||||
`--b` deliberately does not have to exist. Clearing a reference left
|
||||
behind by a hand-deleted or hand-renamed page is the main reason this
|
||||
command exists, and in that case the target is exactly what is missing.
|
||||
Idempotent: removing a link that is already gone is a no-op.
|
||||
"""
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
page_a = _find_page(pages, a)
|
||||
page_b = pages.get(b)
|
||||
|
||||
body_a = remove_link_bullets(page_a.body, b)
|
||||
changed_a = strip_frontmatter_ref(page_a, b) or body_a != page_a.body
|
||||
|
||||
changed_b = False
|
||||
body_b = ""
|
||||
if page_b is not None:
|
||||
body_b = remove_link_bullets(page_b.body, a)
|
||||
changed_b = strip_frontmatter_ref(page_b, a) or body_b != page_b.body
|
||||
|
||||
if dry_run:
|
||||
typer.echo(f"[dry-run] '{a}': {'would update' if changed_a else 'no reference to remove'}")
|
||||
if page_b is None:
|
||||
typer.echo(f"[dry-run] '{b}': not a page - only '{a}' would be updated")
|
||||
else:
|
||||
typer.echo(f"[dry-run] '{b}': {'would update' if changed_b else 'no reference to remove'}")
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
if changed_a:
|
||||
write_page(page_a.path, page_a.frontmatter, body_a)
|
||||
if changed_b and page_b is not None:
|
||||
write_page(page_b.path, page_b.frontmatter, body_b)
|
||||
|
||||
if not changed_a and not changed_b:
|
||||
success(f"No link between '{a}' and '{b}' to remove; nothing changed.")
|
||||
return
|
||||
if page_b is None:
|
||||
success(f"Removed '{a}' -> '{b}' ('{b}' is not a page, so only '{a}' was updated).")
|
||||
return
|
||||
success(f"Unlinked '{a}' <-> '{b}'")
|
||||
|
||||
|
||||
@app.command("link-source")
|
||||
def xref_link_source(
|
||||
source: str = typer.Option(..., "--source", help="Exact source page title, e.g. 'Source - Docker Cheatsheet'"),
|
||||
entities: str = typer.Option(..., "--entities", help="Comma-separated entity/concept titles the source mentions"),
|
||||
dry_run: bool = typer.Option(False, "--dry-run", help="Preview which pages would be linked instead of writing"),
|
||||
):
|
||||
pages = load_kb_pages(config.KB_DIR)
|
||||
source_page = _find_page(pages, source)
|
||||
names = parse_list(entities)
|
||||
|
||||
linked: list[str] = []
|
||||
skipped: list[str] = []
|
||||
failed: list[str] = []
|
||||
unrouted: list[str] = []
|
||||
source_changed = False
|
||||
for name in names:
|
||||
page = pages.get(name)
|
||||
if page is None:
|
||||
skipped.append(name)
|
||||
continue
|
||||
sources = page.frontmatter.setdefault("sources", [])
|
||||
if source not in sources:
|
||||
sources.append(source)
|
||||
body = add_see_also_bullet(page.body, source)
|
||||
|
||||
# The way back. Until this existed the command wrote only the targets,
|
||||
# so a source page's own `entities:`/`concepts:` stayed as `new` left
|
||||
# them - and an ingest that creates its concept pages *after* the
|
||||
# source page (which it must, since their titles come out of the
|
||||
# extraction) left them empty with no command able to fill them.
|
||||
field = _back_reference_field(source_page, page)
|
||||
if field is None:
|
||||
unrouted.append(name)
|
||||
else:
|
||||
entries = source_page.frontmatter.setdefault(field, [])
|
||||
if name not in entries:
|
||||
entries.append(name)
|
||||
source_changed = True
|
||||
|
||||
if not dry_run:
|
||||
try:
|
||||
write_page(page.path, page.frontmatter, body)
|
||||
except OSError as exc:
|
||||
failed.append(f"{name} ({exc})")
|
||||
continue
|
||||
linked.append(name)
|
||||
|
||||
if source_changed and not dry_run:
|
||||
try:
|
||||
write_page(source_page.path, source_page.frontmatter, source_page.body)
|
||||
except OSError as exc:
|
||||
fail(
|
||||
f"Targets were updated but writing '{source}' failed: {exc}. Its reference "
|
||||
f"arrays are now behind - fix the write failure and re-run (idempotent)."
|
||||
)
|
||||
if unrouted:
|
||||
typer.echo(
|
||||
f"Not recorded on '{source}' (no matching reference field for their collection): "
|
||||
f"{', '.join(unrouted)}"
|
||||
)
|
||||
|
||||
if linked:
|
||||
verb = "Would link" if dry_run else "Linked"
|
||||
typer.echo(f"{verb} source '{source}' to: {', '.join(linked)}")
|
||||
if skipped:
|
||||
typer.echo(f"Skipped (page not found): {', '.join(skipped)}")
|
||||
if failed:
|
||||
typer.echo(f"Failed to write (fix and re-run for just these names): {', '.join(failed)}")
|
||||
|
||||
if dry_run:
|
||||
typer.echo("No files written (--dry-run).")
|
||||
return
|
||||
|
||||
if skipped or failed:
|
||||
parts = []
|
||||
if skipped:
|
||||
parts.append(f"page(s) not found: {', '.join(skipped)}")
|
||||
if failed:
|
||||
parts.append(f"page(s) failed to write: {', '.join(failed)}")
|
||||
fail(f"Linked {len(linked)}/{len(names)} page(s); " + "; ".join(parts))
|
||||
|
||||
success(f"Linked source '{source}' to {len(names)} page(s)")
|
||||
Reference in new issue
Block a user