Files
chemenu/tools/chemenu/commands/provenance_cmd.py
T
torben be78ad20af
CI / verify (push) Successful in 1m12s
Release / release (push) Successful in 36s
tools: command records, Provenance group - examples, exit lines per cause (#142)
Files changed:
- CHANGES.md
- VERSION
- tools/CONTRACT.md
- tools/chemenu/commands/provenance_cmd.py
2026-09-26 08:51:53 +02:00

279 lines
11 KiB
Python

"""`wikitool sources ...` - raw-file <-> wiki provenance tooling.
This is the deterministic backbone for citation backtracing: it never guesses
which raw file backs a claim, it only reports what the frontmatter and inline
`[^cite-id]` footnotes already declare. Filling in those declarations
correctly is still the LLM's job.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Optional
import typer
from chemenu import cli_contract, config
from chemenu.commands._util import fail, rel_path, success
from chemenu.provenance import (
broken_raw_refs,
citing_pages,
legacy_source_pages,
page_raw_files,
source_pages_by_raw_file,
source_raw_files,
uncovered_raw_files,
)
from chemenu.kb_scan import load_kb_pages
app = typer.Typer(help="Trace and lint raw-file <-> wiki-page provenance.")
def _normalize_raw_path(raw: str) -> str:
"""Accept an absolute path or a path relative to the repo root or to raw/,
and return it as a path relative to the repo root (matching how it is
stored in `raw_files:`)."""
candidate = Path(raw)
if candidate.is_absolute():
try:
return str(candidate.relative_to(config.ROOT))
except ValueError:
return str(candidate)
if candidate.exists():
return str(candidate)
if (config.ROOT / candidate).exists():
return str(candidate)
if (config.RAW_DIR / candidate).exists():
return str((Path("raw") / candidate))
return raw
@app.command("coverage")
@cli_contract.record(cli_contract.CommandRecord(
path="sources coverage",
summary="List raw files with no source page, broken `raw_files:` references, and legacy "
"directory/URL-only source pages.",
synopsis=(cli_contract.Variant(usage="sources coverage [--json]"),),
properties=cli_contract.Properties(
effect=cli_contract.Effect.READ,
idempotent=cli_contract.Idempotent.YES,
atomic="Read-only",
budget=cli_contract.Budget.COUNTED,
),
notes=(
"Lists raw files with no source page, broken `raw_files:` references, and legacy "
"directory/URL-only source pages.",
"`--json` prints the same lists as JSON.",
"Never fails; read-only and safe to retry freely.",
),
failures=(),
examples=(
"tools/wikitool sources coverage",
"tools/wikitool sources coverage --json",
),
see_also=(
"`wikitool sources trace` - follows one file or page",
"`wiki-ingest` skill - turns an uncovered raw file into a source page",
),
))
def coverage(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
"""Report raw files with no source page, broken raw_files: references, and
source pages still using a legacy directory/URL-only `source:` field."""
pages = load_kb_pages(config.KB_DIR)
report = {
"uncovered_raw_files": uncovered_raw_files(config.RAW_DIR, pages),
"broken_raw_refs": broken_raw_refs(pages),
"legacy_source_pages": legacy_source_pages(pages),
}
if json_out:
typer.echo(json.dumps(report, indent=2))
return
typer.echo(f"Uncovered raw files: {len(report['uncovered_raw_files'])}")
for f in report["uncovered_raw_files"]:
typer.echo(f" - {f}")
typer.echo(f"Broken raw_files references: {len(report['broken_raw_refs'])}")
for item in report["broken_raw_refs"]:
typer.echo(f" - [[{item['page']}]] -> {item['raw_path']}")
typer.echo(f"Legacy (directory/URL-only) source pages: {len(report['legacy_source_pages'])}")
for item in report["legacy_source_pages"]:
typer.echo(f" - [[{item['page']}]] ({item['reason']}): {item['source']}")
@app.command("trace")
@cli_contract.record(cli_contract.CommandRecord(
path="sources trace",
summary="Trace provenance in either direction: raw file, or page.",
synopsis=(cli_contract.Variant(usage='sources trace --raw <path> | --page "<Title>"'),),
properties=cli_contract.Properties(
effect=cli_contract.Effect.READ,
idempotent=cli_contract.Idempotent.YES,
atomic="Read-only",
budget=cli_contract.Budget.COUNTED,
),
notes=(
"`--raw <path>`: raw file -> the source page(s) covering it -> the pages citing those.",
"`--page \"<Title>\"`: page -> its sources -> their raw files.",
"Read-only.",
),
failures=(
cli_contract.Failure(
cause="Neither or both of `--raw`/`--page` given, or `--page` names an unknown page",
reaction="Fix the argument and retry",
),
cli_contract.Failure(
cause="`--raw` names a file no source page covers - reported as a plain finding "
"plus exit 1, not the usual `ERROR`-prefixed rejection",
reaction="Nothing to retry: the file is uncovered. Ingest it, or check the path",
),
),
examples=(
'tools/wikitool sources trace --page "Docker"',
'tools/wikitool sources trace --raw "raw/articles/llm-wiki.md"',
),
see_also=(
"`wikitool sources coverage` - every uncovered raw file at once",
"`wikitool xref link-source` - links a source page to what it mentions",
),
))
def trace(
raw: Optional[str] = typer.Option(None, "--raw", help="Raw file path to trace forward from"),
page: Optional[str] = typer.Option(None, "--page", help="Wiki page title to trace backward from"),
):
"""Trace provenance in either direction: --raw shows which source pages
cover a raw file and which wiki pages cite it; --page shows which sources
and raw files back a given wiki page."""
if bool(raw) == bool(page):
fail("Provide exactly one of --raw or --page")
pages = load_kb_pages(config.KB_DIR)
if raw:
raw_key = _normalize_raw_path(raw)
by_raw = source_pages_by_raw_file(pages)
source_titles = by_raw.get(raw_key, [])
if not source_titles:
typer.echo(f"No source page covers {raw_key}")
raise typer.Exit(code=1)
for source_title in source_titles:
typer.echo(f"{raw_key}")
typer.echo(f" covered by: [[{source_title}]]")
citers = citing_pages(pages, source_title)
if citers:
for c in citers:
typer.echo(f" cited by: [[{c}]]")
else:
typer.echo(" cited by: (nothing yet)")
return
target_page = pages.get(page)
if target_page is None:
fail(f"No page titled '{page}' found")
sources = target_page.frontmatter.get("sources") or []
typer.echo(f"[[{page}]]")
if not sources:
typer.echo(" sources: (none listed)")
for source_title in sources:
typer.echo(f" sources: [[{source_title}]]")
source_page = pages.get(source_title)
if source_page is None:
typer.echo(" (source page not found)")
continue
for raw_path in source_raw_files(source_page):
typer.echo(f" raw file: {raw_path}")
raw_files = page_raw_files(pages, target_page)
typer.echo(f" all raw files (incl. inline citations): {raw_files or '(none)'}")
def build_provenance_index(kb_dir: Path, raw_dir: Path) -> str:
pages = load_kb_pages(kb_dir)
by_raw = source_pages_by_raw_file(pages)
all_raw = sorted(str(p.relative_to(config.ROOT)) for p in config.iter_raw_files(raw_dir))
uncovered = uncovered_raw_files(raw_dir, pages)
lines: list[str] = []
lines.append("# Provenance Index")
lines.append("")
lines.append("Generated by `tools/wikitool sources rebuild-index`. Do not hand-edit.")
lines.append("")
lines.append("Maps every raw source file to the wiki source page(s) that cover it, and")
lines.append("every wiki page that cites that source (via frontmatter `sources:` or an")
lines.append("inline `[^cite-id]` footnote).")
lines.append("")
lines.append("## Coverage Summary")
lines.append("")
lines.append(f"- **Total raw files:** {len(all_raw)}")
lines.append(f"- **Covered:** {len(all_raw) - len(uncovered)}")
lines.append(f"- **Uncovered:** {len(uncovered)}")
lines.append("")
lines.append("---")
lines.append("")
lines.append("## Raw Files")
lines.append("")
for raw_path in all_raw:
lines.append(f"### `{raw_path}`")
lines.append("")
source_titles = by_raw.get(raw_path, [])
if not source_titles:
lines.append("No source page covers this file yet.")
lines.append("")
continue
for source_title in source_titles:
lines.append(f"- Covered by: [[{source_title}]]")
citers = citing_pages(pages, source_title)
if citers:
lines.append(f" - Cited by: {', '.join(f'[[{c}]]' for c in citers)}")
else:
lines.append(" - Cited by: (nothing yet)")
lines.append("")
return "\n".join(lines) + "\n"
@app.command("rebuild-index")
@cli_contract.record(cli_contract.CommandRecord(
path="sources rebuild-index",
summary="Regenerate the `kb/provenance.md` reverse index.",
synopsis=(cli_contract.Variant(usage="sources rebuild-index [--dry-run]"),),
properties=cli_contract.Properties(
effect=cli_contract.Effect.WRITE,
idempotent=cli_contract.Idempotent.YES,
atomic="Yes - the single provenance file is regenerated from scratch",
budget=cli_contract.Budget.COUNTED,
),
notes=(
"Regenerates the `kb/provenance.md` reverse index (raw file -> source page -> citing "
"pages) from scratch.",
"`--dry-run` prints the result instead of writing `kb/provenance.md`.",
),
failures=(cli_contract.Failure(
cause="An I/O error while writing `kb/provenance.md` (rare)",
reaction="Safe to retry freely",
),),
examples=(
"tools/wikitool sources rebuild-index",
"tools/wikitool sources rebuild-index --dry-run",
),
never=(
"Never hand-edit `kb/provenance.md` - re-run this command instead.",
),
see_also=(
"`wikitool index rebuild` - the page catalog, rebuilt alongside",
"`instructions/publish-cycle.md` - where a write session runs this",
),
))
def rebuild_index(
dry_run: bool = typer.Option(False, "--dry-run", help="Print the result instead of writing kb/provenance.md"),
):
"""Regenerate the `kb/provenance.md` reverse index."""
content = build_provenance_index(config.KB_DIR, config.RAW_DIR)
provenance_file = config.KB_DIR / "provenance.md"
if dry_run:
# nl=False so the preview is byte-identical to the file that would be
# written; see the same note in index_build.py.
typer.echo(content, nl=False)
return
provenance_file.write_text(content, encoding="utf-8")
success(f"Rebuilt {rel_path(provenance_file)}")