Files changed: - AGENTS.md - CHANGES.md - VERSION - instructions/wiki-query/SKILL.md - tools/CONTRACT.md - tools/chemenu/api.py - tools/chemenu/commands/search.py - tools/chemenu/mcp/server.py - tools/chemenu/search/service.py - tools/chemenu/search/types.py - tools/chemenu/tests/test_api.py - tools/chemenu/tests/test_mcp_server.py - tools/chemenu/tests/test_search.py
208 lines
8.7 KiB
Python
208 lines
8.7 KiB
Python
"""The in-process entry point: point Chemenu at a corpus and read from it.
|
|
|
|
This is the seam the MCP server (Gitea #19) is built on, and the reason it is
|
|
worth having as its own module rather than as "call the command functions
|
|
yourself": it fixes the two things that made an in-process caller a second-class
|
|
one.
|
|
|
|
**A corpus you name, not the one this package happens to sit in.** `Corpus`
|
|
holds the root, threads it into the loader and the search backend, and points
|
|
`config` at it for the duration of each call so the parts that reach for
|
|
`config` directly - the `TypeResolver` singleton, which has to find `types/` -
|
|
follow too. A test proves no path of the developer's checkout is read while a
|
|
foreign root is set.
|
|
|
|
**Values and exceptions, not exit codes.** The functions return the same
|
|
structures the CLI's `--json` forms print - one wire contract, with the CLI as
|
|
its executable specification - and raise `ChemenuError` where the CLI would
|
|
print `ERROR` and leave through `typer.Exit(1)`.
|
|
|
|
Read-only, structurally: nothing under `chemenu.commands` is imported, so `new`,
|
|
`touch`, `xref`, `cite`, `publish`, `migrate` and `version bump` are not
|
|
reachable from here at all. That is the property #19 asks for - the write
|
|
functions do not exist in this surface rather than being filtered out of it.
|
|
|
|
Import cost is the whole read core and nothing else: `yaml` and `jsonschema`,
|
|
plus the standard library. No `typer`, no `rich`.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import datetime
|
|
from pathlib import Path
|
|
from typing import Any, Iterable, Optional
|
|
|
|
from chemenu import config
|
|
from chemenu.corpus_cache import CorpusCache
|
|
from chemenu.errors import BackendError, ChemenuError, ValidationError
|
|
from chemenu.lint_core import run_lint
|
|
from chemenu.search import filters
|
|
from chemenu.search.registry import resolve
|
|
from chemenu.search.service import run_search, unreadable_pages
|
|
from chemenu.search.types import DEFAULT_LIMIT, Predicate, SearchQuery
|
|
from chemenu.types_core import describe_type, list_types
|
|
|
|
# Distinguishes "the caller did not pass a revision" from "the caller passed
|
|
# None", which is itself a meaningful answer: no commit, because the tree is
|
|
# dirty or is not a checkout.
|
|
_UNREAD = object()
|
|
|
|
__all__ = [
|
|
"Corpus",
|
|
"ChemenuError",
|
|
"ValidationError",
|
|
"BackendError",
|
|
]
|
|
|
|
|
|
class Corpus:
|
|
"""One corpus tree, read repeatedly.
|
|
|
|
`root` follows `config.resolve_root()`: an explicit path, else
|
|
`$CHEMENU_ROOT`, else the checkout this package lives in. `kb_dir` defaults
|
|
to `<root>/kb` and is separate only because the search backend already
|
|
distinguishes the two.
|
|
|
|
Holds a `CorpusCache`, so a long-lived caller parses the corpus once per
|
|
commit instead of once per request - and never answers from a superseded
|
|
parse, because a dirty tree is not cached. Not thread-safe: a server serving
|
|
concurrent requests holds the lock, for the reason given in
|
|
`chemenu/corpus_cache.py`.
|
|
"""
|
|
|
|
def __init__(self, root: Optional[Path | str] = None, kb_dir: Optional[Path | str] = None):
|
|
self.root = config.resolve_root(root)
|
|
self.kb_dir = Path(kb_dir) if kb_dir is not None else self.root / "kb"
|
|
self._cache = CorpusCache(self.kb_dir, self.root)
|
|
|
|
@property
|
|
def revision(self) -> Optional[str]:
|
|
"""The commit every answer from this corpus is stamped with, or None
|
|
when the tree is dirty or is not a git checkout - in which case the
|
|
answer corresponds to no commit, and says so."""
|
|
return self._cache.current_revision()
|
|
|
|
def _rooted(self):
|
|
"""Point `config` at this corpus for the duration of one call.
|
|
|
|
Threading a root through every argument gets the search backend and the
|
|
corpus loader, and misses the module-level `TypeResolver` singleton -
|
|
which resolves `types/` and is what `Page.kind` goes through. Without
|
|
this, a foreign corpus is read with *this* checkout's type specs, and
|
|
the page's `kind` is an answer about the wrong instance.
|
|
|
|
Process-wide while open, so `Corpus` inherits `config.rooted()`'s
|
|
thread-safety constraint: one lock per process, held by the caller.
|
|
"""
|
|
return config.rooted(self.root)
|
|
|
|
def _stamp(
|
|
self, payload: dict[str, Any], revision: Optional[str] = _UNREAD
|
|
) -> dict[str, Any]:
|
|
"""Every response carries the revision it was computed from.
|
|
|
|
A stale checkout otherwise answers confidently and wrongly, which is the
|
|
failure `SOUL.md` names as the cardinal one. The stamp turns a silent
|
|
stale answer into a visible one.
|
|
|
|
A caller that loaded the corpus passes the revision it got back, rather
|
|
than letting this ask again: between the load and the stamp the tree can
|
|
move, and the honest answer is the revision the pages actually came
|
|
from. Callers that read no pages (`types`) ask for the current one.
|
|
"""
|
|
if revision is _UNREAD:
|
|
revision = self._cache.current_revision()
|
|
payload["commit"] = revision
|
|
payload["as_of"] = datetime.datetime.now(datetime.timezone.utc).isoformat()
|
|
return payload
|
|
|
|
def search(
|
|
self,
|
|
text: Optional[str] = None,
|
|
predicates: Iterable[str] = (),
|
|
regex: bool = False,
|
|
limit: int = DEFAULT_LIMIT,
|
|
sort: Optional[str] = None,
|
|
backend: Optional[str] = None,
|
|
) -> dict[str, Any]:
|
|
"""`wikitool search --json`, as a value.
|
|
|
|
`predicates` takes the raw `--field` strings, so the CLI and this share
|
|
one parser and cannot drift on what `entity_type=system` means.
|
|
"""
|
|
raw = list(predicates)
|
|
if not text and not raw:
|
|
raise ValidationError(
|
|
"Nothing to search for: give a query, or at least one predicate."
|
|
)
|
|
parsed: tuple[Predicate, ...] = tuple(filters.parse_predicate(p) for p in raw)
|
|
backends = resolve(backend, self.kb_dir, self.root)
|
|
query = SearchQuery(
|
|
text=text, predicates=parsed, regex=regex, limit=limit, sort=sort
|
|
)
|
|
|
|
with self._rooted():
|
|
pages, revision = self._cache.load()
|
|
result = run_search(query, pages, backends, self.kb_dir)
|
|
return self._stamp({
|
|
"query": text,
|
|
"predicates": [p.render() for p in parsed],
|
|
"backend": ",".join(b.name for b in backends),
|
|
# Same shape the CLI's `--json` prints: `count` is what came back,
|
|
# `total` is how many matched before `limit` cut it.
|
|
"count": len(result.hits),
|
|
"total": result.total,
|
|
"truncated": result.truncated,
|
|
"limit": result.limit,
|
|
"results": [hit.as_dict() for hit in result.hits],
|
|
"unreadable": unreadable_pages(pages),
|
|
}, revision)
|
|
|
|
def lint(self) -> dict[str, Any]:
|
|
"""`wikitool lint --json`, as a value.
|
|
|
|
The JSON form only - `lint` without a flag writes a report into
|
|
`reports/`, and a read surface does not write into the tree it is
|
|
reading.
|
|
"""
|
|
with self._rooted():
|
|
return self._stamp(run_lint(self.kb_dir))
|
|
|
|
def types(self) -> dict[str, Any]:
|
|
"""`wikitool types list --json`, as a value."""
|
|
with self._rooted():
|
|
return self._stamp({"types": list_types()})
|
|
|
|
def describe_type(self, name: str) -> dict[str, Any]:
|
|
"""`wikitool types describe <name> --json`, as a value. Raises
|
|
`UnknownType` (a `ValidationError`) for a name that does not exist."""
|
|
with self._rooted():
|
|
return self._stamp(describe_type(name))
|
|
|
|
def status(self) -> dict[str, Any]:
|
|
"""A composed snapshot: how big the corpus is and what lint says about
|
|
it, without the full report.
|
|
|
|
Composed here on purpose. There is no `wikitool status` to wrap -
|
|
`wiki-status` is a *skill* that assembles `kb/index.md`, `lint` and
|
|
`kb/log.md` - so this is a new surface, and saying so is what keeps
|
|
anyone from looking for the CLI command it mirrors.
|
|
"""
|
|
with self._rooted():
|
|
pages, revision = self._cache.load()
|
|
report = run_lint(self.kb_dir)
|
|
collections: dict[str, int] = {}
|
|
for page in pages.values():
|
|
name = filters.collection_of(page.path, self.kb_dir)
|
|
if name:
|
|
collections[name] = collections.get(name, 0) + 1
|
|
return self._stamp({
|
|
"pages": len(pages),
|
|
"collections": dict(sorted(collections.items())),
|
|
"findings": {
|
|
key: len(value)
|
|
for key, value in sorted(report.items())
|
|
if isinstance(value, list)
|
|
},
|
|
"unreadable": unreadable_pages(pages),
|
|
}, revision)
|