Files
chemenu/tools/chemenu/api.py
T
torben bb097f614b
CI / verify (push) Failing after 40s
Release / release (push) Successful in 36s
search: Pfad und Titel vollstaendig in der Trefferzeile, Trunkierung wird benannt (schliesst #100)
Files changed:
- AGENTS.md
- CHANGES.md
- VERSION
- instructions/wiki-query/SKILL.md
- tools/CONTRACT.md
- tools/chemenu/api.py
- tools/chemenu/commands/search.py
- tools/chemenu/mcp/server.py
- tools/chemenu/search/service.py
- tools/chemenu/search/types.py
- tools/chemenu/tests/test_api.py
- tools/chemenu/tests/test_mcp_server.py
- tools/chemenu/tests/test_search.py
2026-09-15 21:26:39 +02:00

208 lines
8.7 KiB
Python

"""The in-process entry point: point Chemenu at a corpus and read from it.
This is the seam the MCP server (Gitea #19) is built on, and the reason it is
worth having as its own module rather than as "call the command functions
yourself": it fixes the two things that made an in-process caller a second-class
one.
**A corpus you name, not the one this package happens to sit in.** `Corpus`
holds the root, threads it into the loader and the search backend, and points
`config` at it for the duration of each call so the parts that reach for
`config` directly - the `TypeResolver` singleton, which has to find `types/` -
follow too. A test proves no path of the developer's checkout is read while a
foreign root is set.
**Values and exceptions, not exit codes.** The functions return the same
structures the CLI's `--json` forms print - one wire contract, with the CLI as
its executable specification - and raise `ChemenuError` where the CLI would
print `ERROR` and leave through `typer.Exit(1)`.
Read-only, structurally: nothing under `chemenu.commands` is imported, so `new`,
`touch`, `xref`, `cite`, `publish`, `migrate` and `version bump` are not
reachable from here at all. That is the property #19 asks for - the write
functions do not exist in this surface rather than being filtered out of it.
Import cost is the whole read core and nothing else: `yaml` and `jsonschema`,
plus the standard library. No `typer`, no `rich`.
"""
from __future__ import annotations
import datetime
from pathlib import Path
from typing import Any, Iterable, Optional
from chemenu import config
from chemenu.corpus_cache import CorpusCache
from chemenu.errors import BackendError, ChemenuError, ValidationError
from chemenu.lint_core import run_lint
from chemenu.search import filters
from chemenu.search.registry import resolve
from chemenu.search.service import run_search, unreadable_pages
from chemenu.search.types import DEFAULT_LIMIT, Predicate, SearchQuery
from chemenu.types_core import describe_type, list_types
# Distinguishes "the caller did not pass a revision" from "the caller passed
# None", which is itself a meaningful answer: no commit, because the tree is
# dirty or is not a checkout.
_UNREAD = object()
__all__ = [
"Corpus",
"ChemenuError",
"ValidationError",
"BackendError",
]
class Corpus:
"""One corpus tree, read repeatedly.
`root` follows `config.resolve_root()`: an explicit path, else
`$CHEMENU_ROOT`, else the checkout this package lives in. `kb_dir` defaults
to `<root>/kb` and is separate only because the search backend already
distinguishes the two.
Holds a `CorpusCache`, so a long-lived caller parses the corpus once per
commit instead of once per request - and never answers from a superseded
parse, because a dirty tree is not cached. Not thread-safe: a server serving
concurrent requests holds the lock, for the reason given in
`chemenu/corpus_cache.py`.
"""
def __init__(self, root: Optional[Path | str] = None, kb_dir: Optional[Path | str] = None):
self.root = config.resolve_root(root)
self.kb_dir = Path(kb_dir) if kb_dir is not None else self.root / "kb"
self._cache = CorpusCache(self.kb_dir, self.root)
@property
def revision(self) -> Optional[str]:
"""The commit every answer from this corpus is stamped with, or None
when the tree is dirty or is not a git checkout - in which case the
answer corresponds to no commit, and says so."""
return self._cache.current_revision()
def _rooted(self):
"""Point `config` at this corpus for the duration of one call.
Threading a root through every argument gets the search backend and the
corpus loader, and misses the module-level `TypeResolver` singleton -
which resolves `types/` and is what `Page.kind` goes through. Without
this, a foreign corpus is read with *this* checkout's type specs, and
the page's `kind` is an answer about the wrong instance.
Process-wide while open, so `Corpus` inherits `config.rooted()`'s
thread-safety constraint: one lock per process, held by the caller.
"""
return config.rooted(self.root)
def _stamp(
self, payload: dict[str, Any], revision: Optional[str] = _UNREAD
) -> dict[str, Any]:
"""Every response carries the revision it was computed from.
A stale checkout otherwise answers confidently and wrongly, which is the
failure `SOUL.md` names as the cardinal one. The stamp turns a silent
stale answer into a visible one.
A caller that loaded the corpus passes the revision it got back, rather
than letting this ask again: between the load and the stamp the tree can
move, and the honest answer is the revision the pages actually came
from. Callers that read no pages (`types`) ask for the current one.
"""
if revision is _UNREAD:
revision = self._cache.current_revision()
payload["commit"] = revision
payload["as_of"] = datetime.datetime.now(datetime.timezone.utc).isoformat()
return payload
def search(
self,
text: Optional[str] = None,
predicates: Iterable[str] = (),
regex: bool = False,
limit: int = DEFAULT_LIMIT,
sort: Optional[str] = None,
backend: Optional[str] = None,
) -> dict[str, Any]:
"""`wikitool search --json`, as a value.
`predicates` takes the raw `--field` strings, so the CLI and this share
one parser and cannot drift on what `entity_type=system` means.
"""
raw = list(predicates)
if not text and not raw:
raise ValidationError(
"Nothing to search for: give a query, or at least one predicate."
)
parsed: tuple[Predicate, ...] = tuple(filters.parse_predicate(p) for p in raw)
backends = resolve(backend, self.kb_dir, self.root)
query = SearchQuery(
text=text, predicates=parsed, regex=regex, limit=limit, sort=sort
)
with self._rooted():
pages, revision = self._cache.load()
result = run_search(query, pages, backends, self.kb_dir)
return self._stamp({
"query": text,
"predicates": [p.render() for p in parsed],
"backend": ",".join(b.name for b in backends),
# Same shape the CLI's `--json` prints: `count` is what came back,
# `total` is how many matched before `limit` cut it.
"count": len(result.hits),
"total": result.total,
"truncated": result.truncated,
"limit": result.limit,
"results": [hit.as_dict() for hit in result.hits],
"unreadable": unreadable_pages(pages),
}, revision)
def lint(self) -> dict[str, Any]:
"""`wikitool lint --json`, as a value.
The JSON form only - `lint` without a flag writes a report into
`reports/`, and a read surface does not write into the tree it is
reading.
"""
with self._rooted():
return self._stamp(run_lint(self.kb_dir))
def types(self) -> dict[str, Any]:
"""`wikitool types list --json`, as a value."""
with self._rooted():
return self._stamp({"types": list_types()})
def describe_type(self, name: str) -> dict[str, Any]:
"""`wikitool types describe <name> --json`, as a value. Raises
`UnknownType` (a `ValidationError`) for a name that does not exist."""
with self._rooted():
return self._stamp(describe_type(name))
def status(self) -> dict[str, Any]:
"""A composed snapshot: how big the corpus is and what lint says about
it, without the full report.
Composed here on purpose. There is no `wikitool status` to wrap -
`wiki-status` is a *skill* that assembles `kb/index.md`, `lint` and
`kb/log.md` - so this is a new surface, and saying so is what keeps
anyone from looking for the CLI command it mirrors.
"""
with self._rooted():
pages, revision = self._cache.load()
report = run_lint(self.kb_dir)
collections: dict[str, int] = {}
for page in pages.values():
name = filters.collection_of(page.path, self.kb_dir)
if name:
collections[name] = collections.get(name, 0) + 1
return self._stamp({
"pages": len(pages),
"collections": dict(sorted(collections.items())),
"findings": {
key: len(value)
for key, value in sorted(report.items())
if isinstance(value, list)
},
"unreadable": unreadable_pages(pages),
}, revision)