Files changed: - AGENTS.md - CHANGES.md - VERSION - instructions/wiki-query/SKILL.md - tools/CONTRACT.md - tools/chemenu/api.py - tools/chemenu/commands/search.py - tools/chemenu/mcp/server.py - tools/chemenu/search/service.py - tools/chemenu/search/types.py - tools/chemenu/tests/test_api.py - tools/chemenu/tests/test_mcp_server.py - tools/chemenu/tests/test_search.py
236 lines
8.6 KiB
Python
236 lines
8.6 KiB
Python
"""`wikitool search` - find pages without reading `kb/index.md`.
|
|
|
|
This command exists to make retrieval cheap. Before it, the documented way to
|
|
find a page was to read the whole generated index; at a few hundred pages that
|
|
is tens of thousands of tokens spent to learn three filenames. A search returns
|
|
the same pointers for a fraction of it.
|
|
|
|
Two halves, deliberately kept separate:
|
|
|
|
- Text search is answered by a pluggable backend (`rg` today) - see
|
|
`chemenu/search/`.
|
|
- Frontmatter predicates (`--field`) are evaluated here, in-process, on the
|
|
structured YAML rather than on its rendering. With no text at all this is a
|
|
pure structured query, which is how "systems with no sources, oldest
|
|
first" is asked without a second command.
|
|
|
|
Scope is `kb/` only. `instructions/` is discovered through
|
|
`wikitool instructions list`, because a procedure is found by what it is *for*
|
|
(its description), not by keywords in its body.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
|
|
import typer
|
|
|
|
from chemenu.commands._util import fail, today_iso
|
|
from chemenu.search import filters
|
|
from chemenu.search.filters import PredicateError
|
|
from chemenu.search.registry import UnknownBackend, resolve
|
|
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing
|
|
from chemenu.search.service import (
|
|
load_pages_by_path,
|
|
run_search,
|
|
sort_hits,
|
|
unreadable_pages,
|
|
)
|
|
from chemenu.search.types import DEFAULT_LIMIT, Predicate, SearchQuery, SearchResult
|
|
|
|
# Re-exported so `from chemenu.commands.search import run_search` keeps
|
|
# resolving. The core lives in `chemenu/search/service.py`, which imports no
|
|
# CLI machinery; this module is the terminal adapter over it.
|
|
__all__ = [
|
|
"load_pages_by_path",
|
|
"run_search",
|
|
"sort_hits",
|
|
"unreadable_pages",
|
|
"render_table",
|
|
"search_command",
|
|
]
|
|
|
|
SUMMARY_WIDTH = 84
|
|
|
|
# One hit per line, ` | `-separated, in the order score, kind, title, path,
|
|
# summary. Three properties are load-bearing and should survive any edit here:
|
|
#
|
|
# 1. **The path is present.** It was not, and the instructions that drive this
|
|
# command tell an agent to "read only the pages the search points at" - which
|
|
# it could not do, because nothing here pointed anywhere. What a session did
|
|
# instead was run `grep -rl` over `kb/` for the filenames, a second search
|
|
# that can find no page this one missed (the backend *is* `rg` over `kb/`).
|
|
# 2. **Title and path are never truncated.** The title is the wiki's only
|
|
# identifier for a page (AGENTS.md invariant 2) and the argument `xref add`,
|
|
# `cite add` and `touch` all take; a title clipped to a column width is not
|
|
# one. The old fixed 34-char field clipped four of five hits in the report
|
|
# that prompted this. Only the summary is lossy, which is why it goes last.
|
|
# 3. **The separator is unambiguous.** A `|` cannot occur in a title - the
|
|
# wikilink syntax reserves it, so a page carrying one could not be linked at
|
|
# all - and a `|` in the summary is harmless, because the summary is the
|
|
# final field: split on " | " with maxsplit=4 and prose cannot shift a
|
|
# column.
|
|
#
|
|
# Column padding is gone with the widths: it aligned the table for an eye, and
|
|
# the reader here is an agent that pays for the spaces by the token.
|
|
SEPARATOR = " | "
|
|
|
|
|
|
def _truncate(text: str, width: int) -> str:
|
|
text = " ".join(text.split())
|
|
return text if len(text) <= width else text[: width - 1] + "\u2026"
|
|
|
|
|
|
def _count_line(result: SearchResult) -> str:
|
|
"""The last line: how many hits, and whether that is all of them.
|
|
|
|
A bare `N result(s).` reads as the whole answer, so it is only used when it
|
|
is one. A capped search says what it capped, which is the number the caller
|
|
would otherwise have to run a second, unlimited search to learn.
|
|
"""
|
|
if not result.truncated:
|
|
return f"{len(result.hits)} result(s)."
|
|
return (
|
|
f"{len(result.hits)} of {result.total} result(s) - "
|
|
f"raise --limit (0 for all) or narrow the query."
|
|
)
|
|
|
|
|
|
def render_table(result: SearchResult, show_matches: bool) -> str:
|
|
if not result.hits:
|
|
return "No matches."
|
|
lines = []
|
|
for hit in result.hits:
|
|
kind = hit.kind or "?"
|
|
if hit.subtype:
|
|
kind = f"{kind}/{hit.subtype}"
|
|
lines.append(
|
|
SEPARATOR.join(
|
|
(
|
|
f"{hit.score:.1f}",
|
|
kind,
|
|
hit.title,
|
|
hit.path,
|
|
_truncate(hit.summary, SUMMARY_WIDTH),
|
|
)
|
|
)
|
|
)
|
|
if show_matches:
|
|
for match in hit.matches:
|
|
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
|
|
lines.append("")
|
|
lines.append(_count_line(result))
|
|
return "\n".join(lines)
|
|
|
|
|
|
def search_command(
|
|
text: str = typer.Argument(
|
|
None,
|
|
help="Text to search for. Omit it to run a pure frontmatter query.",
|
|
),
|
|
field: list[str] = typer.Option(
|
|
None,
|
|
"--field",
|
|
"-f",
|
|
help="Frontmatter predicate, repeatable (AND). Forms: field=value, "
|
|
"field~substring, 'field>=value', 'field:*' (present), '!field' (absent).",
|
|
),
|
|
kind: str = typer.Option(None, "--kind", help="Shorthand for --field kind=<value>."),
|
|
subtype: str = typer.Option(None, "--subtype", help="Shorthand for --field subtype=<value>."),
|
|
collection: str = typer.Option(
|
|
None, "--collection", help="Shorthand for --field collection=<value>."
|
|
),
|
|
tag: str = typer.Option(None, "--tag", help="Shorthand for --field tags=<value>."),
|
|
regex: bool = typer.Option(
|
|
False, "--regex", help="Treat the query as a regex. Off by default: terms are literal."
|
|
),
|
|
limit: int = typer.Option(
|
|
DEFAULT_LIMIT,
|
|
"--limit",
|
|
help="Maximum number of results. 0 for no limit. A capped result says so.",
|
|
),
|
|
sort: str = typer.Option(
|
|
None, "--sort", help="Sort by a result field; prefix with '-' to reverse, e.g. -modified."
|
|
),
|
|
backend: str = typer.Option(
|
|
None,
|
|
"--backend",
|
|
help="Search backend(s), comma-separated. Default 'rg' (or $WIKITOOL_SEARCH_BACKEND).",
|
|
),
|
|
show_matches: bool = typer.Option(
|
|
False, "--matches", help="Print the matching lines under each result."
|
|
),
|
|
json_out: bool = typer.Option(False, "--json", help="Print the results as JSON."),
|
|
):
|
|
"""Search kb/ by text, by frontmatter, or by both."""
|
|
raw_predicates = list(field or [])
|
|
for value, name in ((kind, "kind"), (subtype, "subtype"), (collection, "collection")):
|
|
if value:
|
|
raw_predicates.append(f"{name}={value}")
|
|
if tag:
|
|
raw_predicates.append(f"tags={tag}")
|
|
|
|
if not text and not raw_predicates:
|
|
fail("Nothing to search for: give a query, or at least one --field predicate.")
|
|
|
|
try:
|
|
predicates: tuple[Predicate, ...] = tuple(
|
|
filters.parse_predicate(raw) for raw in raw_predicates
|
|
)
|
|
except PredicateError as exc:
|
|
fail(str(exc))
|
|
|
|
try:
|
|
backends = resolve(backend)
|
|
except UnknownBackend as exc:
|
|
fail(str(exc))
|
|
|
|
query = SearchQuery(
|
|
text=text,
|
|
predicates=predicates,
|
|
regex=regex,
|
|
limit=limit,
|
|
sort=sort,
|
|
)
|
|
|
|
pages = load_pages_by_path()
|
|
try:
|
|
result = run_search(query, pages, backends)
|
|
except PredicateError as exc:
|
|
fail(str(exc))
|
|
except RipgrepMissing as exc:
|
|
fail(str(exc))
|
|
except RipgrepFailed as exc:
|
|
fail(str(exc))
|
|
|
|
unreadable = unreadable_pages(pages)
|
|
|
|
if json_out:
|
|
payload = {
|
|
"generated": today_iso(),
|
|
"query": text,
|
|
"predicates": [p.render() for p in predicates],
|
|
"backend": ",".join(b.name for b in backends),
|
|
# `count` keeps its meaning - how many results are in this payload -
|
|
# so a consumer written against the old shape reads the same number
|
|
# it always did. `total`/`truncated`/`limit` are what it could not
|
|
# ask before.
|
|
"count": len(result.hits),
|
|
"total": result.total,
|
|
"truncated": result.truncated,
|
|
"limit": result.limit,
|
|
"results": [hit.as_dict() for hit in result.hits],
|
|
# Always present, usually empty. A caller that has to look for the
|
|
# key to learn whether it should worry will not look.
|
|
"unreadable": unreadable,
|
|
}
|
|
typer.echo(json.dumps(payload, indent=2))
|
|
return
|
|
|
|
typer.echo(render_table(result, show_matches))
|
|
for entry in unreadable:
|
|
typer.echo(
|
|
f"WARN unreadable frontmatter: {entry['path']} ({entry['reason']}) - "
|
|
"this page cannot match any --field predicate",
|
|
err=True,
|
|
)
|