Files
chemenu/tools/chemenu/commands/search.py
T
torben bb097f614b
CI / verify (push) Failing after 40s
Release / release (push) Successful in 36s
search: Pfad und Titel vollstaendig in der Trefferzeile, Trunkierung wird benannt (schliesst #100)
Files changed:
- AGENTS.md
- CHANGES.md
- VERSION
- instructions/wiki-query/SKILL.md
- tools/CONTRACT.md
- tools/chemenu/api.py
- tools/chemenu/commands/search.py
- tools/chemenu/mcp/server.py
- tools/chemenu/search/service.py
- tools/chemenu/search/types.py
- tools/chemenu/tests/test_api.py
- tools/chemenu/tests/test_mcp_server.py
- tools/chemenu/tests/test_search.py
2026-09-15 21:26:39 +02:00

236 lines
8.6 KiB
Python

"""`wikitool search` - find pages without reading `kb/index.md`.
This command exists to make retrieval cheap. Before it, the documented way to
find a page was to read the whole generated index; at a few hundred pages that
is tens of thousands of tokens spent to learn three filenames. A search returns
the same pointers for a fraction of it.
Two halves, deliberately kept separate:
- Text search is answered by a pluggable backend (`rg` today) - see
`chemenu/search/`.
- Frontmatter predicates (`--field`) are evaluated here, in-process, on the
structured YAML rather than on its rendering. With no text at all this is a
pure structured query, which is how "systems with no sources, oldest
first" is asked without a second command.
Scope is `kb/` only. `instructions/` is discovered through
`wikitool instructions list`, because a procedure is found by what it is *for*
(its description), not by keywords in its body.
"""
from __future__ import annotations
import json
import typer
from chemenu.commands._util import fail, today_iso
from chemenu.search import filters
from chemenu.search.filters import PredicateError
from chemenu.search.registry import UnknownBackend, resolve
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing
from chemenu.search.service import (
load_pages_by_path,
run_search,
sort_hits,
unreadable_pages,
)
from chemenu.search.types import DEFAULT_LIMIT, Predicate, SearchQuery, SearchResult
# Re-exported so `from chemenu.commands.search import run_search` keeps
# resolving. The core lives in `chemenu/search/service.py`, which imports no
# CLI machinery; this module is the terminal adapter over it.
__all__ = [
"load_pages_by_path",
"run_search",
"sort_hits",
"unreadable_pages",
"render_table",
"search_command",
]
SUMMARY_WIDTH = 84
# One hit per line, ` | `-separated, in the order score, kind, title, path,
# summary. Three properties are load-bearing and should survive any edit here:
#
# 1. **The path is present.** It was not, and the instructions that drive this
# command tell an agent to "read only the pages the search points at" - which
# it could not do, because nothing here pointed anywhere. What a session did
# instead was run `grep -rl` over `kb/` for the filenames, a second search
# that can find no page this one missed (the backend *is* `rg` over `kb/`).
# 2. **Title and path are never truncated.** The title is the wiki's only
# identifier for a page (AGENTS.md invariant 2) and the argument `xref add`,
# `cite add` and `touch` all take; a title clipped to a column width is not
# one. The old fixed 34-char field clipped four of five hits in the report
# that prompted this. Only the summary is lossy, which is why it goes last.
# 3. **The separator is unambiguous.** A `|` cannot occur in a title - the
# wikilink syntax reserves it, so a page carrying one could not be linked at
# all - and a `|` in the summary is harmless, because the summary is the
# final field: split on " | " with maxsplit=4 and prose cannot shift a
# column.
#
# Column padding is gone with the widths: it aligned the table for an eye, and
# the reader here is an agent that pays for the spaces by the token.
SEPARATOR = " | "
def _truncate(text: str, width: int) -> str:
text = " ".join(text.split())
return text if len(text) <= width else text[: width - 1] + "\u2026"
def _count_line(result: SearchResult) -> str:
"""The last line: how many hits, and whether that is all of them.
A bare `N result(s).` reads as the whole answer, so it is only used when it
is one. A capped search says what it capped, which is the number the caller
would otherwise have to run a second, unlimited search to learn.
"""
if not result.truncated:
return f"{len(result.hits)} result(s)."
return (
f"{len(result.hits)} of {result.total} result(s) - "
f"raise --limit (0 for all) or narrow the query."
)
def render_table(result: SearchResult, show_matches: bool) -> str:
if not result.hits:
return "No matches."
lines = []
for hit in result.hits:
kind = hit.kind or "?"
if hit.subtype:
kind = f"{kind}/{hit.subtype}"
lines.append(
SEPARATOR.join(
(
f"{hit.score:.1f}",
kind,
hit.title,
hit.path,
_truncate(hit.summary, SUMMARY_WIDTH),
)
)
)
if show_matches:
for match in hit.matches:
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
lines.append("")
lines.append(_count_line(result))
return "\n".join(lines)
def search_command(
text: str = typer.Argument(
None,
help="Text to search for. Omit it to run a pure frontmatter query.",
),
field: list[str] = typer.Option(
None,
"--field",
"-f",
help="Frontmatter predicate, repeatable (AND). Forms: field=value, "
"field~substring, 'field>=value', 'field:*' (present), '!field' (absent).",
),
kind: str = typer.Option(None, "--kind", help="Shorthand for --field kind=<value>."),
subtype: str = typer.Option(None, "--subtype", help="Shorthand for --field subtype=<value>."),
collection: str = typer.Option(
None, "--collection", help="Shorthand for --field collection=<value>."
),
tag: str = typer.Option(None, "--tag", help="Shorthand for --field tags=<value>."),
regex: bool = typer.Option(
False, "--regex", help="Treat the query as a regex. Off by default: terms are literal."
),
limit: int = typer.Option(
DEFAULT_LIMIT,
"--limit",
help="Maximum number of results. 0 for no limit. A capped result says so.",
),
sort: str = typer.Option(
None, "--sort", help="Sort by a result field; prefix with '-' to reverse, e.g. -modified."
),
backend: str = typer.Option(
None,
"--backend",
help="Search backend(s), comma-separated. Default 'rg' (or $WIKITOOL_SEARCH_BACKEND).",
),
show_matches: bool = typer.Option(
False, "--matches", help="Print the matching lines under each result."
),
json_out: bool = typer.Option(False, "--json", help="Print the results as JSON."),
):
"""Search kb/ by text, by frontmatter, or by both."""
raw_predicates = list(field or [])
for value, name in ((kind, "kind"), (subtype, "subtype"), (collection, "collection")):
if value:
raw_predicates.append(f"{name}={value}")
if tag:
raw_predicates.append(f"tags={tag}")
if not text and not raw_predicates:
fail("Nothing to search for: give a query, or at least one --field predicate.")
try:
predicates: tuple[Predicate, ...] = tuple(
filters.parse_predicate(raw) for raw in raw_predicates
)
except PredicateError as exc:
fail(str(exc))
try:
backends = resolve(backend)
except UnknownBackend as exc:
fail(str(exc))
query = SearchQuery(
text=text,
predicates=predicates,
regex=regex,
limit=limit,
sort=sort,
)
pages = load_pages_by_path()
try:
result = run_search(query, pages, backends)
except PredicateError as exc:
fail(str(exc))
except RipgrepMissing as exc:
fail(str(exc))
except RipgrepFailed as exc:
fail(str(exc))
unreadable = unreadable_pages(pages)
if json_out:
payload = {
"generated": today_iso(),
"query": text,
"predicates": [p.render() for p in predicates],
"backend": ",".join(b.name for b in backends),
# `count` keeps its meaning - how many results are in this payload -
# so a consumer written against the old shape reads the same number
# it always did. `total`/`truncated`/`limit` are what it could not
# ask before.
"count": len(result.hits),
"total": result.total,
"truncated": result.truncated,
"limit": result.limit,
"results": [hit.as_dict() for hit in result.hits],
# Always present, usually empty. A caller that has to look for the
# key to learn whether it should worry will not look.
"unreadable": unreadable,
}
typer.echo(json.dumps(payload, indent=2))
return
typer.echo(render_table(result, show_matches))
for entry in unreadable:
typer.echo(
f"WARN unreadable frontmatter: {entry['path']} ({entry['reason']}) - "
"this page cannot match any --field predicate",
err=True,
)