search: Pfad und Titel vollstaendig in der Trefferzeile, Trunkierung wird benannt (schliesst #100)
CI / verify (push) Failing after 40s
Release / release (push) Successful in 36s

Files changed:
- AGENTS.md
- CHANGES.md
- VERSION
- instructions/wiki-query/SKILL.md
- tools/CONTRACT.md
- tools/chemenu/api.py
- tools/chemenu/commands/search.py
- tools/chemenu/mcp/server.py
- tools/chemenu/search/service.py
- tools/chemenu/search/types.py
- tools/chemenu/tests/test_api.py
- tools/chemenu/tests/test_mcp_server.py
- tools/chemenu/tests/test_search.py
This commit is contained in:
torben committed 2026-09-15 21:26:39 +02:00
1 parent 55f65c1ab1
commit bb097f614b
13 files changed
+352 -41

No files matched your search

+69 -14
View File
@@ -35,7 +35,7 @@ from chemenu.search.service import (
sort_hits,
unreadable_pages,
)
from chemenu.search.types import Predicate, SearchHit, SearchQuery
from chemenu.search.types import DEFAULT_LIMIT, Predicate, SearchQuery, SearchResult
# Re-exported so `from chemenu.commands.search import run_search` keeps
# resolving. The core lives in `chemenu/search/service.py`, which imports no
@@ -49,32 +49,76 @@ __all__ = [
"search_command",
]
TITLE_WIDTH = 34
SUMMARY_WIDTH = 84
# One hit per line, ` | `-separated, in the order score, kind, title, path,
# summary. Three properties are load-bearing and should survive any edit here:
#
# 1. **The path is present.** It was not, and the instructions that drive this
# command tell an agent to "read only the pages the search points at" - which
# it could not do, because nothing here pointed anywhere. What a session did
# instead was run `grep -rl` over `kb/` for the filenames, a second search
# that can find no page this one missed (the backend *is* `rg` over `kb/`).
# 2. **Title and path are never truncated.** The title is the wiki's only
# identifier for a page (AGENTS.md invariant 2) and the argument `xref add`,
# `cite add` and `touch` all take; a title clipped to a column width is not
# one. The old fixed 34-char field clipped four of five hits in the report
# that prompted this. Only the summary is lossy, which is why it goes last.
# 3. **The separator is unambiguous.** A `|` cannot occur in a title - the
# wikilink syntax reserves it, so a page carrying one could not be linked at
# all - and a `|` in the summary is harmless, because the summary is the
# final field: split on " | " with maxsplit=4 and prose cannot shift a
# column.
#
# Column padding is gone with the widths: it aligned the table for an eye, and
# the reader here is an agent that pays for the spaces by the token.
SEPARATOR = " | "
def _truncate(text: str, width: int) -> str:
text = " ".join(text.split())
return text if len(text) <= width else text[: width - 1] + "\u2026"
def render_table(hits: list[SearchHit], show_matches: bool) -> str:
if not hits:
def _count_line(result: SearchResult) -> str:
"""The last line: how many hits, and whether that is all of them.
A bare `N result(s).` reads as the whole answer, so it is only used when it
is one. A capped search says what it capped, which is the number the caller
would otherwise have to run a second, unlimited search to learn.
"""
if not result.truncated:
return f"{len(result.hits)} result(s)."
return (
f"{len(result.hits)} of {result.total} result(s) - "
f"raise --limit (0 for all) or narrow the query."
)
def render_table(result: SearchResult, show_matches: bool) -> str:
if not result.hits:
return "No matches."
lines = []
for hit in hits:
for hit in result.hits:
kind = hit.kind or "?"
if hit.subtype:
kind = f"{kind}/{hit.subtype}"
lines.append(
f"{hit.score:6.1f} {_truncate(hit.title, TITLE_WIDTH):<{TITLE_WIDTH}} "
f"{kind:<18} {_truncate(hit.summary, SUMMARY_WIDTH)}"
SEPARATOR.join(
(
f"{hit.score:.1f}",
kind,
hit.title,
hit.path,
_truncate(hit.summary, SUMMARY_WIDTH),
)
)
)
if show_matches:
for match in hit.matches:
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
lines.append(f" {hit.path}:{match.line}: {_truncate(match.text, 100)}")
lines.append("")
lines.append(f"{len(hits)} result(s).")
lines.append(_count_line(result))
return "\n".join(lines)
@@ -99,7 +143,11 @@ def search_command(
regex: bool = typer.Option(
False, "--regex", help="Treat the query as a regex. Off by default: terms are literal."
),
limit: int = typer.Option(20, "--limit", help="Maximum number of results. 0 for no limit."),
limit: int = typer.Option(
DEFAULT_LIMIT,
"--limit",
help="Maximum number of results. 0 for no limit. A capped result says so.",
),
sort: str = typer.Option(
None, "--sort", help="Sort by a result field; prefix with '-' to reverse, e.g. -modified."
),
@@ -146,7 +194,7 @@ def search_command(
pages = load_pages_by_path()
try:
hits = run_search(query, pages, backends)
result = run_search(query, pages, backends)
except PredicateError as exc:
fail(str(exc))
except RipgrepMissing as exc:
@@ -162,8 +210,15 @@ def search_command(
"query": text,
"predicates": [p.render() for p in predicates],
"backend": ",".join(b.name for b in backends),
"count": len(hits),
"results": [hit.as_dict() for hit in hits],
# `count` keeps its meaning - how many results are in this payload -
# so a consumer written against the old shape reads the same number
# it always did. `total`/`truncated`/`limit` are what it could not
# ask before.
"count": len(result.hits),
"total": result.total,
"truncated": result.truncated,
"limit": result.limit,
"results": [hit.as_dict() for hit in result.hits],
# Always present, usually empty. A caller that has to look for the
# key to learn whether it should worry will not look.
"unreadable": unreadable,
@@ -171,7 +226,7 @@ def search_command(
typer.echo(json.dumps(payload, indent=2))
return
typer.echo(render_table(hits, show_matches))
typer.echo(render_table(result, show_matches))
for entry in unreadable:
typer.echo(
f"WARN unreadable frontmatter: {entry['path']} ({entry['reason']}) - "