feat: MCP-Leseserver, Bibliotheksgrenze, Haertung des Lesepfads, Publish-Remote-Gate scharf (2.4.0)
CI / verify (push) Successful in 52s
Release / release (push) Successful in 37s

Files changed:
- .gitea/workflows/ci.yml
- CHANGES.md
- README.md
- VERSION
- instructions/mcp-read-server.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/api.py
- tools/chemenu/commands/doctor.py
- tools/chemenu/commands/lint.py
- tools/chemenu/commands/search.py
- tools/chemenu/commands/types_cmd.py
- tools/chemenu/config.py
- tools/chemenu/corpus_cache.py
- tools/chemenu/errors.py
- tools/chemenu/frontmatter_io.py
- tools/chemenu/lint_core.py
- tools/chemenu/mcp/__init__.py
- tools/chemenu/mcp/__main__.py
- tools/chemenu/mcp/server.py
- tools/chemenu/page.py
- tools/chemenu/search/filters.py
- tools/chemenu/search/registry.py
- tools/chemenu/search/ripgrep.py
- tools/chemenu/search/service.py
- tools/chemenu/tests/conftest.py
- tools/chemenu/tests/test_api.py
- tools/chemenu/tests/test_corpus_cache.py
- tools/chemenu/tests/test_doctor.py
- tools/chemenu/tests/test_frontmatter_io.py
- tools/chemenu/tests/test_instructions_cmd.py
- tools/chemenu/tests/test_mcp_server.py
- tools/chemenu/tests/test_new_page.py
- tools/chemenu/tests/test_search.py
- tools/chemenu/type_resolver.py
- tools/chemenu/types_core.py
- tools/requirements-mcp.txt
This commit is contained in:
torben committed 2026-09-02 07:19:32 +02:00
1 parent d1cf2e0327
commit 576df2cddd
37 files changed
+3037 -625

No files matched your search

+30 -71
View File
@@ -21,90 +21,38 @@ Scope is `kb/` only. `instructions/` is discovered through
from __future__ import annotations
import json
from pathlib import Path
import typer
from chemenu import config
from chemenu.commands._util import fail, today_iso
from chemenu.frontmatter_io import read_page
from chemenu.kb_scan import iter_kb_pages
from chemenu.page import Page
from chemenu.search import filters
from chemenu.search.base import page_key
from chemenu.search.filters import PredicateError
from chemenu.search.fuse import reciprocal_rank_fusion
from chemenu.search.registry import UnknownBackend, resolve
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing, build_hit
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing
from chemenu.search.service import (
load_pages_by_path,
run_search,
sort_hits,
unreadable_pages,
)
from chemenu.search.types import Predicate, SearchHit, SearchQuery
# Re-exported so `from chemenu.commands.search import run_search` keeps
# resolving. The core lives in `chemenu/search/service.py`, which imports no
# CLI machinery; this module is the terminal adapter over it.
__all__ = [
"load_pages_by_path",
"run_search",
"sort_hits",
"unreadable_pages",
"render_table",
"search_command",
]
TITLE_WIDTH = 34
SUMMARY_WIDTH = 84
def load_pages_by_path(kb_dir: Path | None = None, root: Path | None = None) -> dict[str, Page]:
"""Every page under `kb/`, keyed by repo-relative path.
Path-keyed rather than title-keyed on purpose: `load_kb_pages()` drops one
of two pages sharing a stem, and search should still find both - a
duplicate title is a lint finding, not a reason to hide a page.
"""
kb_dir = kb_dir or config.KB_DIR
root = root or config.ROOT
pages: dict[str, Page] = {}
for path in iter_kb_pages(kb_dir):
frontmatter, body = read_page(path)
pages[page_key(path, root)] = Page(path=path, frontmatter=frontmatter, body=body)
return pages
def _sort_key(hit: SearchHit, field: str):
value = hit.as_dict().get(field)
if value is None:
# Missing values sort last in either direction rather than crashing on
# a None comparison.
return (1, "")
if isinstance(value, (int, float)):
return (0, value)
return (0, str(value).lower())
def sort_hits(hits: list[SearchHit], sort: str | None) -> list[SearchHit]:
"""Sort by a hit field. A leading `-` reverses, e.g. `--sort -confidence`."""
if not sort:
return hits
descending = sort.startswith("-")
field = sort.lstrip("-")
ordered = sorted(hits, key=lambda h: _sort_key(h, field), reverse=descending)
return ordered
def run_search(
query: SearchQuery,
pages: dict[str, Page],
backends: list,
kb_dir: Path | None = None,
) -> list[SearchHit]:
"""Answer a query. Pure: no I/O beyond whatever a backend does."""
filters.validate_fields(query.predicates, pages)
if query.text:
rankings = [backend.search(query, pages) for backend in backends]
hits = rankings[0] if len(rankings) == 1 else reciprocal_rank_fusion(rankings)
allowed = filters.apply_predicates(pages, query.predicates, kb_dir)
hits = [hit for hit in hits if hit.path in allowed]
else:
selected = filters.apply_predicates(pages, query.predicates, kb_dir)
hits = [
build_hit(page, key, [], query, backend="frontmatter", kb_dir=kb_dir)
for key, page in selected.items()
]
hits.sort(key=lambda h: h.title.lower())
hits = sort_hits(hits, query.sort)
return hits[: query.limit] if query.limit else hits
def _truncate(text: str, width: int) -> str:
text = " ".join(text.split())
return text if len(text) <= width else text[: width - 1] + "\u2026"
@@ -206,6 +154,8 @@ def search_command(
except RipgrepFailed as exc:
fail(str(exc))
unreadable = unreadable_pages(pages)
if json_out:
payload = {
"generated": today_iso(),
@@ -214,8 +164,17 @@ def search_command(
"backend": ",".join(b.name for b in backends),
"count": len(hits),
"results": [hit.as_dict() for hit in hits],
# Always present, usually empty. A caller that has to look for the
# key to learn whether it should worry will not look.
"unreadable": unreadable,
}
typer.echo(json.dumps(payload, indent=2))
return
typer.echo(render_table(hits, show_matches))
for entry in unreadable:
typer.echo(
f"WARN unreadable frontmatter: {entry['path']} ({entry['reason']}) - "
"this page cannot match any --field predicate",
err=True,
)