54d9540c08
Files changed: - .wikitool-kb.json - AGENTS.md - CHANGES.md - INSTALL-MCP.md - INSTALL.md - README.md - VERSION - instructions/capture-session.md - instructions/dev/issue-tracking.md - instructions/german-terminology.md - instructions/kb-profiles.md - instructions/migrate-corpus.md - instructions/migrations/5.0.0-confidence-removal.md - instructions/private-instance.md - instructions/setup-instance.md - instructions/wiki-lint/SKILL.md - instructions/wiki-manage/SKILL.md - instructions/wiki-query/SKILL.md - kb/CONTRACT.md - kb/CONVENTIONS.md - kb/CONVENTIONS.md.template - kb/concepts/architectures/Consolidation Tiers.md - kb/concepts/architectures/Context Isolation.md - kb/concepts/architectures/Cross-platform Agent Skills.md - kb/concepts/architectures/Episodic Memory.md - kb/concepts/architectures/Hybrid Search.md - kb/concepts/architectures/Implementation Spectrum.md - kb/concepts/architectures/Knowledge Graph.md - kb/concepts/architectures/LLM Wiki Pattern.md - kb/concepts/architectures/MCP-Leseserver.md - kb/concepts/architectures/Memory Lifecycle.md - kb/concepts/architectures/OKF Compatibility.md - kb/concepts/architectures/Optional Instance Context File.md - kb/concepts/architectures/Personalization Plane.md - kb/concepts/architectures/Procedural Memory.md - kb/concepts/architectures/RAG.md - kb/concepts/architectures/Scale Ceiling.md - kb/concepts/architectures/Semantic Memory.md - kb/concepts/architectures/Three-Layer Architecture.md - kb/concepts/architectures/Token Economics.md - kb/concepts/architectures/Working Memory.md - kb/concepts/decisions/Delete Rather Than Anonymize.md - kb/concepts/decisions/Denylist over Allowlist.md - kb/concepts/decisions/Diff-Reviewable Agent Edits.md - kb/concepts/decisions/Dual Licensing by File Plan.md - kb/concepts/decisions/Issue Label Scheme.md - kb/concepts/decisions/KB Stack Versioning.md - kb/concepts/decisions/Structural Enforcement over Documented Rule.md - kb/concepts/patterns/Audit Trail.md - kb/concepts/patterns/BM25.md - kb/concepts/patterns/Command Round-Trip Integrity.md - kb/concepts/patterns/Confidence Scoring.md - kb/concepts/patterns/Contradiction Resolution.md - kb/concepts/patterns/Entity Extraction.md - kb/concepts/patterns/Filter on Ingest.md - kb/concepts/patterns/Forgetting.md - kb/concepts/patterns/Graph Traversal.md - kb/concepts/patterns/Mesh Sync.md - kb/concepts/patterns/Quality Scoring.md - kb/concepts/patterns/Reciprocal Rank Fusion.md - kb/concepts/patterns/Self-Healing.md - kb/concepts/patterns/Shared vs Private.md - kb/concepts/patterns/Typed Relationships.md - kb/concepts/patterns/Vector Search.md - kb/concepts/patterns/Work Coordination.md - kb/concepts/problems/Ambient Environment Dependency.md - kb/concepts/problems/Detect-Repair Asymmetry.md - kb/concepts/problems/Green Suite Blind Spot.md - kb/concepts/problems/Naming Convention Conflict.md - kb/concepts/problems/Write-Once Frontmatter Fields.md - kb/concepts/protocols/CPPC.md - kb/concepts/protocols/Modbus.md - kb/concepts/protocols/SSD TRIM.md - kb/concepts/workflows/Anti-Cramming Heuristic.md - kb/concepts/workflows/Bulk Operations.md - kb/concepts/workflows/CI Integration.md - kb/concepts/workflows/Checkpoint Audit.md - kb/concepts/workflows/Claude Code Auto Mode.md - kb/concepts/workflows/Content Quality Control.md - kb/concepts/workflows/Crystallization.md - kb/concepts/workflows/Event-Driven Automation.md - kb/concepts/workflows/Hooks.md - kb/concepts/workflows/Index Scaling.md - kb/concepts/workflows/Iteration and Cost Limits.md - kb/concepts/workflows/KB Migration.md - kb/concepts/workflows/Knowledge Compounding.md - kb/concepts/workflows/Lint Workflow.md - kb/concepts/workflows/Mass-Update Gate.md - kb/concepts/workflows/Multi-Agent Collaboration.md - kb/concepts/workflows/Privacy and Governance.md - kb/concepts/workflows/Publish-Remote Gate.md - kb/concepts/workflows/Quality and Self-Correction.md - kb/concepts/workflows/Semantic Lint Automation.md - kb/concepts/workflows/Session Orientation.md - kb/concepts/workflows/Split Merge Reclassify.md - kb/concepts/workflows/Split Threshold.md - kb/concepts/workflows/Stub Threshold.md - kb/concepts/workflows/Supersession.md - kb/concepts/workflows/User Management.md - kb/concepts/workflows/Workflow Extraction.md - kb/concepts/workflows/Workflow Orchestration.md - kb/entities/people/Andrej Karpathy.md - kb/entities/people/E3DC GmbH.md - kb/entities/people/Rohit Gupta.md - kb/entities/people/Vannevar Bush.md - kb/entities/projects/BCDModule.md - kb/entities/projects/Chemenu.md - kb/entities/projects/andybalholm-edl.md - kb/entities/projects/goresponsiveness.md - kb/entities/projects/ha-core.md - kb/entities/projects/hacs-e3dc.md - kb/entities/projects/hacs-integration-blueprint.md - kb/entities/projects/llm-wiki-skills.md - kb/entities/projects/plugnburn-edl.md - kb/entities/projects/wiki-skills-vanillaflava.md - kb/entities/projects/wiki-skills.md - kb/entities/systems/AGENTS.md.md - kb/entities/systems/CLAUDE.md.md - kb/entities/systems/E3DC.md - kb/entities/systems/ENVIRONMENT.md.md - kb/entities/systems/Memex.md - kb/entities/systems/Tolkien Gateway.md - kb/entities/technologies/Arch Linux.md - kb/entities/technologies/Disk Encryption.md - kb/entities/technologies/Docker.md - kb/entities/technologies/GRUB.md - kb/entities/technologies/Gitea Actions.md - kb/entities/technologies/Gitea.md - kb/entities/technologies/Go.md - kb/entities/technologies/Home Assistant.md - kb/entities/technologies/Kernel PM Governors.md - kb/entities/technologies/LVM.md - kb/entities/technologies/Linux Kernel.md - kb/entities/technologies/MQTT.md - kb/entities/technologies/OPC UA.md - kb/entities/technologies/Python.md - kb/entities/technologies/Rust.md - kb/entities/technologies/Wine GE.md - kb/entities/technologies/Wine-Staging.md - kb/entities/technologies/acpi-cpufreq.md - kb/entities/technologies/amd-pstate.md - kb/entities/technologies/iii Engine.md - kb/entities/tools/AUR.md - kb/entities/tools/Act Runner.md - kb/entities/tools/Agent Memory.md - kb/entities/tools/Aura.md - kb/entities/tools/Bottles.md - kb/entities/tools/ChatGPT.md - kb/entities/tools/Claude Code.md - kb/entities/tools/Codex CLI.md - kb/entities/tools/Dataview.md - kb/entities/tools/GPG.md - kb/entities/tools/GitHub Copilot.md - kb/entities/tools/Gitea MCP Server.md - kb/entities/tools/Lutris.md - kb/entities/tools/Marp.md - kb/entities/tools/Mistral Vibe.md - kb/entities/tools/NotebookLM.md - kb/entities/tools/Obsidian Web Clipper.md - kb/entities/tools/Obsidian.md - kb/entities/tools/OpenAI Codex.md - kb/entities/tools/OpenCode.md - kb/entities/tools/Pi.md - kb/entities/tools/Proton.md - kb/entities/tools/Steam.md - kb/entities/tools/Wine.md - kb/entities/tools/awesome-llm-wiki.md - kb/entities/tools/farzaa gist.md - kb/entities/tools/gdeploy.md - kb/entities/tools/makepkg.md - kb/entities/tools/pascalandy schema.md - kb/entities/tools/qmd.md - kb/entities/tools/wikitool.md - kb/index.md - kb/log.md - raw/CONTRACT.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/api.py - tools/chemenu/cli.py - tools/chemenu/commands/confidence_decay.py - tools/chemenu/commands/docs_verify.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/index_build.py - tools/chemenu/commands/new_page.py - tools/chemenu/commands/search.py - tools/chemenu/commands/touch.py - tools/chemenu/commands/version_cmd.py - tools/chemenu/conventions.py - tools/chemenu/corpus_diff.py - tools/chemenu/frontmatter_io.py - tools/chemenu/lint_core.py - tools/chemenu/mcp/server.py - tools/chemenu/page.py - tools/chemenu/search/base.py - tools/chemenu/search/filters.py - tools/chemenu/search/ripgrep.py - tools/chemenu/search/service.py - tools/chemenu/search/types.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_api.py - tools/chemenu/tests/test_confidence_decay.py - tools/chemenu/tests/test_corpus_diff.py - tools/chemenu/tests/test_docs_verify.py - tools/chemenu/tests/test_frontmatter_io.py - tools/chemenu/tests/test_index_build.py - tools/chemenu/tests/test_kb_scan.py - tools/chemenu/tests/test_lint.py - tools/chemenu/tests/test_mcp_server.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_page_ops.py - tools/chemenu/tests/test_provenance.py - tools/chemenu/tests/test_raw_cmd.py - tools/chemenu/tests/test_search.py - tools/chemenu/tests/test_touch.py - tools/chemenu/tests/test_type_resolver.py - tools/chemenu/tests/test_version_cmd.py - tools/chemenu/tests/test_xref.py - tools/chemenu/version.py - types/concept.md - types/concept.schema.yaml - types/entity.md - types/entity.schema.yaml - types/instruction.md - types/type-spec.md
232 lines
9.3 KiB
Python
232 lines
9.3 KiB
Python
"""The lexical backend: `rg` over `kb/`, parsed from its JSON output.
|
|
|
|
Why shell out instead of scanning in Python: `rg` is already the retrieval
|
|
layer the agent instructions point at, it handles large trees fast, and its
|
|
`--json` mode gives line numbers and matched text without reparsing files.
|
|
|
|
Three safety properties are load-bearing and must survive any edit here:
|
|
|
|
1. The query is passed as an *argv element*, never through a shell. There is
|
|
no `shell=True` anywhere in this module, so a query containing `;`, `$(...)`
|
|
or backticks is searched for literally rather than executed.
|
|
2. `--fixed-strings` is the default. A user-supplied regex is opt-in via
|
|
`--regex`, so an accidental `.*` in a search term is a literal, and a
|
|
pathological pattern cannot be introduced without asking for one.
|
|
3. A user-supplied pattern is evaluated **only** by `rg`, whose engine is
|
|
linear in the input. Nothing here hands it to Python's `re`, which
|
|
backtracks - see `_contains`.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import subprocess
|
|
from pathlib import Path
|
|
from typing import Iterable
|
|
|
|
from chemenu import config
|
|
from chemenu.errors import BackendError
|
|
from chemenu.page import Page
|
|
from chemenu.search.base import page_key
|
|
from chemenu.search.filters import collection_of
|
|
from chemenu.search.types import Match, SearchHit, SearchQuery
|
|
|
|
# Only the first few matching lines per page are kept. The hit is a pointer for
|
|
# deciding whether to open the page, not a substitute for reading it, and an
|
|
# unbounded excerpt list is exactly the token cost this command exists to avoid.
|
|
MAX_MATCHES_PER_PAGE = 3
|
|
|
|
# Ranking tiers. An *exact* title match is worth more than a title that merely
|
|
# contains the term, because "longhorn" should surface the page called Longhorn
|
|
# ahead of every page whose title mentions it - otherwise a well-connected
|
|
# source page outranks the subject it is about.
|
|
EXACT_TITLE_WEIGHT = 10.0
|
|
TITLE_WEIGHT = 5.0
|
|
SUMMARY_WEIGHT = 3.0
|
|
LINE_WEIGHT = 1.0
|
|
|
|
# A hang-breaker, not a performance budget. A fixed-string search over this
|
|
# corpus costs 6 ms and a deliberately broad regex 1.4 s, so nothing legitimate
|
|
# comes near this; it exists so that a pathological pattern, a corpus on a
|
|
# stalled network mount, or an `rg` that never returns fails as an error
|
|
# instead of holding the caller open forever while its output buffers into the
|
|
# heap. The caller sees the ordinary `RipgrepFailed` path.
|
|
RIPGREP_TIMEOUT_SECONDS = 30.0
|
|
|
|
|
|
class RipgrepMissing(BackendError):
|
|
"""Raised when the `rg` executable is not on PATH."""
|
|
|
|
|
|
class RipgrepFailed(BackendError):
|
|
"""Raised when `rg` exits with an error (exit code 2 or above), or had to
|
|
be killed for running past `RIPGREP_TIMEOUT_SECONDS`."""
|
|
|
|
|
|
def build_argv(query: SearchQuery, root: Path) -> list[str]:
|
|
"""The exact command line. Split out so a test can assert the safety
|
|
properties above without running anything."""
|
|
argv = ["rg", "--json", "--smart-case", "--glob", "*.md"]
|
|
if not query.regex:
|
|
argv.append("--fixed-strings")
|
|
# `--` terminates option parsing: a query starting with `-` is a search
|
|
# term, not a flag.
|
|
argv += ["--", query.text or "", str(root)]
|
|
return argv
|
|
|
|
|
|
def _iter_match_records(stdout: str) -> Iterable[dict]:
|
|
for line in stdout.splitlines():
|
|
if not line.strip():
|
|
continue
|
|
try:
|
|
record = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
continue
|
|
if record.get("type") == "match":
|
|
yield record.get("data", {})
|
|
|
|
|
|
def _record_path(data: dict) -> str | None:
|
|
path = data.get("path") or {}
|
|
return path.get("text")
|
|
|
|
|
|
class RipgrepBackend:
|
|
"""Lexical search over page bodies and frontmatter text."""
|
|
|
|
name = "rg"
|
|
|
|
def __init__(self, search_root: Path | None = None, repo_root: Path | None = None):
|
|
# Two roots, because they answer different questions: `search_root` is
|
|
# what rg walks, `repo_root` is what the resulting paths are made
|
|
# relative to so they match the keys in `pages`. They differ only in
|
|
# tests, but conflating them makes the backend untestable outside the
|
|
# real repo.
|
|
self.search_root = search_root or config.KB_DIR
|
|
self.repo_root = repo_root or config.ROOT
|
|
|
|
def search(self, query: SearchQuery, pages: dict[str, Page]) -> list[SearchHit]:
|
|
if not query.text:
|
|
return []
|
|
|
|
argv = build_argv(query, self.search_root)
|
|
try:
|
|
proc = subprocess.run(
|
|
argv,
|
|
capture_output=True,
|
|
text=True,
|
|
check=False,
|
|
timeout=RIPGREP_TIMEOUT_SECONDS,
|
|
)
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise RipgrepFailed(
|
|
f"rg did not finish within {RIPGREP_TIMEOUT_SECONDS:g}s and was killed. "
|
|
"A search that takes this long is a pathological pattern or an "
|
|
"unresponsive corpus directory, not a slow answer - narrow the query "
|
|
"or use a fixed string instead of --regex."
|
|
) from exc
|
|
except FileNotFoundError as exc: # pragma: no cover - depends on host
|
|
raise RipgrepMissing(
|
|
"ripgrep (rg) is not installed or not on PATH. It is the search "
|
|
"backend; install it (e.g. `pacman -S ripgrep`, `apt install ripgrep`) "
|
|
"and retry. wikitool deliberately has no Python fallback: a fallback "
|
|
"would answer differently from the documented backend without saying so."
|
|
) from exc
|
|
|
|
# rg exits 1 for "no matches found", which is an answer, not a failure.
|
|
if proc.returncode >= 2:
|
|
raise RipgrepFailed(f"rg exited {proc.returncode}: {proc.stderr.strip()}")
|
|
|
|
by_path: dict[str, list[Match]] = {}
|
|
for data in _iter_match_records(proc.stdout):
|
|
raw_path = _record_path(data)
|
|
if raw_path is None:
|
|
continue
|
|
key = page_key(Path(raw_path), self.repo_root)
|
|
if key not in pages:
|
|
# Not a page: a generated index, a contract, or a file outside
|
|
# the corpus. The pages dict is the authority on what exists.
|
|
continue
|
|
found = by_path.setdefault(key, [])
|
|
if len(found) >= MAX_MATCHES_PER_PAGE:
|
|
continue
|
|
found.append(
|
|
Match(
|
|
line=data.get("line_number", 0),
|
|
text=(data.get("lines", {}).get("text") or "").rstrip("\n"),
|
|
)
|
|
)
|
|
|
|
hits = [
|
|
build_hit(pages[key], key, matches, query, backend=self.name, kb_dir=self.search_root)
|
|
for key, matches in by_path.items()
|
|
]
|
|
hits.sort(key=lambda h: (-h.score, h.title.lower()))
|
|
return hits
|
|
|
|
|
|
def _contains(haystack: str, query: SearchQuery) -> bool:
|
|
"""Literal containment, used only for the title and summary ranking boosts.
|
|
|
|
It never evaluates `query.text` as a regex, even when `query.regex` is set.
|
|
It used to, via `re.search`, and Python's engine backtracks: `(\\w+\\s?)+$`
|
|
against 114 characters of ordinary page text does not terminate in eight
|
|
seconds, while a pattern that fails deterministically takes 0.2 ms - the
|
|
difference is the pattern, not the haystack. `build_hit` calls this twice
|
|
per hit, and a pattern as cheap as `\\w` matches every page, so one request
|
|
bought two unbounded searches per page in the corpus.
|
|
|
|
Deleting the branch rather than bounding it is the right trade: `rg` has
|
|
already applied the pattern with a linear engine by the time we get here,
|
|
and the page is a hit *because* of that. What is lost is only the extra
|
|
weight a regex hit in the title would have scored - and since a summary and
|
|
an H1 are themselves lines in the file, `rg` still counts them. A pattern
|
|
that is mostly literal (`longhorn`) still earns its boost through the test
|
|
below; one that is not gets ranked by match count alone.
|
|
"""
|
|
if not query.text:
|
|
return False
|
|
return query.text.lower() in haystack.lower()
|
|
|
|
|
|
def build_hit(
|
|
page: Page,
|
|
key: str,
|
|
matches: list[Match],
|
|
query: SearchQuery,
|
|
backend: str,
|
|
kb_dir: Path | None = None,
|
|
) -> SearchHit:
|
|
"""Turn a page plus its matching lines into an enriched, scored hit.
|
|
|
|
Ranking is deliberately crude and explainable: an exact title match
|
|
outweighs a partial one, which outweighs a summary match, which outweighs
|
|
body matches. An agent scanning results should be able to predict the
|
|
order, which a tuned scorer would not give.
|
|
"""
|
|
summary = str(page.frontmatter.get("summary") or "")
|
|
score = LINE_WEIGHT * len(matches)
|
|
if query.text and page.title.lower() == query.text.lower():
|
|
score += EXACT_TITLE_WEIGHT
|
|
elif _contains(page.title, query):
|
|
score += TITLE_WEIGHT
|
|
if _contains(summary, query):
|
|
score += SUMMARY_WEIGHT
|
|
|
|
tags = page.frontmatter.get("tags") or []
|
|
modified = page.frontmatter.get("modified") or page.frontmatter.get("date")
|
|
|
|
return SearchHit(
|
|
title=page.title,
|
|
path=key,
|
|
collection=collection_of(page.path, kb_dir),
|
|
kind=page.kind,
|
|
subtype=page.subtype,
|
|
summary=summary,
|
|
tags=[str(t) for t in tags] if isinstance(tags, (list, tuple)) else [str(tags)],
|
|
modified=str(modified) if modified else None,
|
|
score=score,
|
|
backend=backend,
|
|
matches=matches,
|
|
)
|