Files
chemenu/tools/chemenu/corpus_cache.py
T
torben d0f08d1fba
CI / verify (push) Successful in 2m42s
CI / pwsh (push) Successful in 1m55s
Release / release (push) Successful in 36s
feat: Windows portability - path separators, LF line endings, UTF-8 decoding and output, msvcrt lock fallback (#152)
Files changed:
- .gitattributes
- CHANGES.md
- VERSION
- raw/CONTRACT.md
- tools/README.md
- tools/chemenu/commands/_util.py
- tools/chemenu/commands/dist_cmd.py
- tools/chemenu/commands/docs_verify.py
- tools/chemenu/commands/doctor.py
- tools/chemenu/commands/eval_cmd.py
- tools/chemenu/commands/git_publish.py
- tools/chemenu/commands/index_build.py
- tools/chemenu/commands/lint.py
- tools/chemenu/commands/log_append.py
- tools/chemenu/commands/migrate_cmd.py
- tools/chemenu/commands/provenance_cmd.py
- tools/chemenu/commands/raw_cmd.py
- tools/chemenu/commands/run_budget.py
- tools/chemenu/commands/upstream_cmd.py
- tools/chemenu/commands/version_cmd.py
- tools/chemenu/commands/work_cmd.py
- tools/chemenu/config.py
- tools/chemenu/corpus_cache.py
- tools/chemenu/filelock.py
- tools/chemenu/frontmatter_io.py
- tools/chemenu/kb_scan.py
- tools/chemenu/kb_state.py
- tools/chemenu/lint_core.py
- tools/chemenu/prerequisites.py
- tools/chemenu/provenance.py
- tools/chemenu/search/base.py
- tools/chemenu/search/ripgrep.py
- tools/chemenu/telemetry/writer.py
- tools/chemenu/tests/test_dist_cmd.py
- tools/chemenu/tests/test_portability.py
- tools/chemenu/tests/test_search.py
- tools/chemenu/tests/test_trace_ingest.py
- tools/chemenu/type_resolver.py
- tools/chemenu/upload.py
- tools/chemenu/version.py
- tools/run_wikitool.py
- tools/trace_ingest.py
2026-10-01 21:15:03 +02:00

125 lines
5.1 KiB
Python

"""Load the corpus once per revision instead of once per query.
`load_pages_by_path()` reads and parses every page under `kb/` on every call.
For a CLI that is the right shape - one call per process, nothing to reuse, and
a cache would only add a way to answer from a tree that has since changed. For
a long-lived reader (the MCP server) it is the opposite: the same corpus is
reparsed for every request, and the cost grows linearly with the corpus.
So the cache is an object a caller holds, not a module-level dict that switches
itself on behind everyone's back. The CLI holds none and behaves exactly as
before; a resident process holds one.
**The key is the commit SHA, and a dirty tree is never cached.** The SHA alone
would be a correctness bug in any checkout someone edits: a session that writes
a page and searches for it would be answered from the parse taken before the
write, with nothing about the SHA having changed. A clean tree is the state the
server actually runs in - it is kept that way by `git fetch && git reset
--hard` - so the fast path is the one that holds there, and every other tree
falls back to reloading.
That same SHA is what a response is stamped with, which is deliberate: the
revision a caller is told about is by construction the revision its answer was
computed from.
"""
from __future__ import annotations
import subprocess
from pathlib import Path
from typing import Optional
from chemenu import config, toolpaths
from chemenu.page import Page
def head_commit(root: Optional[Path] = None) -> Optional[str]:
"""The full SHA of `HEAD`, or None outside a git checkout."""
result = _git(["rev-parse", "HEAD"], root)
if result is None or result.returncode != 0:
return None
return result.stdout.strip() or None
def is_dirty(root: Optional[Path] = None, path: Optional[Path] = None) -> bool:
"""Whether the working tree has uncommitted changes under `path`.
Errs toward dirty: if git cannot answer, the answer is "assume it changed".
A cache that treats "unknown" as clean serves stale pages, which is the one
outcome this module exists to prevent.
"""
root = root or config.ROOT
target = path or config.KB_DIR
try:
relative = Path(target).resolve().relative_to(Path(root).resolve()).as_posix()
except ValueError:
return True
result = _git(["status", "--porcelain", "--", relative], root)
if result is None or result.returncode != 0:
return True
return bool(result.stdout.strip())
def _git(args: list[str], root: Optional[Path] = None):
try:
return subprocess.run(
[toolpaths.git(), *args],
cwd=root or config.ROOT,
capture_output=True,
text=True,
encoding="utf-8",
timeout=10,
check=False,
)
except (OSError, subprocess.SubprocessError):
return None
class CorpusCache:
"""One parsed corpus, reused while the checkout stays on the same commit.
Not thread-safe by itself: a caller serving concurrent requests holds the
lock. Kept out of here because the locking discipline belongs to whoever
owns the request loop, and a lock hidden in a cache is one nobody can see
when they need to reason about it.
"""
def __init__(self, kb_dir: Optional[Path] = None, root: Optional[Path] = None):
self.kb_dir = kb_dir
self.root = root
self._pages: Optional[dict[str, Page]] = None
self._revision: Optional[str] = None
@property
def revision(self) -> Optional[str]:
"""The commit the cached corpus was read at, or None if nothing is
cached (including because the tree was dirty)."""
return self._revision
def current_revision(self) -> Optional[str]:
"""The commit this corpus would be cached under right now: `HEAD` on a
clean tree, None on a dirty one or outside git. None means uncacheable,
which is why it is also what a caller should report as "no revision" -
an answer read out of a dirty tree does not correspond to any commit."""
root = self.root or config.ROOT
if is_dirty(root, self.kb_dir or config.KB_DIR):
return None
return head_commit(root)
def load(self) -> tuple[dict[str, Page], Optional[str]]:
"""(pages, revision). Reparses whenever the revision is not the cached
one, and on every call while the tree is dirty."""
# Imported here rather than at module level: `commands.search` pulls in
# typer, and this module is meant to be importable by a library caller
# that has no CLI. Removing that edge properly is the library-boundary
# work, not this file's job.
from chemenu.commands.search import load_pages_by_path
revision = self.current_revision()
if revision is None or revision != self._revision or self._pages is None:
pages = load_pages_by_path(self.kb_dir, self.root)
if revision is None:
self._pages, self._revision = None, None
return pages, None
self._pages, self._revision = pages, revision
return self._pages, self._revision