feat: MCP-Leseserver, Bibliotheksgrenze, Haertung des Lesepfads, Publish-Remote-Gate scharf (2.4.0)
Files changed: - .gitea/workflows/ci.yml - CHANGES.md - README.md - VERSION - instructions/mcp-read-server.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/api.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/lint.py - tools/chemenu/commands/search.py - tools/chemenu/commands/types_cmd.py - tools/chemenu/config.py - tools/chemenu/corpus_cache.py - tools/chemenu/errors.py - tools/chemenu/frontmatter_io.py - tools/chemenu/lint_core.py - tools/chemenu/mcp/__init__.py - tools/chemenu/mcp/__main__.py - tools/chemenu/mcp/server.py - tools/chemenu/page.py - tools/chemenu/search/filters.py - tools/chemenu/search/registry.py - tools/chemenu/search/ripgrep.py - tools/chemenu/search/service.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_api.py - tools/chemenu/tests/test_corpus_cache.py - tools/chemenu/tests/test_doctor.py - tools/chemenu/tests/test_frontmatter_io.py - tools/chemenu/tests/test_instructions_cmd.py - tools/chemenu/tests/test_mcp_server.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_search.py - tools/chemenu/type_resolver.py - tools/chemenu/types_core.py - tools/requirements-mcp.txt
This commit is contained in:
1 parent
d1cf2e0327
commit
576df2cddd
37 files changed
+3037
-625
No files matched your search
@@ -254,12 +254,19 @@ def check_publish_remotes() -> Check:
|
||||
configured and no allowlist. That is the shape a private instance has after
|
||||
it adds the public upstream, and it is exactly when a wrong `--remote`
|
||||
stops being a typo and starts being a disclosure.
|
||||
|
||||
Both absent states say **armed** or **not armed** rather than only naming
|
||||
the file. AGENTS.md lists this among the three limits enforced in code, so a
|
||||
line that reports the file's absence and leaves the reader to infer what
|
||||
that means about the gate is how a checkout ends up trusting a safeguard
|
||||
that is not running - which is worse than having none.
|
||||
"""
|
||||
urls = git_publish.read_allowed_push_urls()
|
||||
if urls is not None:
|
||||
return Check(
|
||||
"publish-remotes", "OK",
|
||||
f"{len(urls)} allowed push target(s) in {config.PUBLISH_REMOTES_FILENAME}",
|
||||
f"Gate armed: {len(urls)} allowed push target(s) in "
|
||||
f"{config.PUBLISH_REMOTES_FILENAME}",
|
||||
)
|
||||
result = subprocess.run(
|
||||
["git", "remote"], cwd=config.ROOT, capture_output=True, text=True
|
||||
@@ -268,13 +275,15 @@ def check_publish_remotes() -> Check:
|
||||
if len(remotes) > 1:
|
||||
return Check(
|
||||
"publish-remotes", "WARN",
|
||||
f"{len(remotes)} remotes ({', '.join(remotes)}) and no publish allowlist",
|
||||
f"Gate not armed: {len(remotes)} remotes ({', '.join(remotes)}) and no "
|
||||
f"{config.PUBLISH_REMOTES_FILENAME} - every one of them is a legal publish target",
|
||||
f"Create {config.PUBLISH_REMOTES_FILENAME} naming the push URL this checkout "
|
||||
"may publish to - see instructions/gates.md",
|
||||
)
|
||||
return Check(
|
||||
"publish-remotes", "OK",
|
||||
f"No {config.PUBLISH_REMOTES_FILENAME} (unrestricted; one remote configured)",
|
||||
f"Gate not armed: no {config.PUBLISH_REMOTES_FILENAME} - any push target passes "
|
||||
"(1 remote configured, nothing to confuse it with)",
|
||||
)
|
||||
|
||||
|
||||
|
||||
+28
-416
@@ -1,16 +1,12 @@
|
||||
"""Deterministic structural health checks for the wiki.
|
||||
"""`wikitool lint` - the terminal adapter over `chemenu.lint_core`.
|
||||
|
||||
This intentionally covers only what can be computed mechanically: broken
|
||||
wikilinks, orphan pages, index/page drift, frontmatter schema gaps, and
|
||||
filename/title mismatches. Semantic judgment (contradictions, staleness,
|
||||
what's worth writing about next) stays with the LLM - this report gives it a
|
||||
verified factual foundation instead of requiring it to re-derive these facts
|
||||
by reading every page.
|
||||
The checks, the report and the hard-error rule live in `chemenu/lint_core.py`,
|
||||
which imports no CLI machinery. This module owns only what a terminal needs:
|
||||
the flags, where the report file lands, and the exit code.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
@@ -18,416 +14,32 @@ import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import rel_path, success
|
||||
from chemenu.frontmatter_io import frontmatter_error
|
||||
from chemenu.markdown_code import strip_code_spans
|
||||
from chemenu.provenance import broken_raw_refs as find_broken_raw_refs
|
||||
from chemenu.provenance import duplicate_raw_file_owners as find_duplicate_raw_file_owners
|
||||
from chemenu.provenance import extract_inline_cites
|
||||
from chemenu.provenance import legacy_citation_markers as find_legacy_citation_markers
|
||||
from chemenu.provenance import legacy_source_pages as find_legacy_source_pages
|
||||
from chemenu.provenance import orphan_footnote_defs as find_orphan_footnote_defs
|
||||
from chemenu.provenance import uncovered_raw_files as find_uncovered_raw_files
|
||||
from chemenu.provenance import undefined_footnote_refs as find_undefined_footnote_refs
|
||||
from chemenu.kb_scan import (
|
||||
GENERATED_INDEX,
|
||||
WIKILINK_RE,
|
||||
build_link_graph,
|
||||
find_duplicate_title_paths,
|
||||
inbound_links,
|
||||
load_kb_pages,
|
||||
)
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
# Style guide's one mechanically-checkable rule (hard oracle: a plain count).
|
||||
# The rest of the style guide (tone, AI-phrase avoidance) is a soft/proxy judgment
|
||||
# and stays with the LLM - see wiki-manage/wiki-ingest skill guidance, not lint.
|
||||
#
|
||||
# The unit is a quote, not a `>` line. It used to be the line, which measured
|
||||
# the wrap width the rule has no opinion about: one quotation written long
|
||||
# counted 1 and the same quotation wrapped at 100 columns counted 4. An author
|
||||
# who took the finding seriously made the page harder to read to quiet it.
|
||||
QUOTE_LIMIT = 2
|
||||
|
||||
# How many hub pages `most_linked` reports. Purely informational (wiki-status
|
||||
# surfaces it); not a finding, so the cutoff only bounds report size.
|
||||
MOST_LINKED_COUNT = 10
|
||||
|
||||
|
||||
def count_quote_blocks(body: str) -> int:
|
||||
"""How many distinct blockquotes `body` carries.
|
||||
|
||||
A run of consecutive `>` lines is one quote; a blank line or any
|
||||
non-quoted line ends it. Code is masked out first, so a `>` inside a
|
||||
fenced shell transcript is a prompt, not a quotation.
|
||||
|
||||
Lazy continuation - a quote whose wrapped lines drop the `>` - reads here
|
||||
as two quotes rather than one. That over-counts in the direction the limit
|
||||
already errs on, and the corpus prefixes every line, so the alternative
|
||||
(tracking paragraph state) buys nothing.
|
||||
"""
|
||||
count, in_quote = 0, False
|
||||
for line in strip_code_spans(body).splitlines():
|
||||
is_quote = line.lstrip().startswith(">")
|
||||
if is_quote and not in_quote:
|
||||
count += 1
|
||||
in_quote = is_quote
|
||||
return count
|
||||
|
||||
|
||||
def run_lint(kb_dir: Path) -> dict:
|
||||
pages = load_kb_pages(kb_dir)
|
||||
duplicate_titles = find_duplicate_title_paths(kb_dir, config.ROOT)
|
||||
|
||||
# Pages whose frontmatter can't be parsed read back as `{}` everywhere
|
||||
# else, which would let them slip past every frontmatter-driven check
|
||||
# below with no finding at all - so they are detected explicitly.
|
||||
frontmatter_errors = []
|
||||
for title, page in sorted(pages.items()):
|
||||
reason = frontmatter_error(page.path)
|
||||
if reason is None and not page.frontmatter.get("type"):
|
||||
reason = "missing `type:` field"
|
||||
if reason is not None:
|
||||
frontmatter_errors.append({"page": title, "error": reason})
|
||||
|
||||
graph = build_link_graph(pages)
|
||||
broken_links = [
|
||||
{"page": title, "target": target}
|
||||
for title, targets in graph.items()
|
||||
for target in sorted(targets)
|
||||
if target not in pages
|
||||
]
|
||||
|
||||
inbound = inbound_links({t: v for t, v in graph.items() if t != "index"})
|
||||
orphan_pages = sorted(
|
||||
title
|
||||
for title, sources in inbound.items()
|
||||
if not sources
|
||||
and title not in ("index", "log")
|
||||
# comparison pages are not linked to by design; index.md is sufficient coverage
|
||||
and pages[title].kind != "comparison"
|
||||
)
|
||||
|
||||
# Same link graph, opposite end: the most-linked-to pages are the wiki's
|
||||
# hubs. Reported (not judged) so `wiki-status` can show them without
|
||||
# re-deriving the graph.
|
||||
inbound_counts = {title: len(sources) for title, sources in inbound.items()}
|
||||
most_linked = [
|
||||
{"page": title, "inbound": count}
|
||||
for title, count in sorted(inbound_counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
||||
if count > 0
|
||||
][:MOST_LINKED_COUNT]
|
||||
|
||||
# The catalog is sharded: `kb/index.md` is a map carrying counts and links,
|
||||
# and the page rows live in a generated INDEX.md per collection/area. Both
|
||||
# halves have to be read, or every page reads as missing from the index.
|
||||
index_text = "".join(
|
||||
path.read_text(encoding="utf-8")
|
||||
for path in [kb_dir / "index.md", *sorted(kb_dir.rglob(GENERATED_INDEX))]
|
||||
if path.exists()
|
||||
)
|
||||
index_links = {m.group(1).strip() for m in WIKILINK_RE.finditer(index_text)}
|
||||
missing_from_index = sorted(set(pages) - index_links - {"index", "log"})
|
||||
dangling_index_entries = sorted(index_links - set(pages))
|
||||
|
||||
title_mismatches = []
|
||||
for title, page in sorted(pages.items()):
|
||||
if page.kind not in ("entity", "concept"):
|
||||
continue
|
||||
h1 = page.h1_title
|
||||
if h1 is not None and h1 != title:
|
||||
title_mismatches.append({"page": title, "h1": h1})
|
||||
|
||||
unmarked_provenance = []
|
||||
for title, page in sorted(pages.items()):
|
||||
if page.kind not in ("entity", "concept"):
|
||||
continue
|
||||
sources_list = page.frontmatter.get("sources") or []
|
||||
if not sources_list and page.frontmatter.get("provenance") != "general":
|
||||
unmarked_provenance.append(title)
|
||||
|
||||
citation_frontmatter_drift = []
|
||||
for title, page in sorted(pages.items()):
|
||||
sources_list = set(page.frontmatter.get("sources") or [])
|
||||
cited = {cited_title for cited_title, _file in extract_inline_cites(page.body)}
|
||||
cited.discard(title) # a source page citing itself for a specific file within it is not drift
|
||||
for missing_source in sorted(cited - sources_list):
|
||||
citation_frontmatter_drift.append({"page": title, "cited_but_not_in_sources": missing_source})
|
||||
|
||||
legacy_citation_markers = find_legacy_citation_markers(pages)
|
||||
undefined_footnote_refs = find_undefined_footnote_refs(pages)
|
||||
orphan_footnote_defs = find_orphan_footnote_defs(pages)
|
||||
|
||||
# The frontmatter half of the link graph. `broken_links` above only walks
|
||||
# `[[wikilinks]]` in page *bodies*, so a `related:`/`sources:`/`entities:`
|
||||
# entry naming a page that does not exist - a rename that was not
|
||||
# propagated, a deleted page, or a URL pasted where a title belongs - used
|
||||
# to pass every check. Which fields hold page titles is declared by each
|
||||
# type-spec's `page_ref_fields:`, not hardcoded here.
|
||||
dangling_frontmatter_refs = []
|
||||
for title, page in sorted(pages.items()):
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
continue
|
||||
try:
|
||||
ref_fields = resolver.get_page_ref_fields(type_path, page.path)
|
||||
except ValueError:
|
||||
continue # unresolvable type is already reported as type_resolution_errors
|
||||
for field in ref_fields:
|
||||
for target in page.frontmatter.get(field) or []:
|
||||
if target not in pages:
|
||||
dangling_frontmatter_refs.append(
|
||||
{"page": title, "field": field, "target": target}
|
||||
)
|
||||
|
||||
quote_limit_violations = []
|
||||
for title, page in sorted(pages.items()):
|
||||
quote_count = count_quote_blocks(page.body)
|
||||
if quote_count > QUOTE_LIMIT:
|
||||
quote_limit_violations.append({"page": title, "quote_count": quote_count})
|
||||
|
||||
# Type system validation. Lint reports are not validated here: they are
|
||||
# written to `reports/` outside kb/ and are never pages, so nothing this
|
||||
# loop scans can be one.
|
||||
invalid_type_paths = []
|
||||
type_resolution_errors = []
|
||||
schema_validation_errors = []
|
||||
|
||||
for title, page in sorted(pages.items()):
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
continue
|
||||
|
||||
# Check if type path is valid
|
||||
if not type_path.endswith('.md'):
|
||||
invalid_type_paths.append({"page": title, "type": type_path, "error": "Type path must end with .md"})
|
||||
continue
|
||||
|
||||
# Try to resolve and validate the type
|
||||
try:
|
||||
resolver.load_type_spec(type_path, page.path)
|
||||
|
||||
# Try schema validation
|
||||
try:
|
||||
resolver.validate_frontmatter(page.frontmatter, type_path, page.path)
|
||||
except ValueError as schema_error:
|
||||
schema_validation_errors.append({"page": title, "type": type_path, "error": str(schema_error)})
|
||||
|
||||
except ValueError as resolution_error:
|
||||
type_resolution_errors.append({"page": title, "type": type_path, "error": str(resolution_error)})
|
||||
|
||||
return {
|
||||
"generated": date.today().isoformat(),
|
||||
"page_count": len(pages),
|
||||
"frontmatter_errors": frontmatter_errors,
|
||||
"broken_links": broken_links,
|
||||
"orphan_pages": orphan_pages,
|
||||
"most_linked": most_linked,
|
||||
"inbound_counts": inbound_counts,
|
||||
"missing_from_index": missing_from_index,
|
||||
"dangling_index_entries": dangling_index_entries,
|
||||
"title_mismatches": title_mismatches,
|
||||
"duplicate_titles": duplicate_titles,
|
||||
"uncovered_raw_files": find_uncovered_raw_files(config.RAW_DIR, pages),
|
||||
"broken_raw_refs": find_broken_raw_refs(pages),
|
||||
"duplicate_raw_file_owners": find_duplicate_raw_file_owners(pages),
|
||||
"legacy_source_pages": find_legacy_source_pages(pages),
|
||||
"unmarked_provenance": unmarked_provenance,
|
||||
"citation_frontmatter_drift": citation_frontmatter_drift,
|
||||
"legacy_citation_markers": legacy_citation_markers,
|
||||
"undefined_footnote_refs": undefined_footnote_refs,
|
||||
"orphan_footnote_defs": orphan_footnote_defs,
|
||||
"dangling_frontmatter_refs": dangling_frontmatter_refs,
|
||||
"quote_limit_violations": quote_limit_violations,
|
||||
"invalid_type_paths": invalid_type_paths,
|
||||
"type_resolution_errors": type_resolution_errors,
|
||||
"schema_validation_errors": schema_validation_errors,
|
||||
}
|
||||
|
||||
|
||||
def _section(lines: list[str], title: str, items: list, formatter) -> None:
|
||||
lines.append(f"## {title}")
|
||||
lines.append("")
|
||||
if not items:
|
||||
lines.append("None found.")
|
||||
else:
|
||||
for item in items:
|
||||
lines.append(f"- {formatter(item)}")
|
||||
lines.append("")
|
||||
|
||||
|
||||
def render_markdown(report: dict) -> str:
|
||||
lines = [f"# Structural Lint Report ({report['generated']})", ""]
|
||||
lines.append(f"Scanned {report['page_count']} pages under `wiki/`. This report covers only")
|
||||
lines.append("mechanically-verifiable structural issues; see the Semantic Review section")
|
||||
lines.append("below for judgment calls the LLM should complete.")
|
||||
lines.append("")
|
||||
|
||||
_section(
|
||||
lines, "Unreadable Frontmatter", report["frontmatter_errors"],
|
||||
lambda i: f"[[{i['page']}]] - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Broken Wikilinks", report["broken_links"],
|
||||
lambda i: f"[[{i['page']}]] links to missing [[{i['target']}]]",
|
||||
)
|
||||
_section(lines, "Orphan Pages (no inbound links)", report["orphan_pages"], lambda i: f"[[{i}]]")
|
||||
_section(
|
||||
lines, f"Most-Linked Pages (top {MOST_LINKED_COUNT} hubs)", report["most_linked"],
|
||||
lambda i: f"[[{i['page']}]] - {i['inbound']} inbound link(s)",
|
||||
)
|
||||
_section(lines, "Pages Missing from index.md", report["missing_from_index"], lambda i: f"[[{i}]]")
|
||||
_section(lines, "Dangling index.md Entries", report["dangling_index_entries"], lambda i: f"[[{i}]]")
|
||||
_section(
|
||||
lines, "Duplicate Titles (naming collisions)", report["duplicate_titles"],
|
||||
lambda i: f"`{i['stem']}` -> {', '.join(f'`{p}`' for p in i['paths'])}",
|
||||
)
|
||||
_section(
|
||||
lines, "Filename / H1 Title Mismatches", report["title_mismatches"],
|
||||
lambda i: f"[[{i['page']}]] H1 is '{i['h1']}'",
|
||||
)
|
||||
_section(
|
||||
lines, "Uncovered Raw Files (no source page)", report["uncovered_raw_files"],
|
||||
lambda i: f"`{i}`",
|
||||
)
|
||||
_section(
|
||||
lines, "Broken raw_files References", report["broken_raw_refs"],
|
||||
lambda i: f"[[{i['page']}]] -> `{i['raw_path']}` (does not exist)",
|
||||
)
|
||||
_section(
|
||||
lines, "Raw Files With More Than One Owner", report["duplicate_raw_file_owners"],
|
||||
lambda i: f"`{i['raw_file']}` is claimed by " + ", ".join(f"[[{t}]]" for t in i["owners"]),
|
||||
)
|
||||
_section(
|
||||
lines, "Legacy source: Field (not yet migrated to raw_files:)", report["legacy_source_pages"],
|
||||
lambda i: f"[[{i['page']}]] source: `{i['source']}` ({i['reason']})",
|
||||
)
|
||||
_section(
|
||||
lines, "Pages Missing provenance: general Marker", report["unmarked_provenance"],
|
||||
lambda i: f"[[{i}]] has no sources and is not marked `provenance: general`",
|
||||
)
|
||||
_section(
|
||||
lines, "Citation / Frontmatter Drift", report["citation_frontmatter_drift"],
|
||||
lambda i: f"[[{i['page']}]] cites [[{i['cited_but_not_in_sources']}]] inline but it is missing from frontmatter `sources:`",
|
||||
)
|
||||
_section(
|
||||
lines, "Legacy Citation Markers (pre-migration `^[[...]]`)", report["legacy_citation_markers"],
|
||||
lambda i: f"[[{i['page']}]] still has `{i['marker']}` - run `wikitool cite add` and replace it with the `[^cite-id]` it prints",
|
||||
)
|
||||
_section(
|
||||
lines, "Undefined Footnote References", report["undefined_footnote_refs"],
|
||||
lambda i: f"[[{i['page']}]] references `[^{i['ref']}]`, which has no `[^{i['ref']}]: [[...]]` definition",
|
||||
)
|
||||
_section(
|
||||
lines, "Orphan Footnote Definitions", report["orphan_footnote_defs"],
|
||||
lambda i: f"[[{i['page']}]] defines `[^{i['id']}]` (-> [[{i['source']}]]) but nothing references it - run `wikitool cite sync`",
|
||||
)
|
||||
_section(
|
||||
lines, "Dangling Frontmatter References", report["dangling_frontmatter_refs"],
|
||||
lambda i: f"[[{i['page']}]] `{i['field']}:` names `{i['target']}`, which is not a page",
|
||||
)
|
||||
_section(
|
||||
lines, "Invalid Type Paths", report["invalid_type_paths"],
|
||||
lambda i: f"[[{i['page']}]] has type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Type Resolution Errors", report["type_resolution_errors"],
|
||||
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, "Schema Validation Errors", report["schema_validation_errors"],
|
||||
lambda i: f"[[{i['page']}]] type: `{i['type']}` - {i['error']}",
|
||||
)
|
||||
_section(
|
||||
lines, f"Pages Exceeding Quote Limit (>{QUOTE_LIMIT}/page)", report["quote_limit_violations"],
|
||||
lambda i: f"[[{i['page']}]] has {i['quote_count']} quotes - trim or confirm they're load-bearing",
|
||||
)
|
||||
|
||||
lines.append("## Semantic Review (LLM to complete)")
|
||||
lines.append("")
|
||||
lines.append("- Contradictions across pages: TODO")
|
||||
lines.append("- Stale claims (unconfirmed >6 months): TODO")
|
||||
lines.append("- Suggested new pages / missing cross-references: TODO")
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# Sections that always carry content but are not findings, so the summary
|
||||
# handles them separately: a hub list is a statistic, and the semantic review
|
||||
# is the checklist that follows the report rather than part of it.
|
||||
INFORMATIONAL_SECTIONS = ("Most-Linked Pages",)
|
||||
SEMANTIC_REVIEW_SECTION = "Semantic Review"
|
||||
|
||||
|
||||
def _split_sections(markdown: str) -> tuple[str, list[tuple[str, str]]]:
|
||||
"""Cut a rendered report into its preamble and (title, body) sections."""
|
||||
preamble, *rest = markdown.split("\n## ")
|
||||
sections = []
|
||||
for part in rest:
|
||||
title, _, body = part.partition("\n")
|
||||
sections.append((title.strip(), body.strip()))
|
||||
return preamble.rstrip(), sections
|
||||
|
||||
|
||||
def render_summary(report: dict) -> str:
|
||||
"""The same report with the empty sections removed.
|
||||
|
||||
On a healthy corpus the full report is better than 90% "None found.", so
|
||||
reading it in the terminal means paging past the answer. The file on disk
|
||||
stays complete - this is what gets printed, and the written path underneath
|
||||
it is how the rest is reached without running lint a second time.
|
||||
"""
|
||||
preamble, sections = _split_sections(render_markdown(report))
|
||||
findings, trailing = [], []
|
||||
for title, body in sections:
|
||||
if title.startswith(SEMANTIC_REVIEW_SECTION):
|
||||
trailing.append((title, body))
|
||||
elif body != "None found." and not title.startswith(INFORMATIONAL_SECTIONS):
|
||||
findings.append((title, body))
|
||||
lines = [preamble, ""]
|
||||
if not findings:
|
||||
lines += ["No structural findings.", ""]
|
||||
for title, body in findings + trailing:
|
||||
lines += [f"## {title}", "", body, ""]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def default_report_path(report: dict) -> Path:
|
||||
"""Where a report goes when the caller names no path.
|
||||
|
||||
`reports/` is derived and gitignored ([reports/CONTRACT.md]), so writing
|
||||
here by default costs the tree nothing.
|
||||
"""
|
||||
return config.ROOT / "reports" / f"Lint Report {report['generated']}.md"
|
||||
|
||||
|
||||
# Findings that make a tree structurally wrong rather than merely untidy.
|
||||
# `orphan_pages` is deliberately absent: many pages are validly reachable
|
||||
# through the index or navigation only. `quote_limit_violations` is advisory
|
||||
# too - it flags a habit, not a broken tree.
|
||||
#
|
||||
# One definition, used by `lint --fail-on-error` and by the eval scorecard: if
|
||||
# the two disagreed, a run could pass its score while lint refused it.
|
||||
HARD_ERROR_KEYS = (
|
||||
"frontmatter_errors",
|
||||
"broken_links",
|
||||
"dangling_index_entries",
|
||||
"duplicate_titles",
|
||||
"broken_raw_refs",
|
||||
"duplicate_raw_file_owners",
|
||||
"legacy_source_pages",
|
||||
"citation_frontmatter_drift",
|
||||
"legacy_citation_markers",
|
||||
"undefined_footnote_refs",
|
||||
"orphan_footnote_defs",
|
||||
"dangling_frontmatter_refs",
|
||||
"invalid_type_paths",
|
||||
"type_resolution_errors",
|
||||
"schema_validation_errors",
|
||||
from chemenu.lint_core import (
|
||||
HARD_ERROR_KEYS,
|
||||
MOST_LINKED_COUNT,
|
||||
QUOTE_LIMIT,
|
||||
count_quote_blocks,
|
||||
default_report_path,
|
||||
has_hard_errors,
|
||||
render_markdown,
|
||||
render_summary,
|
||||
run_lint,
|
||||
)
|
||||
|
||||
|
||||
def has_hard_errors(report: dict) -> bool:
|
||||
return any(report.get(key) for key in HARD_ERROR_KEYS)
|
||||
# Re-exported: `from chemenu.commands.lint import run_lint` still resolves, and
|
||||
# so does every other name the tests and sibling commands already import.
|
||||
__all__ = [
|
||||
"HARD_ERROR_KEYS",
|
||||
"MOST_LINKED_COUNT",
|
||||
"QUOTE_LIMIT",
|
||||
"count_quote_blocks",
|
||||
"default_report_path",
|
||||
"has_hard_errors",
|
||||
"render_markdown",
|
||||
"render_summary",
|
||||
"run_lint",
|
||||
"lint_command",
|
||||
]
|
||||
|
||||
|
||||
def lint_command(
|
||||
|
||||
@@ -21,90 +21,38 @@ Scope is `kb/` only. `instructions/` is discovered through
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.commands._util import fail, today_iso
|
||||
from chemenu.frontmatter_io import read_page
|
||||
from chemenu.kb_scan import iter_kb_pages
|
||||
from chemenu.page import Page
|
||||
from chemenu.search import filters
|
||||
from chemenu.search.base import page_key
|
||||
from chemenu.search.filters import PredicateError
|
||||
from chemenu.search.fuse import reciprocal_rank_fusion
|
||||
from chemenu.search.registry import UnknownBackend, resolve
|
||||
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing, build_hit
|
||||
from chemenu.search.ripgrep import RipgrepFailed, RipgrepMissing
|
||||
from chemenu.search.service import (
|
||||
load_pages_by_path,
|
||||
run_search,
|
||||
sort_hits,
|
||||
unreadable_pages,
|
||||
)
|
||||
from chemenu.search.types import Predicate, SearchHit, SearchQuery
|
||||
|
||||
# Re-exported so `from chemenu.commands.search import run_search` keeps
|
||||
# resolving. The core lives in `chemenu/search/service.py`, which imports no
|
||||
# CLI machinery; this module is the terminal adapter over it.
|
||||
__all__ = [
|
||||
"load_pages_by_path",
|
||||
"run_search",
|
||||
"sort_hits",
|
||||
"unreadable_pages",
|
||||
"render_table",
|
||||
"search_command",
|
||||
]
|
||||
|
||||
TITLE_WIDTH = 34
|
||||
SUMMARY_WIDTH = 84
|
||||
|
||||
|
||||
def load_pages_by_path(kb_dir: Path | None = None, root: Path | None = None) -> dict[str, Page]:
|
||||
"""Every page under `kb/`, keyed by repo-relative path.
|
||||
|
||||
Path-keyed rather than title-keyed on purpose: `load_kb_pages()` drops one
|
||||
of two pages sharing a stem, and search should still find both - a
|
||||
duplicate title is a lint finding, not a reason to hide a page.
|
||||
"""
|
||||
kb_dir = kb_dir or config.KB_DIR
|
||||
root = root or config.ROOT
|
||||
pages: dict[str, Page] = {}
|
||||
for path in iter_kb_pages(kb_dir):
|
||||
frontmatter, body = read_page(path)
|
||||
pages[page_key(path, root)] = Page(path=path, frontmatter=frontmatter, body=body)
|
||||
return pages
|
||||
|
||||
|
||||
def _sort_key(hit: SearchHit, field: str):
|
||||
value = hit.as_dict().get(field)
|
||||
if value is None:
|
||||
# Missing values sort last in either direction rather than crashing on
|
||||
# a None comparison.
|
||||
return (1, "")
|
||||
if isinstance(value, (int, float)):
|
||||
return (0, value)
|
||||
return (0, str(value).lower())
|
||||
|
||||
|
||||
def sort_hits(hits: list[SearchHit], sort: str | None) -> list[SearchHit]:
|
||||
"""Sort by a hit field. A leading `-` reverses, e.g. `--sort -confidence`."""
|
||||
if not sort:
|
||||
return hits
|
||||
descending = sort.startswith("-")
|
||||
field = sort.lstrip("-")
|
||||
ordered = sorted(hits, key=lambda h: _sort_key(h, field), reverse=descending)
|
||||
return ordered
|
||||
|
||||
|
||||
def run_search(
|
||||
query: SearchQuery,
|
||||
pages: dict[str, Page],
|
||||
backends: list,
|
||||
kb_dir: Path | None = None,
|
||||
) -> list[SearchHit]:
|
||||
"""Answer a query. Pure: no I/O beyond whatever a backend does."""
|
||||
filters.validate_fields(query.predicates, pages)
|
||||
|
||||
if query.text:
|
||||
rankings = [backend.search(query, pages) for backend in backends]
|
||||
hits = rankings[0] if len(rankings) == 1 else reciprocal_rank_fusion(rankings)
|
||||
allowed = filters.apply_predicates(pages, query.predicates, kb_dir)
|
||||
hits = [hit for hit in hits if hit.path in allowed]
|
||||
else:
|
||||
selected = filters.apply_predicates(pages, query.predicates, kb_dir)
|
||||
hits = [
|
||||
build_hit(page, key, [], query, backend="frontmatter", kb_dir=kb_dir)
|
||||
for key, page in selected.items()
|
||||
]
|
||||
hits.sort(key=lambda h: h.title.lower())
|
||||
|
||||
hits = sort_hits(hits, query.sort)
|
||||
return hits[: query.limit] if query.limit else hits
|
||||
|
||||
|
||||
def _truncate(text: str, width: int) -> str:
|
||||
text = " ".join(text.split())
|
||||
return text if len(text) <= width else text[: width - 1] + "\u2026"
|
||||
@@ -206,6 +154,8 @@ def search_command(
|
||||
except RipgrepFailed as exc:
|
||||
fail(str(exc))
|
||||
|
||||
unreadable = unreadable_pages(pages)
|
||||
|
||||
if json_out:
|
||||
payload = {
|
||||
"generated": today_iso(),
|
||||
@@ -214,8 +164,17 @@ def search_command(
|
||||
"backend": ",".join(b.name for b in backends),
|
||||
"count": len(hits),
|
||||
"results": [hit.as_dict() for hit in hits],
|
||||
# Always present, usually empty. A caller that has to look for the
|
||||
# key to learn whether it should worry will not look.
|
||||
"unreadable": unreadable,
|
||||
}
|
||||
typer.echo(json.dumps(payload, indent=2))
|
||||
return
|
||||
|
||||
typer.echo(render_table(hits, show_matches))
|
||||
for entry in unreadable:
|
||||
typer.echo(
|
||||
f"WARN unreadable frontmatter: {entry['path']} ({entry['reason']}) - "
|
||||
"this page cannot match any --field predicate",
|
||||
err=True,
|
||||
)
|
||||
@@ -7,35 +7,29 @@ into context on every skill invocation. A type-spec's own frontmatter
|
||||
declared `.schema.yaml` are the single source of truth; this command only
|
||||
formats what `TypeResolver` already resolves - it does not duplicate or
|
||||
re-derive any type knowledge.
|
||||
|
||||
The resolving half lives in `chemenu/types_core.py`, which imports no CLI
|
||||
machinery. This module is the terminal adapter over it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Dict
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu.commands._util import fail
|
||||
from chemenu.type_resolver import resolver
|
||||
from chemenu.types_core import UnknownType, describe_type, list_types
|
||||
|
||||
app = typer.Typer(help="Discover and describe Chemenu type-spec contracts.")
|
||||
|
||||
|
||||
@app.command("list")
|
||||
def list_types(json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")):
|
||||
def list_types_command(
|
||||
json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON")
|
||||
):
|
||||
"""List every type-spec under types/, with its name, schema, subtype
|
||||
field (if any), base directory, and description."""
|
||||
rows: list[Dict[str, Any]] = []
|
||||
for type_path, frontmatter in resolver.list_type_specs():
|
||||
rows.append({
|
||||
"name": frontmatter.get("name"),
|
||||
"type_path": type_path,
|
||||
"schema": frontmatter.get("schema"),
|
||||
"subtype_field": frontmatter.get("subtype_field"),
|
||||
"root": frontmatter.get("root") or "kb",
|
||||
"base_dir": frontmatter.get("base_dir"),
|
||||
"description": frontmatter.get("description"),
|
||||
})
|
||||
rows = list_types()
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(rows, indent=2))
|
||||
@@ -53,7 +47,7 @@ def list_types(json_out: bool = typer.Option(False, "--json", help="Print raw fi
|
||||
|
||||
|
||||
@app.command("describe")
|
||||
def describe_type(
|
||||
def describe_type_command(
|
||||
name: str = typer.Argument(..., help="Type name, e.g. 'entity' (see `types list`)"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Print raw findings as JSON"),
|
||||
):
|
||||
@@ -61,51 +55,26 @@ def describe_type(
|
||||
with enums where declared), its subtype field if any, and its authoring
|
||||
body - the same information an LLM would otherwise gather by reading the
|
||||
raw type-spec and `.schema.yaml` files directly."""
|
||||
type_path = resolver.find_type_by_name(name)
|
||||
if type_path is None:
|
||||
available = sorted(fm.get("name") for _, fm in resolver.list_type_specs())
|
||||
fail(f"No type-spec named '{name}'. Available: {', '.join(available)}")
|
||||
return # unreachable; keeps type-checkers happy about `type_path` below
|
||||
|
||||
type_spec = resolver.load_type_spec(type_path)
|
||||
frontmatter = type_spec["frontmatter"]
|
||||
body = type_spec["body"]
|
||||
schema = resolver.get_schema(type_path)
|
||||
|
||||
fields: list[Dict[str, Any]] = []
|
||||
if schema is not None:
|
||||
required = set(schema.get("required", []))
|
||||
for field_name, field_schema in schema.get("properties", {}).items():
|
||||
fields.append({
|
||||
"field": field_name,
|
||||
"required": field_name in required,
|
||||
"type": field_schema.get("type"),
|
||||
"enum": field_schema.get("enum"),
|
||||
})
|
||||
try:
|
||||
described = describe_type(name)
|
||||
except UnknownType as exc:
|
||||
fail(str(exc))
|
||||
return # unreachable; keeps type-checkers happy about `described` below
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps({
|
||||
"name": frontmatter.get("name"),
|
||||
"type_path": type_path,
|
||||
"description": frontmatter.get("description"),
|
||||
"schema": frontmatter.get("schema"),
|
||||
"subtype_field": frontmatter.get("subtype_field"),
|
||||
"base_dir": frontmatter.get("base_dir"),
|
||||
"title_prefix": frontmatter.get("title_prefix"),
|
||||
"fields": fields,
|
||||
"body": body.strip(),
|
||||
}, indent=2))
|
||||
typer.echo(json.dumps(described, indent=2))
|
||||
return
|
||||
|
||||
typer.echo(f"# {frontmatter.get('name')} ({type_path})")
|
||||
typer.echo(frontmatter.get("description", ""))
|
||||
fields = described["fields"]
|
||||
typer.echo(f"# {described['name']} ({described['type_path']})")
|
||||
typer.echo(described["description"] or "")
|
||||
typer.echo("")
|
||||
if frontmatter.get("subtype_field"):
|
||||
typer.echo(f"subtype_field: {frontmatter['subtype_field']}")
|
||||
if frontmatter.get("base_dir"):
|
||||
typer.echo(f"base_dir: {frontmatter.get('root') or 'kb'}/{frontmatter['base_dir']}")
|
||||
if frontmatter.get("title_prefix"):
|
||||
typer.echo(f"title_prefix: {frontmatter['title_prefix']!r}")
|
||||
if described["subtype_field"]:
|
||||
typer.echo(f"subtype_field: {described['subtype_field']}")
|
||||
if described["base_dir"]:
|
||||
typer.echo(f"base_dir: {described['root']}/{described['base_dir']}")
|
||||
if described["title_prefix"]:
|
||||
typer.echo(f"title_prefix: {described['title_prefix']!r}")
|
||||
typer.echo("")
|
||||
|
||||
if not fields:
|
||||
@@ -119,4 +88,4 @@ def describe_type(
|
||||
typer.echo("")
|
||||
|
||||
typer.echo("## Authoring guidance")
|
||||
typer.echo(body.strip())
|
||||
typer.echo(described["body"])
|
||||
Reference in new issue
Block a user